commit 2da702073bcdf8dd3e2add17ca5d05c52c16cde3
parent fb8c3674cae5a4a0e4f14a74e683b6192bb5b20c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 03:32:37 -0400
common: report converters — /sweep markdown, /ask answers and report-to-video manifests into report.json, and a report into a starter manifest; archilyzer reports convert / to-manifest
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
7 files changed, 1711 insertions(+), 0 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -162,6 +162,50 @@ export const COMMANDS: Command[] = [
},
},
{
+ path: ["reports", "convert"],
+ usage:
+ "<sweep|ask|manifest> <in> --out <report.json> [--channels-dir <dir>] [--id <id>] [--title <title>] a /sweep report (markdown), an /ask answer or a report-to-video manifest as a report.json, written only when it validates (--channels-dir: widen spans from the cues, find posts' channels)",
+ flags: { out: "string", "channels-dir": "string", id: "string", title: "string" },
+ maxPositionals: 2,
+ run: async ({ positionals, flags }) => {
+ const [from, input] = positionals;
+ const { convertMain, isConvertFrom, CONVERT_FROM } = await import("./reports-convert");
+ if (!isConvertFrom(from) || !input || typeof flags.out !== "string") {
+ console.error(`reports convert: give <${CONVERT_FROM.join("|")}> <in> --out <report.json>`);
+ return 2;
+ }
+ return convertMain({
+ from,
+ input,
+ out: flags.out,
+ ...(typeof flags["channels-dir"] === "string" ? { channelsDir: flags["channels-dir"] } : {}),
+ ...(typeof flags.id === "string" ? { id: flags.id } : {}),
+ ...(typeof flags.title === "string" ? { title: flags.title } : {}),
+ });
+ },
+ },
+ {
+ path: ["reports", "to-manifest"],
+ usage:
+ "<report.json> --out <manifest.json> [--channels-dir <dir>] [--site-origin <url>] a starter report-to-video manifest from a report (a chapter card per section, a claim's still and clips stamped with its verdict, its posts)",
+ flags: { out: "string", "channels-dir": "string", "site-origin": "string" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags }) => {
+ const [input] = positionals;
+ if (!input || typeof flags.out !== "string") {
+ console.error("reports to-manifest: give <report.json> --out <manifest.json>");
+ return 2;
+ }
+ const { toManifestMain } = await import("./reports-convert");
+ return toManifestMain({
+ input,
+ out: flags.out,
+ ...(typeof flags["channels-dir"] === "string" ? { channelsDir: flags["channels-dir"] } : {}),
+ ...(typeof flags["site-origin"] === "string" ? { siteOrigin: flags["site-origin"] } : {}),
+ });
+ },
+ },
+ {
path: ["source", "publish"],
usage:
"[--force] [--check] [--keep-scratch] the scrubbed git mirror, raw tree, history pages (stagit, when installed) and tarball into homepage/public, behind the denied-literal gate (--check: audit and count, write nothing)",
diff --git a/common/bin/reports-convert.ts b/common/bin/reports-convert.ts
@@ -0,0 +1,140 @@
+// `archilyzer reports convert <sweep|ask|manifest> <in> --out <report.json>`
+// and `archilyzer reports to-manifest <report.json> --out <manifest.json>` —
+// the converters (lib/report/convert*.ts) over files.
+//
+// `--channels-dir <dir>` (a `transcripts/channels` tree) lets a conversion
+// read what the documents do not carry: a cited record's cues, so a /sweep or
+// /ask citation's one second widens to whole sentences; the channel that keeps
+// a post cited by its platform link; a post's platform, link and date for a
+// manifest's `posts`. Without it a span is the second plus 10 s and those
+// posts are left out, each with a warning. Nothing is fetched.
+//
+// The output is checked before it is written: `convert` writes only a report
+// the report validator passes, `to-manifest` reads only one. Exit 0 written
+// (warnings on stderr), 1 refused with problems (nothing written), 2 an input
+// that cannot be read.
+
+import { readFile } from "node:fs/promises";
+import path from "node:path";
+import { writeJsonAtomic } from "../lib/jsonFile-server";
+import type { Problem } from "../lib/citations/validate";
+import { askToReport, readAskAnswer } from "../lib/report/convertAsk";
+import { manifestToReport, reportToManifest } from "../lib/report/convertManifest";
+import { sweepToReport } from "../lib/report/convertSweep";
+import type { ConvertContext, Converted } from "../lib/report/convertShared";
+import { diskCuesOf, diskPostChannelOf, diskPostOf } from "../lib/report/convert-server";
+
+type Out = { log: (s: string) => void; error: (s: string) => void };
+
+export const CONVERT_FROM = ["sweep", "ask", "manifest"] as const;
+export type ConvertFrom = (typeof CONVERT_FROM)[number];
+
+export function isConvertFrom(v: unknown): v is ConvertFrom {
+ return typeof v === "string" && (CONVERT_FROM as readonly string[]).includes(v);
+}
+
+function printProblems(out: Out, what: string, problems: Problem[]): void {
+ out.error(`${what}: ${problems.length} problem(s):`);
+ for (const p of problems) out.error(` ${p.path || "(root)"}: ${p.message}`);
+}
+
+async function readInput(file: string, out: Out, what: string): Promise<string | null> {
+ try {
+ return await readFile(file, "utf8");
+ } catch (err) {
+ out.error(`${what}: cannot read ${file}: ${(err as Error).message}`);
+ return null;
+ }
+}
+
+function parseJson(text: string, file: string, out: Out, what: string): unknown {
+ try {
+ return JSON.parse(text);
+ } catch (err) {
+ out.error(`${what}: ${file} is not JSON: ${(err as Error).message}`);
+ return undefined;
+ }
+}
+
+export async function convertMain(
+ opts: { from: ConvertFrom; input: string; out: string; channelsDir?: string; id?: string; title?: string },
+ out: Out = console,
+): Promise<number> {
+ const what = `reports convert ${opts.from}`;
+ const text = await readInput(opts.input, out, what);
+ if (text === null) return 2;
+ const warnings: string[] = [];
+ const ctx: ConvertContext = opts.channelsDir
+ ? { cuesOf: diskCuesOf(opts.channelsDir, warnings), postChannelOf: diskPostChannelOf(opts.channelsDir) }
+ : {};
+ const named = { ...ctx, ...(opts.id ? { id: opts.id } : {}), ...(opts.title ? { title: opts.title } : {}) };
+
+ let converted: Converted;
+ if (opts.from === "sweep") {
+ converted = await sweepToReport(text, named);
+ } else {
+ const raw = parseJson(text, opts.input, out, what);
+ if (raw === undefined) return 2;
+ if (opts.from === "ask") {
+ const answer = readAskAnswer(raw);
+ if (!answer) {
+ out.error(`${what}: ${opts.input} is not an /ask answer ({ answer, sources }, a saved chat or an assistant message)`);
+ return 2;
+ }
+ converted = await askToReport(answer, named);
+ } else {
+ converted = await manifestToReport(raw, named);
+ const hasStills = Object.values(converted.report.citations ?? {}).some((c) => c.kind === "source" && c.image);
+ if (hasStills && path.resolve(path.dirname(opts.input)) !== path.resolve(path.dirname(opts.out))) {
+ warnings.push(
+ "the report's stills are paths relative to the manifest's directory — copy them beside the report (or write it there)",
+ );
+ }
+ }
+ }
+
+ for (const w of [...warnings, ...converted.warnings]) out.error(`warning: ${w}`);
+ if (converted.problems.length) {
+ printProblems(out, `${what}: not written`, converted.problems);
+ return 1;
+ }
+ await writeJsonAtomic(opts.out, converted.report, { mkdir: true });
+ const r = converted.report;
+ const claims = r.sections.reduce((n, s) => n + (s.claims?.length ?? 0), 0);
+ out.log(
+ `${opts.out}: report ${r.id} (${r.kind}) — ${r.sections.length} section(s), ${claims} claim(s), ${Object.keys(r.citations ?? {}).length} citation(s)`,
+ );
+ return 0;
+}
+
+export async function toManifestMain(
+ opts: { input: string; out: string; channelsDir?: string; siteOrigin?: string },
+ out: Out = console,
+): Promise<number> {
+ const what = "reports to-manifest";
+ const text = await readInput(opts.input, out, what);
+ if (text === null) return 2;
+ const raw = parseJson(text, opts.input, out, what);
+ if (raw === undefined) return 2;
+ const result = await reportToManifest(raw, {
+ ...(opts.siteOrigin ? { siteOrigin: opts.siteOrigin } : {}),
+ ...(opts.channelsDir ? { postOf: diskPostOf(opts.channelsDir) } : {}),
+ });
+ if (!result.ok) {
+ printProblems(out, `${what}: ${opts.input} is not a sound report`, result.problems);
+ return 1;
+ }
+ const warnings = [...result.warnings];
+ const timeline = result.manifest.timeline as { type: string }[];
+ if (
+ timeline.some((e) => e.type === "image") &&
+ path.resolve(path.dirname(opts.input)) !== path.resolve(path.dirname(opts.out))
+ ) {
+ warnings.push("the images' src are paths relative to the report's directory — copy the stills beside the manifest");
+ }
+ for (const w of warnings) out.error(`warning: ${w}`);
+ await writeJsonAtomic(opts.out, result.manifest, { mkdir: true });
+ const posts = (result.manifest.posts as unknown[] | undefined)?.length ?? 0;
+ out.log(`${opts.out}: ${timeline.length} timeline entr${timeline.length === 1 ? "y" : "ies"}, ${posts} post(s)`);
+ return 0;
+}
diff --git a/common/lib/report/convert-server.ts b/common/lib/report/convert-server.ts
@@ -0,0 +1,123 @@
+// The converters' disk half: what a channels tree can tell a converter, as the
+// callbacks the converters take — a record's cues, the archive channel that
+// keeps a post, a post's record. The CLI is bin/reports-convert.ts.
+//
+// Reads only TEXT: `channels/<slug>/data/<id>/transcript.cues.json` (behind
+// the text guard, lib/channelMedia.ts — a legacy channel or one mid-migration
+// is "no cues", with a warning, never a read of the wrong tree), and each
+// channel's `posts-archive` and `posts/*.jsonl`. Writes nothing.
+//
+// SERVER-ONLY (node:fs).
+
+import { readdir, readFile } from "node:fs/promises";
+import path from "node:path";
+import { assertChannelTextReadable } from "../channelMedia";
+import { isSafeChannelSegment, isSafeIdSegment } from "../citations/moments";
+import { readAllPosts, readSeenPostIds } from "../posts-server";
+import type { Post } from "../posts";
+import type { Cue, CuesOf, PostChannelOf } from "./convertShared";
+import type { PostOf, PostRecord } from "./convertManifest";
+
+function isCue(v: unknown): v is Cue {
+ if (!v || typeof v !== "object") return false;
+ const c = v as Record<string, unknown>;
+ return typeof c.start === "number" && typeof c.end === "number" && typeof c.text === "string" && c.end >= c.start;
+}
+
+// A record's cues off the channels tree, cached per record. A channel whose
+// text is not readable is one warning and no cues.
+export function diskCuesOf(channelsDir: string, warnings: string[] = []): CuesOf {
+ const guarded = new Map<string, Promise<boolean>>();
+ const cache = new Map<string, Promise<Cue[] | null>>();
+ const readable = (slug: string) => {
+ let p = guarded.get(slug);
+ if (!p) {
+ p = assertChannelTextReadable({ channelsDir }, slug).then(
+ () => true,
+ (err: Error) => {
+ warnings.push(`${slug}: ${err.message} — no cues read from it`);
+ return false;
+ },
+ );
+ guarded.set(slug, p);
+ }
+ return p;
+ };
+ return (channel, id) => {
+ if (!isSafeChannelSegment(channel) || !isSafeIdSegment(id)) return Promise.resolve(null);
+ const key = `${channel}/${id}`;
+ let p = cache.get(key);
+ if (!p) {
+ p = (async () => {
+ if (!(await readable(channel))) return null;
+ let doc: unknown;
+ try {
+ doc = JSON.parse(await readFile(path.join(channelsDir, channel, "data", id, "transcript.cues.json"), "utf8"));
+ } catch {
+ return null;
+ }
+ const cues = (doc as { cues?: unknown })?.cues;
+ if (!Array.isArray(cues)) return null;
+ return cues.filter(isCue).sort((a, b) => a.start - b.start);
+ })();
+ cache.set(key, p);
+ }
+ return p;
+ };
+}
+
+// The channel directories under the tree, by name.
+async function channelSlugs(channelsDir: string): Promise<string[]> {
+ const entries = await readdir(channelsDir, { withFileTypes: true }).catch(() => []);
+ return entries.filter((e) => e.isDirectory() && isSafeChannelSegment(e.name)).map((e) => e.name).sort();
+}
+
+const squash = (v: string) => v.toLowerCase().replace(/^@/, "").replace(/[^a-z0-9]+/g, "");
+
+// The archive channel that keeps a post: every channel's posts archive, read
+// once; among several, the one whose slug carries the post's handle.
+export function diskPostChannelOf(channelsDir: string): PostChannelOf {
+ let index: Promise<Map<string, string[]>> | null = null;
+ const build = async () => {
+ const byId = new Map<string, string[]>();
+ for (const slug of await channelSlugs(channelsDir)) {
+ for (const id of await readSeenPostIds(path.join(channelsDir, slug))) {
+ byId.set(id, [...(byId.get(id) ?? []), slug]);
+ }
+ }
+ return byId;
+ };
+ return async ({ id, handle }) => {
+ index ??= build();
+ const slugs = (await index).get(id) ?? [];
+ if (slugs.length <= 1) return slugs[0] ?? null;
+ const h = handle ? squash(handle.split(".")[0]) : "";
+ return (h && slugs.find((s) => squash(s).includes(h))) || slugs[0];
+ };
+}
+
+// A post's record off its channel's archive, each channel read once.
+export function diskPostOf(channelsDir: string): PostOf {
+ const channels = new Map<string, Promise<Map<string, Post>>>();
+ return async (channel, id) => {
+ if (!isSafeChannelSegment(channel)) return null;
+ let p = channels.get(channel);
+ if (!p) {
+ p = readAllPosts(path.join(channelsDir, channel)).then(
+ (posts) => new Map(posts.map((post) => [post.id, post])),
+ () => new Map(),
+ );
+ channels.set(channel, p);
+ }
+ const post = (await p).get(id);
+ if (!post) return null;
+ const rec: PostRecord = {
+ platform: post.platform === "bluesky" ? "bluesky" : "x",
+ url: post.url,
+ createdAt: post.createdAt,
+ author: post.author,
+ };
+ if (post.authorName) rec.authorName = post.authorName;
+ return rec;
+ };
+}
diff --git a/common/lib/report/convertAsk.ts b/common/lib/report/convertAsk.ts
@@ -0,0 +1,172 @@
+// An /ask ANSWER → `report.json` (kind "sweep").
+//
+// The export's /ask writes an answer (or a running report) in markdown that
+// cites its sources with numbered markers — `[n]`, or `[n @ mm:ss]` for a
+// moment (also `h:mm:ss`, stray spaces tolerated) — where `n` indexes the
+// answer's source list, the retrieved records (export/app/ask/citations.tsx
+// parses the same markers; export/app/lib/askRetrieval.ts `RetrievedVideo` is
+// a source). Accepted as input, all JSON:
+//
+// { "answer": "<md>", "sources": [<source>…], "question"?: "…" }
+// a saved chat ({ "messages": […], "report"?, "reportSources"? }): its
+// report and report sources when it has them, else its last answer and
+// that answer's sources, with the question before it
+// one assistant message ({ "role": "assistant", "content", "sources" })
+//
+// Each marker becomes `[label](cite:<id>)`: a post source a post citation
+// (its words the quote), any other source a span at the marked second —
+// snapped to the nearest second the source's excerpt lines carry, as the
+// /ask page does, so a slightly-off model time lands on a real line — or at
+// its first excerpt line when the marker has no time; its quote is that
+// line's text. Then the answer is structured like a sweep (./convertShared.ts
+// `structureMarkdown`): paragraphs and list items that cite are claims.
+
+import {
+ CitationRegistry,
+ emptyReport,
+ finish,
+ reportIdOf,
+ structureMarkdown,
+ type ConvertContext,
+ type Converted,
+} from "./convertShared";
+
+export type AskSource = {
+ // `<channel>/<id>`: the record (or post) the source is.
+ key: string;
+ title?: string;
+ uploadDate?: string;
+ snippets?: { clock?: string; seconds: number; text: string }[];
+ isPost?: boolean;
+};
+
+export type AskAnswer = { answer: string; sources: AskSource[]; question?: string };
+
+export type AskConvertOptions = ConvertContext & {
+ id?: string;
+ // Default: the question, else "Answer".
+ title?: string;
+ sectionTitle?: string;
+};
+
+const isObj = (v: unknown): v is Record<string, unknown> => !!v && typeof v === "object" && !Array.isArray(v);
+
+function sourcesOf(v: unknown): AskSource[] | null {
+ if (!Array.isArray(v)) return null;
+ return v.filter((s): s is AskSource => isObj(s) && typeof s.key === "string");
+}
+
+// The answer, its sources and its question out of any accepted shape; null for
+// JSON that is none of them.
+export function readAskAnswer(raw: unknown): AskAnswer | null {
+ if (!isObj(raw)) return null;
+ const question = typeof raw.question === "string" ? raw.question : undefined;
+ for (const [text, list] of [
+ ["answer", "sources"],
+ ["report", "reportSources"],
+ ] as const) {
+ const sources = sourcesOf(raw[list]);
+ if (typeof raw[text] === "string" && sources) return { answer: raw[text] as string, sources, question };
+ }
+ if (raw.role === "assistant" && typeof raw.content === "string") {
+ return { answer: raw.content, sources: sourcesOf(raw.sources) ?? [], question };
+ }
+ if (Array.isArray(raw.messages)) {
+ const messages = raw.messages.filter(isObj);
+ for (let i = messages.length - 1; i >= 0; i--) {
+ const m = messages[i];
+ if (m.role !== "assistant" || typeof m.content !== "string" || !m.content.trim()) continue;
+ const asked = messages.slice(0, i).reverse().find((u) => u.role === "user" && typeof u.content === "string");
+ return { answer: m.content, sources: sourcesOf(m.sources) ?? [], question: question ?? (asked?.content as string | undefined) };
+ }
+ }
+ return null;
+}
+
+// `mm:ss` or `h:mm:ss` as whole seconds; null when malformed.
+export function parseClock(s: string): number | null {
+ const parts = s.split(":");
+ if (parts.length < 2 || parts.length > 3) return null;
+ let total = 0;
+ for (const p of parts) {
+ if (!/^\d+$/.test(p)) return null;
+ total = total * 60 + Number(p);
+ }
+ return total;
+}
+
+// One citation marker: a source number, an optional `@ mm:ss`, stray spaces
+// tolerated — not one already followed by a link target.
+const MARKER_RE = /\[\s*(\d+)\s*(?:@\s*(\d{1,2}(?::\d{2}){1,2})\s*)?\](?!\()/g;
+
+// The excerpt line a marker cites: the one nearest the marked second, else
+// the first.
+function snippetAt(src: AskSource, seconds: number | null): { seconds: number; text: string } | null {
+ const lines = src.snippets ?? [];
+ if (!lines.length) return null;
+ if (seconds === null) return lines[0];
+ return lines.reduce((best, s) => (Math.abs(s.seconds - seconds) < Math.abs(best.seconds - seconds) ? s : best));
+}
+
+const dateOfUpload = (d: string | undefined) =>
+ d && /^\d{8}$/.test(d) ? `${d.slice(0, 4)}-${d.slice(4, 6)}-${d.slice(6, 8)}` : undefined;
+
+export async function askToReport(input: AskAnswer, opts: AskConvertOptions = {}): Promise<Converted> {
+ const warnings: string[] = [];
+ const registry = new CitationRegistry(opts, warnings);
+
+ // The markers, outside code, as `[label](cite:<id>)`.
+ const segments = input.answer.split(/(```[\s\S]*?```|`[^`]*`)/g);
+ for (let i = 0; i < segments.length; i += 2) {
+ let out = "";
+ let last = 0;
+ for (const m of segments[i].matchAll(MARKER_RE)) {
+ out += segments[i].slice(last, m.index);
+ last = m.index + m[0].length;
+ const n = Number(m[1]);
+ const src = input.sources[n - 1];
+ if (!src) {
+ warnings.push(`[${m[1]}${m[2] ? ` @ ${m[2]}` : ""}] names no source (the answer has ${input.sources.length})`);
+ out += m[0];
+ continue;
+ }
+ const slash = src.key.indexOf("/");
+ if (slash <= 0) {
+ warnings.push(`source ${n} (${src.key}) is not a <channel>/<id> key — left as it is`);
+ out += m[0];
+ continue;
+ }
+ const channel = src.key.slice(0, slash);
+ const id = src.key.slice(slash + 1);
+ const marked = m[2] ? parseClock(m[2]) : null;
+ let cid: string;
+ if (src.isPost) {
+ const words = (src.snippets ?? []).map((s) => s.text).join("\n").trim() || src.title || id;
+ cid = registry.post(channel, id, words, { date: dateOfUpload(src.uploadDate) });
+ } else {
+ const line = snippetAt(src, marked);
+ if (!line) warnings.push(`source ${n} (${src.key}) has no excerpt lines — its quote is its title`);
+ cid = await registry.span(channel, id, line?.seconds ?? marked ?? 0, line?.text.trim() || src.title || id, {
+ label: src.title,
+ });
+ }
+ out += `[${m[0].slice(1, -1).trim()}](cite:${cid})`;
+ }
+ segments[i] = out + segments[i].slice(last);
+ }
+
+ const doc = await structureMarkdown(
+ segments.join(""),
+ async ({ href }) => (href.startsWith("cite:") ? href.slice("cite:".length) : null),
+ { defaultSection: opts.sectionTitle ?? "Answer" },
+ );
+ const question = input.question?.replace(/\s+/g, " ").trim();
+ const title = opts.title ?? doc.title ?? question ?? "Answer";
+ const report = emptyReport(reportIdOf(opts.id, title), "sweep", title);
+ if (question && question !== title) report.subtitle = question;
+ if (doc.summary) report.summary = doc.summary;
+ report.citations = registry.citations;
+ report.sections = doc.sections;
+ if (!Object.keys(registry.citations).length) warnings.push("the answer cites no source");
+ return finish(report, warnings);
+}
diff --git a/common/lib/report/convertManifest.ts b/common/lib/report/convertManifest.ts
@@ -0,0 +1,552 @@
+// A report-to-video MANIFEST ↔ `report.json`, both ways. The video and the
+// report page can be cut from one source: a manifest becomes a report, and a
+// report becomes a starter manifest to refine in umtool.
+//
+// The manifest is umtool/report-to-video's (its README, "Manifest shape").
+// What maps to what:
+//
+// manifest report
+// ──────────────────────────────────────── ──────────────────────────────────────
+// slug, title, subtitle id, title, subtitle
+// card (style "chapter", or none) a section: heading → title, sub → body
+// clip (channel ?? provenance.channelSlug, a video citation, id = the clip's id;
+// video, cutStart ?? start, an `audioOnly` clip an audio one
+// cutEnd ?? end, quote, date, note)
+// image (src, or its first panel; quote, a source citation (its still = src)
+// title, date, citeUrl, note) and a source of its own
+// posts[] (siteChannel, its id, text, date) a post citation, id = the post's id
+// `claim: { id, verdict }` on entries a claim with that verdict; its first
+// image is its source sentence
+// ledger[] (id, quote, label, date, a claim with no verdict (text = the
+// entryId, channel, video, cite) quote, title = the label)
+//
+// An entry that carries no claim is cited in its section's body, one line
+// each. A post goes with the clip it is attached to (`attachTo`, else the
+// clip it follows by date, else the first) — under that clip's claim, or in
+// its section's body. A claim's text is its ledger entry's quote, else the
+// heading of a card that carries it, else its source sentence, else its first
+// clip's quote. A report with any verdict is a fact-check, else a sweep.
+//
+// Left out, with a warning: a clip with its own media (`src`: no record to
+// cite), an entry without a quote, a post no archive channel is known to keep.
+// A manifest has more than a report (render settings, holds, the deck, cut
+// edits, redactions, the ledger's adjudication) and a report more than a
+// manifest (findings, a claim's own title, citation pads, labels, speakers);
+// neither survives the trip. Image paths are written as the manifest has
+// them: relative to ITS directory (the CLI says so when the report is written
+// elsewhere).
+//
+// A report → a STARTER manifest: no cold open — a title card, then per section
+// a chapter card, its body's citations in order, and per claim its source
+// sentence's still (an `image`) and its video and audio citations (`clip`s),
+// each carrying `claim: { id, verdict }` when the claim has a verdict, and its
+// posts in `posts[]` attached to the claim's last clip. Spans are written as
+// the report has them, so `resolve-windows.mjs` and the clip bench start from
+// the cited sentences. A post needs its platform, its link and its date: from
+// `postOf` (the CLI reads the channel's archive) or, for a numeric X id, its
+// `x.com/i/status/` link and the citation's date; one that has none of them is
+// left out with a warning.
+
+import type { Citation, Source, SourceCitation } from "../citations/schema";
+import { isHttpUrl, isPartialDate } from "../citations/validate";
+import {
+ IdAllocator,
+ emptyReport,
+ finish,
+ citedIds,
+ reportIdOf,
+ spanAt,
+ type CitationMap,
+ type ConvertContext,
+ type Converted,
+} from "./convertShared";
+import type { Claim, Report, Section } from "./schema";
+import { isVerdict, type Verdict } from "./verdicts";
+import { parseReport } from "./validate";
+import type { Problem } from "../citations/validate";
+
+type Obj = Record<string, unknown>;
+
+const isObj = (v: unknown): v is Obj => !!v && typeof v === "object" && !Array.isArray(v);
+const str = (v: unknown): string | undefined => (typeof v === "string" && v.trim() ? v : undefined);
+const num = (v: unknown): number | undefined => (typeof v === "number" && Number.isFinite(v) ? v : undefined);
+
+export type ManifestConvertOptions = ConvertContext & { id?: string; title?: string };
+
+// A manifest entry's claim, when it carries a sound one.
+function claimOfEntry(e: Obj): { id: string; verdict: Verdict } | null {
+ if (!isObj(e.claim) || e.type === "teaser") return null;
+ const id = str(e.claim.id)?.trim();
+ return id && isVerdict(e.claim.verdict) ? { id, verdict: e.claim.verdict } : null;
+}
+
+// The id a post has on its platform: `postId`, else the link's.
+export function postNativeId(post: Obj): string | null {
+ const given = str(post.postId);
+ if (given && /^[A-Za-z0-9_-]{1,128}$/.test(given)) return given;
+ try {
+ const u = new URL(String(post.url ?? ""));
+ const m = /\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname) ?? /\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname);
+ return m ? m[1] : null;
+ } catch {
+ return null;
+ }
+}
+
+const dayOf = (d: unknown) => (typeof d === "string" && /^\d{4}-\d{2}-\d{2}/.test(d) ? d.slice(0, 10) : null);
+
+// The clip a post goes with, as the build attaches it: `attachTo`, else the
+// clip whose date most closely precedes the post's (ties to the later in the
+// cut), else the first.
+function clipForPost(post: Obj, clips: Obj[]): Obj | null {
+ const named = str(post.attachTo);
+ if (named) {
+ const c = clips.find((e) => e.id === named);
+ if (c) return c;
+ }
+ const day = dayOf(post.date);
+ let best: Obj | null = null;
+ if (day) {
+ for (const c of clips) {
+ const cd = dayOf(c.date);
+ if (cd && cd <= day && (!best || cd >= (dayOf(best.date) as string))) best = c;
+ }
+ }
+ return best ?? clips[0] ?? null;
+}
+
+export async function manifestToReport(manifest: unknown, opts: ManifestConvertOptions = {}): Promise<Converted> {
+ const warnings: string[] = [];
+ const m: Obj = isObj(manifest) ? manifest : {};
+ const timeline = (Array.isArray(m.timeline) ? m.timeline : []).filter(isObj);
+ const ledger = (Array.isArray(m.ledger) ? m.ledger : []).filter(isObj);
+ const posts = (Array.isArray(m.posts) ? m.posts : []).filter(isObj);
+ const prov = isObj(m.provenance) ? m.provenance : {};
+ const title = opts.title ?? str(m.title) ?? str(m.slug) ?? "Report";
+ const report = emptyReport(reportIdOf(opts.id ?? str(m.slug), title), "sweep", title);
+ if (str(m.subtitle)) report.subtitle = m.subtitle as string;
+
+ const citations: CitationMap = {};
+ const sources: Record<string, Source> = {};
+ const citeIds = new IdAllocator();
+ const anchors = new IdAllocator();
+ const where = (e: Obj, i: number) => `timeline[${i}] ${str(e.id) ?? "?"}`;
+
+ // ── The citations every entry is ──
+ const citeOf = new Map<Obj, string>();
+ for (const [i, e] of timeline.entries()) {
+ if (e.type === "clip") {
+ if (str(e.src)) {
+ warnings.push(`${where(e, i)}: a clip of its own media (src) cites no record — left out`);
+ continue;
+ }
+ const channel = str(e.channel) ?? str(prov.channelSlug);
+ const video = str(e.video);
+ const start = num(e.cutStart) ?? num(e.start);
+ const end = num(e.cutEnd) ?? num(e.end);
+ const quote = str(e.quote);
+ if (!channel || !video || start === undefined || end === undefined) {
+ warnings.push(`${where(e, i)}: a clip needs a channel, a video, a start and an end — left out`);
+ continue;
+ }
+ if (!quote) {
+ warnings.push(`${where(e, i)}: a clip without a quote — left out (a citation's quote is verbatim)`);
+ continue;
+ }
+ const cid = citeIds.claim(str(e.id) ?? "c", "c");
+ const c: Citation = { kind: e.audioOnly === true ? "audio" : "video", channel, id: video, start, end, quote };
+ if (str(e.date) && isPartialDate(e.date as string)) c.date = e.date as string;
+ if (str(e.note)) c.note = e.note as string;
+ citations[cid] = c;
+ citeOf.set(e, cid);
+ } else if (e.type === "image") {
+ const panels = Array.isArray(e.panels) ? e.panels.filter(isObj) : [];
+ const src = str(e.src) ?? str(panels[0]?.src);
+ if (!src) {
+ warnings.push(`${where(e, i)}: an image with no src — left out`);
+ continue;
+ }
+ if (panels.length > 1) warnings.push(`${where(e, i)}: ${panels.length} panels — the citation's still is the first`);
+ let quote = str(e.quote);
+ if (!quote) {
+ quote = str(e.title) ?? str(e.id) ?? src;
+ warnings.push(`${where(e, i)}: an image without a quote — its citation quotes its title; give it the still's words`);
+ }
+ const cid = citeIds.claim(str(e.id) ?? "a", "a");
+ const sid = citeIds.claim(`src-${cid}`);
+ const source: Source = { kind: "other", title: str(e.title) ?? cid };
+ if (str(e.citeUrl) && isHttpUrl(e.citeUrl as string)) source.url = e.citeUrl as string;
+ if (str(e.date) && isPartialDate(e.date as string)) source.date = e.date as string;
+ sources[sid] = source;
+ const c: SourceCitation = { kind: "source", source: sid, quote, image: src };
+ if (str(e.note)) c.note = e.note as string;
+ citations[cid] = c;
+ citeOf.set(e, cid);
+ }
+ }
+
+ // ── Posts: each a citation, and the clip it goes with ──
+ const clips = timeline.filter((e) => e.type === "clip" && citeOf.has(e));
+ const postsOfClip = new Map<Obj, string[]>();
+ const loosePosts: string[] = [];
+ for (const [i, p] of posts.entries()) {
+ if (p.hide === true) continue;
+ const native = postNativeId(p);
+ const platform = p.platform === "bluesky" ? "bluesky" : "x";
+ const channel =
+ str(p.siteChannel) ??
+ (native && opts.postChannelOf ? await opts.postChannelOf({ platform, id: native, handle: str(p.handle) }) : null);
+ const text = str(p.text);
+ if (!native || !channel || !text) {
+ warnings.push(
+ `posts[${i}] ${str(p.id) ?? "?"}: ${!native ? "no id on its platform" : !text ? "no text" : "no archive channel known to keep it (siteChannel)"} — left out`,
+ );
+ continue;
+ }
+ const cid = citeIds.claim(str(p.id) ?? "p", "p");
+ citations[cid] = {
+ kind: "post",
+ channel,
+ id: native,
+ quote: text,
+ ...(str(p.date) && isPartialDate(p.date as string) ? { date: p.date as string } : {}),
+ };
+ const clip = clipForPost(p, clips);
+ if (clip) postsOfClip.set(clip, [...(postsOfClip.get(clip) ?? []), cid]);
+ else loosePosts.push(cid);
+ }
+
+ // ── Ledger claims: what the ledger says a claim is, and the clip it pins ──
+ const ledgerById = new Map<string, Obj>();
+ for (const l of ledger) if (str(l.id)) ledgerById.set(l.id as string, l);
+ const timelineClaims = new Set(timeline.map(claimOfEntry).filter((c) => c).map((c) => c!.id));
+ const ledgerClaimOfClip = new Map<string, string>();
+ for (const l of ledger) {
+ const id = str(l.id);
+ const pinned = str(l.entryId);
+ if (id && pinned && !timelineClaims.has(id)) ledgerClaimOfClip.set(pinned, id);
+ }
+
+ // ── The walk: sections, claims, bodies ──
+ const sections: { section: Section; body: string[] }[] = [];
+ const claims = new Map<string, { claim: Claim; texts: { from: string; text: string }[] }>();
+ const current = () => {
+ if (!sections.length) {
+ sections.push({ section: { id: anchors.claim("opening"), title: "Opening", claims: [] }, body: [] });
+ }
+ return sections[sections.length - 1];
+ };
+ const claimFor = (id: string, verdict: Verdict | null) => {
+ let k = claims.get(id);
+ if (!k) {
+ const claim: Claim = { id: anchors.claim(id, "k"), text: "", citations: [] };
+ if (verdict) claim.verdict = verdict;
+ k = { claim, texts: [] };
+ claims.set(id, k);
+ current().section.claims!.push(claim);
+ }
+ return k;
+ };
+ const list = (claim: Claim, cid: string) => {
+ if (!claim.citations!.includes(cid)) claim.citations!.push(cid);
+ };
+ const bodyLine = (cid: string) => {
+ const c = citations[cid];
+ const quote = c.quote.replace(/\s+/g, " ").trim();
+ const label =
+ c.kind === "post" ? "post" : c.kind === "source" ? (sources[c.source]?.title ?? cid) : c.kind === "page" ? cid : `${c.id}`;
+ return `- “${quote}” [${label.replace(/[[\]]/g, "")}](cite:${cid})`;
+ };
+
+ for (const e of timeline) {
+ const tagged = claimOfEntry(e);
+ if (e.type === "card") {
+ if (tagged) {
+ const k = claimFor(tagged.id, tagged.verdict);
+ if (str(e.heading)) k.texts.push({ from: "card", text: e.heading as string });
+ continue;
+ }
+ const style = e.style;
+ if ((style === undefined || style === "chapter") && str(e.heading)) {
+ const section: Section = { id: anchors.claim(str(e.id) ?? "section", "section"), title: e.heading as string, claims: [] };
+ sections.push({ section, body: str(e.sub) ? [e.sub as string] : [] });
+ }
+ continue;
+ }
+ const cid = citeOf.get(e);
+ if (!cid) continue;
+ const ledgerClaim = !tagged && str(e.id) ? ledgerClaimOfClip.get(e.id as string) : undefined;
+ const claimId = tagged?.id ?? ledgerClaim;
+ if (claimId) {
+ const k = claimFor(claimId, tagged?.verdict ?? null);
+ const c = citations[cid];
+ if (c.kind === "source" && !k.claim.sourceQuote) {
+ k.claim.sourceQuote = { citation: cid };
+ k.texts.push({ from: "source", text: c.quote });
+ } else {
+ list(k.claim, cid);
+ if (c.kind !== "source") k.texts.push({ from: "clip", text: c.quote });
+ }
+ for (const pid of postsOfClip.get(e) ?? []) list(k.claim, pid);
+ } else {
+ const body = current().body;
+ body.push(bodyLine(cid));
+ for (const pid of postsOfClip.get(e) ?? []) body.push(bodyLine(pid));
+ }
+ }
+
+ // Ledger claims no clip pins: their own moment, in a section of their own.
+ const unpinned = ledger.filter((l) => {
+ const id = str(l.id);
+ return id && !claims.has(id) && !timelineClaims.has(id);
+ });
+ if (unpinned.length) {
+ sections.push({ section: { id: anchors.claim("ledger"), title: "Not clipped", claims: [] }, body: [] });
+ for (const l of unpinned) {
+ const k = claimFor(l.id as string, null);
+ const channel = str(l.channel) ?? str(prov.channelSlug);
+ const video = str(l.video);
+ const second = num(l.cite) ?? num(l.start);
+ const quote = str(l.quote);
+ if (channel && video && second !== undefined && quote) {
+ const span = await spanAt(channel, video, second, quote, opts, num(l.end));
+ const cid = citeIds.claim(`l-${l.id as string}`);
+ citations[cid] = { kind: "video", channel, id: video, start: span.start, end: span.end, quote };
+ list(k.claim, cid);
+ } else {
+ warnings.push(`ledger ${l.id as string}: no channel, video, second and quote to cite — the claim has no citation`);
+ }
+ }
+ }
+ if (loosePosts.length) {
+ const body = current().body;
+ for (const pid of loosePosts) body.push(bodyLine(pid));
+ }
+
+ // ── Claim texts and titles ──
+ for (const [id, { claim, texts }] of claims) {
+ const l = ledgerById.get(id);
+ const pick = (from: string) => texts.find((t) => t.from === from)?.text;
+ const text = str(l?.quote) ?? pick("card") ?? pick("source") ?? pick("clip");
+ if (str(l?.label) && str(l?.quote)) claim.title = l!.label as string;
+ if (text) claim.text = text.replace(/\s+/g, " ").trim();
+ else {
+ claim.text = id;
+ warnings.push(`claim ${id}: nothing in the manifest states it — its text is its id`);
+ }
+ if (!claim.citations!.length) delete claim.citations;
+ }
+
+ report.kind = [...claims.values()].some((k) => k.claim.verdict) ? "factcheck" : "sweep";
+ report.sources = sources;
+ report.citations = citations;
+ report.sections = sections.map(({ section, body }) => {
+ const out: Section = { id: section.id, title: section.title };
+ if (body.length) out.body = body.join("\n");
+ if (section.claims!.length) out.claims = section.claims;
+ return out;
+ });
+ return finish(report, warnings);
+}
+
+// ─── report → manifest ───
+
+// What a post record says that a post citation does not (the CLI reads it off
+// the channel's archive).
+export type PostRecord = {
+ platform: "x" | "bluesky";
+ url: string;
+ createdAt?: string;
+ author?: string;
+ authorName?: string;
+};
+
+export type PostOf = (channel: string, id: string) => Promise<PostRecord | null>;
+
+export type ManifestOptions = {
+ // The archive the cut's QR codes link (provenance.siteOrigin). Default ""
+ // — which `umtool check` blocks on until it is set, as for `umtool new`.
+ siteOrigin?: string;
+ postOf?: PostOf;
+ // Replaces the starter `render` block.
+ render?: Record<string, unknown>;
+ // `generatedOn`; default today.
+ today?: string;
+};
+
+// The render block `umtool new` writes (umtool/lib/projects/scaffold.mjs
+// `skeleton`), so a converted report renders as a new project does. Keep the
+// two alike.
+export const STARTER_RENDER: Readonly<Record<string, unknown>> = Object.freeze({
+ width: 1920,
+ height: 1080,
+ fps: 30,
+ audioRate: 48000,
+ audioChannels: 2,
+ maxHeightSource: 1080,
+ fontRegular: "/usr/share/fonts/TTF/FiraSans-Regular.ttf",
+ fontBold: "/usr/share/fonts/TTF/FiraSans-Bold.ttf",
+ palette: { bg: "#12100c", fg: "#f6f1e6", muted: "#a2957f", accent: "#c8752a", amber: "#ffc860" },
+ transition: 0.4,
+ fetchPad: 3,
+ snapWindow: 1.6,
+ silenceMinDur: 0.09,
+ silenceRelDb: 6,
+ headerHeight: 56,
+ footerHeight: 0,
+ crf: 21,
+ preset: "slow",
+ qr: { scale: 4, quiet: 3, ecc: "M", margin: 28 },
+});
+
+// A manifest entry's `date` is a calendar day.
+const DAY_RE = /^\d{4}-\d{2}-\d{2}$/;
+
+export const CARD_SECONDS = { title: 4, chapter: 3.5 } as const;
+export const IMAGE_SECONDS = 6;
+
+export type ManifestResult =
+ | { ok: true; manifest: Record<string, unknown>; warnings: string[] }
+ | { ok: false; problems: Problem[] };
+
+export async function reportToManifest(raw: unknown, opts: ManifestOptions = {}): Promise<ManifestResult> {
+ const parsed = parseReport(raw);
+ if (!parsed.ok || parsed.problems.length) return { ok: false, problems: parsed.problems };
+ const report: Report = parsed.value;
+ const warnings: string[] = [];
+ const citations = report.citations ?? {};
+ const sources = report.sources ?? {};
+ // Timeline ids are file names in a build: letters, digits, `_` and `-`.
+ const entryIds = new IdAllocator(/[^A-Za-z0-9_-]+/g);
+ const postIds = new IdAllocator(/[^A-Za-z0-9_-]+/g);
+ const timeline: Record<string, unknown>[] = [];
+ const posts: Record<string, unknown>[] = [];
+ const postsDone = new Set<string>();
+ const channels = new Map<string, number>();
+ let lastClip: string | null = null;
+
+ timeline.push({
+ type: "card",
+ id: entryIds.claim("title"),
+ style: "title",
+ seconds: CARD_SECONDS.title,
+ heading: report.title,
+ sub: report.subtitle ?? "",
+ });
+
+ const claimTag = (claim: Claim | null) => (claim?.verdict ? { claim: { id: claim.id, verdict: claim.verdict } } : {});
+
+ const emit = async (cid: string, claim: Claim | null, sourceSentence = false) => {
+ const c = citations[cid];
+ if (!c) return;
+ switch (c.kind) {
+ case "video":
+ case "audio": {
+ const id = entryIds.claim(cid, "c");
+ timeline.push({
+ type: "clip",
+ id,
+ channel: c.channel,
+ video: c.id,
+ start: c.start,
+ end: c.end,
+ cite: Math.floor(c.start),
+ quote: c.quote,
+ ...(c.kind === "audio" ? { audioOnly: true } : {}),
+ ...(c.date && DAY_RE.test(c.date) ? { date: c.date } : {}),
+ ...(c.note ? { note: c.note } : {}),
+ ...claimTag(claim),
+ });
+ channels.set(c.channel, (channels.get(c.channel) ?? 0) + 1);
+ lastClip = id;
+ return;
+ }
+ case "source": {
+ if (!c.image) {
+ warnings.push(
+ `${cid}: ${sourceSentence ? "a claim's source sentence" : "a source citation"} with no still — no image entry (shoot it, then set its image)`,
+ );
+ return;
+ }
+ const source = sources[c.source];
+ const dayOfSource = [c.date, source?.date].find((d) => d && DAY_RE.test(d));
+ timeline.push({
+ type: "image",
+ id: entryIds.claim(cid, "a"),
+ src: c.image,
+ seconds: IMAGE_SECONDS,
+ title: source?.title ?? cid,
+ quote: c.quote,
+ ...(dayOfSource ? { date: dayOfSource } : {}),
+ ...(source?.url ? { citeUrl: source.url } : {}),
+ ...(c.note ? { note: c.note } : {}),
+ ...claimTag(claim),
+ });
+ return;
+ }
+ case "post": {
+ if (postsDone.has(cid)) return;
+ postsDone.add(cid);
+ const rec = opts.postOf ? await opts.postOf(c.channel, c.id) : null;
+ const platform = rec?.platform ?? (/^\d+$/.test(c.id) ? "x" : null);
+ const url = rec?.url ?? (platform === "x" ? `https://x.com/i/status/${c.id}` : null);
+ const date = c.date ?? rec?.createdAt;
+ if (!platform || !url || !date || !/^\d{4}-\d{2}-\d{2}/.test(date)) {
+ warnings.push(
+ `${cid}: post ${c.channel}/${c.id} — ${!platform || !url ? "its platform and link are unknown (give a channels dir)" : "no date"} — left out of posts`,
+ );
+ return;
+ }
+ posts.push({
+ id: postIds.claim(cid, "p"),
+ platform,
+ ...(rec?.authorName ? { author: rec.authorName } : {}),
+ ...(rec?.author ? { handle: rec.author } : {}),
+ date,
+ text: c.quote.slice(0, 3000),
+ url,
+ attachTo: lastClip,
+ siteChannel: c.channel,
+ ...(/^[A-Za-z0-9_-]{1,128}$/.test(c.id) ? { postId: c.id } : {}),
+ });
+ return;
+ }
+ case "page":
+ warnings.push(`${cid}: a page citation has no manifest entry — left out`);
+ return;
+ }
+ };
+
+ for (const section of report.sections) {
+ timeline.push({
+ type: "card",
+ id: entryIds.claim(section.id, "section"),
+ style: "chapter",
+ seconds: CARD_SECONDS.chapter,
+ heading: section.title,
+ });
+ for (const cid of citedIds(section.body)) await emit(cid, null);
+ for (const claim of section.claims ?? []) {
+ if (claim.sourceQuote) await emit(claim.sourceQuote.citation, claim, true);
+ const order = [...(claim.citations ?? []), ...citedIds(claim.findings)];
+ for (const cid of [...new Set(order)]) {
+ if (cid !== claim.sourceQuote?.citation) await emit(cid, claim);
+ }
+ }
+ }
+
+ const channelSlug = [...channels.entries()].sort((a, b) => b[1] - a[1])[0]?.[0] ?? "";
+ const manifest: Record<string, unknown> = {
+ schemaVersion: 1,
+ slug: report.id,
+ title: report.title,
+ subtitle: report.subtitle ?? "",
+ generatedOn: opts.today ?? new Date().toISOString().slice(0, 10),
+ provenance: { siteOrigin: opts.siteOrigin ?? "", channelSlug, channel: "" },
+ render: { ...(opts.render ?? STARTER_RENDER) },
+ timelineNodes: [],
+ timeline,
+ ...(posts.length ? { posts } : {}),
+ };
+ return { ok: true, manifest, warnings };
+}
diff --git a/common/lib/report/convertShared.ts b/common/lib/report/convertShared.ts
@@ -0,0 +1,628 @@
+// THE CONVERTERS' SHARED HALF — what bringing a cited document from elsewhere
+// into the citation model takes, whatever the document was:
+//
+// - reading a link a report cites with (`parseCitationHref`): an archive
+// viewer moment (`<origin>/?v=<channel>%2F<id>&t=<s>`, what /sweep and
+// /ask write), an archive post (`…&vm=post`), a moment page
+// (`/m/<key>/`), or a post on its own platform (x.com, bsky.app);
+// - turning ONE cited second into a span (`spanAt`): a record's cues, when
+// the caller can read them, widened to whole sentences by the one widening
+// (lib/cueWiden.mjs); else the second plus a default span;
+// - the markdown engine (`structureMarkdown`): a document's headings into
+// sections, its list items and paragraphs that carry a citation into
+// claims, everything else into the sections' bodies, and every citing link
+// rewritten to `[label](cite:<id>)`;
+// - ids, and the finished report checked by the report validator.
+//
+// The converters themselves are ./convertSweep.ts, ./convertAsk.ts and
+// ./convertManifest.ts. Nothing here reads the disk: the cues and the channel
+// that keeps a post arrive as callbacks (./convert-server.ts has the disk
+// ones), so every converter runs in a test on literals.
+
+import { widen } from "../cueWiden.mjs";
+import { MAX_CITATION_SPAN_SECONDS, REF_ID_RE, type Citation } from "../citations/schema";
+import { citeHref, extractCiteRefs } from "../citations/inline";
+import { parseMomentPath, roundMomentSeconds } from "../citations/moments";
+import { isPartialDate, type Problem } from "../citations/validate";
+import { REPORT_FORMAT, REPORT_ID_RE, REPORT_VERSION, type Claim, type Report, type Section } from "./schema";
+import { parseReport } from "./validate";
+
+export type Cue = { start: number; end: number; text: string };
+
+// A record's cues, sorted by start, or null when they cannot be read.
+export type CuesOf = (channel: string, id: string) => Promise<readonly Cue[] | null>;
+
+export type PostPlatformName = "x" | "bluesky";
+
+// The archive channel that keeps a post, or null.
+export type PostChannelOf = (post: { platform: PostPlatformName; id: string; handle?: string }) => Promise<string | null>;
+
+export type ConvertContext = {
+ cuesOf?: CuesOf;
+ postChannelOf?: PostChannelOf;
+ // A cited second with no cues to read becomes [second, second + this].
+ spanSeconds?: number;
+};
+
+// What a converter hands back: the report, what it had to guess or leave out
+// (warnings), and the report validator's problems — a report with problems is
+// not to be written.
+export type Converted = { report: Report; warnings: string[]; problems: Problem[] };
+
+// A cited second with no cues is this long. A /sweep or /ask citation carries
+// one second, never an end; ten seconds is about a spoken sentence.
+export const DEFAULT_SPAN_SECONDS = 10;
+
+// ─── Links ───
+
+export type ParsedCitationHref =
+ // An archive viewer moment, or a moment page: `seconds` is where it starts;
+ // a moment page also carries its end.
+ | { kind: "span"; channel: string; id: string; seconds: number; end?: number }
+ // An archived post on the archive.
+ | { kind: "post"; channel: string; id: string }
+ // A post on its own platform: the archive channel is not in the link.
+ | { kind: "post-original"; platform: PostPlatformName; id: string; handle?: string };
+
+const X_HOSTS = new Set(["x.com", "twitter.com", "mobile.twitter.com", "www.x.com", "www.twitter.com", "mobile.x.com"]);
+const BSKY_HOSTS = new Set(["bsky.app", "www.bsky.app"]);
+
+// What a link cites, or null for a link that cites nothing the model knows (a
+// platform's video page, an article, a relative link).
+export function parseCitationHref(href: string): ParsedCitationHref | null {
+ let u: URL;
+ try {
+ u = new URL(href.trim());
+ } catch {
+ return null;
+ }
+ if (u.protocol !== "http:" && u.protocol !== "https:") return null;
+ const host = u.hostname.toLowerCase();
+
+ if (X_HOSTS.has(host)) {
+ const m = /^\/([^/]+)\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname);
+ if (!m) return null;
+ return { kind: "post-original", platform: "x", id: m[2], ...(m[1] !== "i" ? { handle: m[1] } : {}) };
+ }
+ if (BSKY_HOSTS.has(host)) {
+ const m = /^\/profile\/([^/]+)\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname);
+ if (!m) return null;
+ return { kind: "post-original", platform: "bluesky", id: m[2], handle: m[1] };
+ }
+
+ const moment = parseMomentPath(u.pathname);
+ if (moment) {
+ return moment.kind === "span"
+ ? { kind: "span", channel: moment.channel, id: moment.id, seconds: moment.start, end: moment.end }
+ : { kind: "post", channel: moment.channel, id: moment.id };
+ }
+
+ // The viewer: `v` is `<channel>/<id>` (URL-encoded or not), `t` whole seconds.
+ const v = u.searchParams.get("v");
+ if (!v) return null;
+ const slash = v.indexOf("/");
+ if (slash <= 0 || slash === v.length - 1) return null;
+ const channel = v.slice(0, slash);
+ const id = v.slice(slash + 1);
+ if (u.searchParams.get("vm") === "post") return { kind: "post", channel, id };
+ const t = u.searchParams.get("t");
+ const seconds = t && /^\d+(\.\d+)?$/.test(t) ? Number(t) : 0;
+ return { kind: "span", channel, id, seconds };
+}
+
+// ─── Spans ───
+
+const EPS = 0.02;
+
+const wordCount = (s: string) => s.split(/\s+/).filter((w) => /[\p{L}\p{N}]/u.test(w)).length;
+
+const round2 = (n: number) => roundMomentSeconds(n);
+
+// The cue a cited second points at: the viewer's `t` is a cue's start,
+// floored, so the cue that starts in [t, t + 1) is the one cited; else the cue
+// that holds t; else the nearest one after it.
+function cueIndexAt(cues: readonly Cue[], t: number): number {
+ const starting = cues.findIndex((c) => c.start >= t - EPS && c.start < t + 1);
+ if (starting >= 0) return starting;
+ const holding = cues.findIndex((c) => c.start <= t + EPS && c.end > t);
+ if (holding >= 0) return holding;
+ const after = cues.findIndex((c) => c.start > t);
+ return after >= 0 ? after : cues.length - 1;
+}
+
+export type Span = { start: number; end: number; from: "cues" | "default" | "link" };
+
+// One cited second as a span. With the record's cues: the cue it cites, run
+// on cue by cue until it holds as many words as the quote, then widened to
+// whole sentences — and if the widened span is longer than a citation may be,
+// the unwidened cue run, and failing that the default. Without cues (or past
+// their end): [second, second + spanSeconds]. An `end` the link already gave
+// (a moment page) is kept as it is.
+export async function spanAt(
+ channel: string,
+ id: string,
+ seconds: number,
+ quote: string,
+ ctx: ConvertContext,
+ end?: number,
+): Promise<Span> {
+ if (end !== undefined && end > seconds) return { start: round2(seconds), end: round2(end), from: "link" };
+ const fallback: Span = {
+ start: round2(seconds),
+ end: round2(seconds + (ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS)),
+ from: "default",
+ };
+ const cues = ctx.cuesOf ? await ctx.cuesOf(channel, id) : null;
+ if (!cues || cues.length === 0 || seconds > cues[cues.length - 1].end) return fallback;
+ const i = cueIndexAt(cues, seconds);
+ const want = Math.max(1, wordCount(quote));
+ let j = i;
+ let have = wordCount(cues[i].text);
+ while (have < want && j + 1 < cues.length && cues[j + 1].end - cues[i].start <= MAX_CITATION_SPAN_SECONDS) {
+ j += 1;
+ have += wordCount(cues[j].text);
+ }
+ const fits = (s: number, e: number) => e > s && e - s <= MAX_CITATION_SPAN_SECONDS && round2(e) > round2(s);
+ const w = widen(cues, cues[i].start, cues[j].end);
+ if (fits(w.start, w.end)) return { start: round2(w.start), end: round2(w.end), from: "cues" };
+ if (fits(cues[i].start, cues[j].end)) return { start: round2(cues[i].start), end: round2(cues[j].end), from: "cues" };
+ return fallback;
+}
+
+// ─── Ids ───
+
+// Unique ids in one namespace, each a reference id (lib/citations/schema.ts
+// REF_ID_RE): a wanted id is kept when it is free and sound, else made sound
+// and suffixed `-2`, `-3`, ….
+export class IdAllocator {
+ private readonly taken = new Set<string>();
+ private readonly counters = new Map<string, number>();
+
+ constructor(private readonly charset: RegExp = /[^A-Za-z0-9_.:-]+/g) {}
+
+ has(id: string): boolean {
+ return this.taken.has(id);
+ }
+
+ claim(wanted: string, fallback = "x"): string {
+ let base = wanted.replace(this.charset, "-").replace(/^[^A-Za-z0-9]+/, "").slice(0, 56);
+ if (!base) base = fallback;
+ let id = base;
+ for (let n = 2; this.taken.has(id) || !REF_ID_RE.test(id); n++) id = `${base}-${n}`;
+ this.taken.add(id);
+ return id;
+ }
+
+ // The next free `<prefix>01`, `<prefix>02`, ….
+ next(prefix: string): string {
+ let n = this.counters.get(prefix) ?? 0;
+ let id: string;
+ do {
+ n += 1;
+ id = `${prefix}${String(n).padStart(2, "0")}`;
+ } while (this.taken.has(id));
+ this.counters.set(prefix, n);
+ this.taken.add(id);
+ return id;
+ }
+}
+
+// A heading or a title as an id: lowercase words joined by `-`.
+export function slugOf(text: string, max = 48): string {
+ return text
+ .normalize("NFKD")
+ .replace(/[̀-ͯ]/g, "")
+ .toLowerCase()
+ .replace(/[^a-z0-9]+/g, "-")
+ .replace(/^-+|-+$/g, "")
+ .slice(0, max)
+ .replace(/-+$/, "");
+}
+
+// A report id (a lowercase slug, REPORT_ID_RE) from what the caller asked for,
+// else from the title.
+export function reportIdOf(wanted: string | undefined, title: string): string {
+ if (wanted !== undefined) return wanted;
+ const slug = slugOf(title, 64);
+ return REPORT_ID_RE.test(slug) ? slug : "report";
+}
+
+// ─── Markdown ───
+
+// `[label](href)`: a label may hold one level of brackets (a title with
+// `[live]` in it); the href runs to the first space or `)`, optionally in
+// `<…>`, optionally followed by a quoted title.
+const LINK_RE = /\[((?:[^[\]]|\[[^[\]]*\])*)\]\(\s*<?([^)\s>]+)>?(?:\s+"[^"]*")?\s*\)/g;
+const CODE_SPAN_RE = /(?<!`)(`+)(?!`)[\s\S]*?(?<!`)\1(?!`)/g;
+const HEADING_RE = /^ {0,3}(#{1,6})\s+(.*?)\s*#*\s*$/;
+const FENCE_RE = /^ {0,3}(`{3,}|~{3,})/;
+const LIST_ITEM_RE = /^ ?([-*+]|\d{1,9}[.)])\s+/;
+const QUOTE_LINE_RE = /^ {0,3}>/;
+const RULE_RE = /^ {0,3}([-*_])(\s*\1){2,}\s*$/;
+
+type MdHeading = { kind: "heading"; level: number; text: string };
+type MdBlock = { kind: "block"; md: string; code: boolean };
+type MdPart = MdHeading | MdBlock;
+
+// The markdown as headings and blocks: a fenced block, a top-level list item
+// (with its continuation and nested lines), a run of quote lines (and the
+// lines that lazily continue it), or a paragraph. Blank lines and rules
+// separate them.
+function partsOf(md: string): MdPart[] {
+ const out: MdPart[] = [];
+ const lines = md.replace(/\r\n?/g, "\n").split("\n");
+ let cur: string[] = [];
+ let curQuote = false;
+ const flush = () => {
+ if (cur.length) out.push({ kind: "block", md: cur.join("\n"), code: false });
+ cur = [];
+ curQuote = false;
+ };
+ for (let i = 0; i < lines.length; i++) {
+ const line = lines[i];
+ const fence = FENCE_RE.exec(line);
+ if (fence) {
+ flush();
+ const close = new RegExp(`^ {0,3}${fence[1][0] === "`" ? "`" : "~"}{${fence[1].length},}\\s*$`);
+ const body = [line];
+ while (++i < lines.length) {
+ body.push(lines[i]);
+ if (close.test(lines[i])) break;
+ }
+ out.push({ kind: "block", md: body.join("\n"), code: true });
+ continue;
+ }
+ const heading = HEADING_RE.exec(line);
+ if (heading) {
+ flush();
+ out.push({ kind: "heading", level: heading[1].length, text: heading[2] });
+ continue;
+ }
+ if (!/\S/.test(line) || RULE_RE.test(line)) {
+ flush();
+ continue;
+ }
+ if (LIST_ITEM_RE.test(line)) {
+ flush();
+ cur.push(line);
+ continue;
+ }
+ const quote = QUOTE_LINE_RE.test(line);
+ if (quote && cur.length && !curQuote) flush();
+ if (quote) curQuote = true;
+ cur.push(line);
+ }
+ flush();
+ return out;
+}
+
+// Markdown as the plain text a claim is: no list or quote markers, no
+// emphasis or code ticks, a link as its label, one line.
+export function plainText(md: string): string {
+ return md
+ .split("\n")
+ .map((l) => l.replace(LIST_ITEM_RE, "").replace(/^ {0,3}(>\s?)+/, ""))
+ .join(" ")
+ .replace(LINK_RE, (_m, label: string) => label)
+ .replace(/(\*\*|__|\*|_|`)(?=\S)([\s\S]*?\S)\1/g, "$2")
+ .replace(/\s+/g, " ")
+ .trim();
+}
+
+// The quoted words in a stretch of text: every “…” or "…" run, joined by an
+// ellipsis (a report quotes one passage as `"A" … "B"`).
+export function quotedIn(text: string): string | null {
+ const runs = [...text.matchAll(/“([^”]+)”|"([^"]+)"/g)]
+ .map((m) => (m[1] ?? m[2]).trim())
+ .filter((q) => wordCount(q) > 0);
+ return runs.length ? runs.join(" … ") : null;
+}
+
+const TRIM_SEP_RE = /^[\s—–\-:;,|]+|[\s—–\-:;,|(]+$/g;
+
+// What a citing link resolves to: the id of the citation it now names, or null
+// to leave the link as it is.
+export type CiteResolver = (link: {
+ href: string;
+ label: string;
+ // The quoted words nearest before the link in its block, else in the block
+ // before it (a quote line followed by its citation line), else the block's
+ // plain text.
+ quote: string;
+}) => Promise<string | null>;
+
+export type StructuredMarkdown = {
+ title: string | null;
+ summary: string;
+ sections: Section[];
+};
+
+type Blk = { md: string; text: string; ids: string[]; interleaved: boolean };
+
+function stripMarkers(md: string): string {
+ const lines = md.split("\n");
+ const allQuoted = lines.every((l) => QUOTE_LINE_RE.test(l) || !/\S/.test(l));
+ return lines
+ .map((l, i) => {
+ let s = i === 0 ? l.replace(LIST_ITEM_RE, "") : l.replace(/^ {2,4}/, "");
+ if (allQuoted) s = s.replace(/^ {0,3}>\s?/, "");
+ return s;
+ })
+ .join("\n")
+ .trim();
+}
+
+// One block with its citing links rewritten. A link inside a code span is
+// text, as it is to lib/citations/inline.ts.
+async function rewriteBlock(md: string, previousText: string, resolve: CiteResolver): Promise<Blk> {
+ const scan = md.replace(CODE_SPAN_RE, (s) => s.replace(/[^\n]/g, " "));
+ const links = [...scan.matchAll(LINK_RE)];
+ const ownText = plainText(md.replace(LINK_RE, "")).replace(TRIM_SEP_RE, "");
+ const ids: string[] = [];
+ let out = "";
+ let last = 0;
+ const withoutLinks: string[] = [];
+ let between = 0;
+ for (const m of links) {
+ const start = m.index;
+ const end = start + m[0].length;
+ const label = md.slice(start + 1, start + 1 + m[1].length);
+ const href = m[2];
+ const before = plainText(md.slice(last, start));
+ const quote =
+ quotedIn(md.slice(last, start)) ??
+ quotedIn(md.slice(0, start)) ??
+ quotedIn(md.slice(end)) ??
+ (ownText ||
+ quotedIn(previousText) ||
+ previousText.replace(TRIM_SEP_RE, "") ||
+ label);
+ const id = await resolve({ href, label, quote });
+ out += md.slice(last, start);
+ withoutLinks.push(md.slice(last, start));
+ if (id) {
+ if (ids.length && before.replace(TRIM_SEP_RE, "")) between += 1;
+ ids.push(id);
+ out += `[${label}](${citeHref(id)})`;
+ } else {
+ out += md.slice(start, end);
+ withoutLinks.push(label);
+ }
+ last = end;
+ }
+ out += md.slice(last);
+ withoutLinks.push(md.slice(last));
+ const text = plainText(withoutLinks.join("")).replace(TRIM_SEP_RE, "");
+ return { md: out, text, ids: [...new Set(ids)], interleaved: between > 0 };
+}
+
+// The engine. `sectionLevel` headings (the shallowest below the title) open
+// sections; the first `#` heading, when it comes before any section, is the
+// title; deeper headings stay in the body. In a section, a block that cites
+// becomes a claim — a block that is ONLY citations (a `— [title @ 1:02](…)`
+// line under a quote) lends them to the block before it, which becomes the
+// claim — and every other block joins the section's body. Before the first
+// section, everything is the summary (its links rewritten, no claims). With no
+// section headings at all, everything after the title is one section,
+// `defaultSection`.
+export async function structureMarkdown(
+ md: string,
+ resolve: CiteResolver,
+ { defaultSection }: { defaultSection: string },
+): Promise<StructuredMarkdown> {
+ const parts = partsOf(md);
+ let title: string | null = null;
+ const firstHeading = parts.findIndex((p) => p.kind === "heading");
+ const firstBlock = parts.findIndex((p) => p.kind === "block");
+ if (firstHeading >= 0 && (parts[firstHeading] as MdHeading).level === 1 && (firstBlock < 0 || firstHeading < firstBlock)) {
+ title = plainText((parts[firstHeading] as MdHeading).text);
+ parts.splice(firstHeading, 1);
+ }
+ const levels = parts.filter((p): p is MdHeading => p.kind === "heading").map((p) => p.level);
+ const sectionLevel = levels.length ? Math.min(...levels) : null;
+
+ const anchors = new IdAllocator(/[^a-z0-9-]+/g);
+ const summary: string[] = [];
+ const sections: { section: Section; body: string[] }[] = [];
+ const open = (heading: string) => {
+ const id = anchors.claim(slugOf(heading) || "section", "section");
+ sections.push({ section: { id, title: heading, claims: [] }, body: [] });
+ };
+ if (sectionLevel === null) open(defaultSection);
+
+ // The block before, while it is still a candidate to take a citation-only
+ // block's citations: its rewritten markdown, its text, and where it went.
+ let prev: { blk: Blk; into: "body" | "claim"; at: number } | null = null;
+ let prevText = "";
+
+ for (const part of parts) {
+ if (part.kind === "heading") {
+ prev = null;
+ prevText = "";
+ if (part.level === sectionLevel) {
+ open(plainText(part.text));
+ continue;
+ }
+ const line = `${"#".repeat(part.level)} ${part.text}`;
+ if (sections.length) sections[sections.length - 1].body.push(line);
+ else summary.push(line);
+ continue;
+ }
+ if (part.code) {
+ if (sections.length) sections[sections.length - 1].body.push(part.md);
+ else summary.push(part.md);
+ prev = null;
+ prevText = "";
+ continue;
+ }
+ const blk = await rewriteBlock(part.md, prevText, resolve);
+ prevText = plainText(part.md.replace(LINK_RE, ""));
+ if (!sections.length) {
+ summary.push(blk.md);
+ continue;
+ }
+ const cur = sections[sections.length - 1];
+ const claims = cur.section.claims!;
+ if (!blk.ids.length) {
+ cur.body.push(blk.md);
+ prev = { blk, into: "body", at: cur.body.length - 1 };
+ continue;
+ }
+ if (!blk.text && prev) {
+ // Citations alone: they cite the block before.
+ if (prev.into === "body") {
+ cur.body.splice(prev.at, 1);
+ claims.push(claimOf(anchors, { ...prev.blk, md: `${prev.blk.md}\n${blk.md}`, ids: blk.ids, interleaved: false }));
+ } else {
+ const claim = claims[prev.at];
+ claim.citations = [...new Set([...(claim.citations ?? []), ...blk.ids])];
+ }
+ prev = null;
+ continue;
+ }
+ claims.push(claimOf(anchors, blk));
+ prev = { blk, into: "claim", at: claims.length - 1 };
+ }
+
+ return {
+ title,
+ summary: summary.join("\n\n").trim(),
+ sections: sections.map(({ section, body }) => {
+ const out: Section = { id: section.id, title: section.title };
+ const text = body.join("\n\n").trim();
+ if (text) out.body = text;
+ if (section.claims!.length) out.claims = section.claims;
+ return out;
+ }),
+ };
+}
+
+function claimOf(anchors: IdAllocator, blk: Blk): Claim {
+ const claim: Claim = { id: anchors.next("k"), text: blk.text || plainText(blk.md), citations: blk.ids };
+ if (blk.interleaved) claim.findings = stripMarkers(blk.md);
+ return claim;
+}
+
+// ─── The finished report ───
+
+export function emptyReport(id: string, kind: Report["kind"], title: string): Report {
+ return { format: REPORT_FORMAT, version: REPORT_VERSION, id, kind, title, sections: [] };
+}
+
+// The report with empty optional maps left out, checked.
+export function finish(report: Report, warnings: string[]): Converted {
+ const out: Report = { ...report };
+ if (out.citations && Object.keys(out.citations).length === 0) delete out.citations;
+ if (out.sources && Object.keys(out.sources).length === 0) delete out.sources;
+ if (out.summary !== undefined && !/\S/.test(out.summary)) delete out.summary;
+ const parsed = parseReport(out);
+ return { report: out, warnings, problems: parsed.problems };
+}
+
+// Every `cite:` id a markdown names, in order.
+export function citedIds(md: string | undefined): string[] {
+ return extractCiteRefs(md).map((r) => r.id);
+}
+
+// A citation map entry, typed for the converters that build one.
+export type CitationMap = Record<string, Citation>;
+
+// ─── The citations a converter collects ───
+
+const CLOCK_SUFFIX_RE = /\s*@\s*\d{1,2}(?::\d{2}){1,2}\s*$/;
+const POST_LABEL_DATE_RE = /,\s*(\d{4}-\d{2}-\d{2}(?:T[^\s,]*)?)\s*$/;
+
+// A citing link's label as the citation's display name: one line, without the
+// `@ mm:ss` the moment already carries.
+export function labelOf(label: string): string | undefined {
+ const s = plainText(label).replace(CLOCK_SUFFIX_RE, "").trim();
+ return s || undefined;
+}
+
+// The citations one document cites, each once: a span by its record and
+// second, a post by its channel and id. Ids are `c01`, `c02`, … for spans and
+// `p01`, … for posts, in order of first citing.
+export class CitationRegistry {
+ readonly citations: CitationMap = {};
+ private readonly ids = new IdAllocator();
+ private readonly byKey = new Map<string, string>();
+ private readonly warnedCues = new Set<string>();
+
+ constructor(
+ private readonly ctx: ConvertContext,
+ private readonly warnings: string[],
+ ) {}
+
+ async span(
+ channel: string,
+ id: string,
+ seconds: number,
+ quote: string,
+ opts: { label?: string; end?: number; kind?: "video" | "audio"; date?: string } = {},
+ ): Promise<string> {
+ const key = `span ${channel}/${id} ${seconds} ${opts.end ?? ""}`;
+ const known = this.byKey.get(key);
+ if (known) return known;
+ const span = await spanAt(channel, id, seconds, quote, this.ctx, opts.end);
+ if (span.from === "default" && this.ctx.cuesOf && !this.warnedCues.has(`${channel}/${id}`)) {
+ this.warnedCues.add(`${channel}/${id}`);
+ this.warnings.push(
+ `${channel}/${id}: no cues to widen from — its citations end ${this.ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS} s after the cited second`,
+ );
+ }
+ const cid = this.ids.next("c");
+ this.citations[cid] = {
+ kind: opts.kind ?? "video",
+ channel,
+ id,
+ start: span.start,
+ end: span.end,
+ quote,
+ ...(opts.label ? { label: opts.label } : {}),
+ ...(opts.date ? { date: opts.date } : {}),
+ };
+ this.byKey.set(key, cid);
+ return cid;
+ }
+
+ post(channel: string, id: string, quote: string, opts: { label?: string; date?: string } = {}): string {
+ const key = `post ${channel}/${id}`;
+ const known = this.byKey.get(key);
+ if (known) return known;
+ const cid = this.ids.next("p");
+ this.citations[cid] = {
+ kind: "post",
+ channel,
+ id,
+ quote,
+ ...(opts.label ? { label: opts.label } : {}),
+ ...(opts.date ? { date: opts.date } : {}),
+ };
+ this.byKey.set(key, cid);
+ return cid;
+ }
+
+ // A citing link (an archive moment or post, a moment page, a post on its
+ // platform) as a citation's id; null, with a warning when it looked like a
+ // citation, for a link to leave as it is.
+ async link(href: string, label: string, quote: string): Promise<string | null> {
+ const parsed = parseCitationHref(href);
+ if (!parsed) return null;
+ if (parsed.kind === "span") {
+ return this.span(parsed.channel, parsed.id, parsed.seconds, quote, { label: labelOf(label), end: parsed.end });
+ }
+ const date = POST_LABEL_DATE_RE.exec(label)?.[1];
+ const postOpts = { label: labelOf(label), ...(date && isPartialDate(date) ? { date } : {}) };
+ if (parsed.kind === "post") return this.post(parsed.channel, parsed.id, quote, postOpts);
+ const channel = this.ctx.postChannelOf
+ ? await this.ctx.postChannelOf({ platform: parsed.platform, id: parsed.id, handle: parsed.handle })
+ : null;
+ if (!channel) {
+ this.warnings.push(
+ `${href}: no archive channel keeps this post${this.ctx.postChannelOf ? "" : " (give a channels dir to look it up)"} — left as a plain link`,
+ );
+ return null;
+ }
+ return this.post(channel, parsed.id, quote, postOpts);
+ }
+}
diff --git a/common/lib/report/convertSweep.ts b/common/lib/report/convertSweep.ts
@@ -0,0 +1,52 @@
+// A /sweep REPORT (markdown) → `report.json` (kind "sweep").
+//
+// What /sweep writes (mcp/src/instructions.ts): `## sections` of findings,
+// each cited as `[title @ mm:ss](<origin>/?v=<channel>%2F<id>&t=<seconds>)`
+// for a video and `[post by <author>, <date>](<url>)` for a post — the url an
+// archive post (`…&vm=post`) or the post on its platform. The engine
+// (./convertShared.ts `structureMarkdown`) makes the document's first `#`
+// heading the title, the text before the first section the summary, each
+// section a section, and each list item or paragraph that cites a claim; every
+// citing link becomes `[label](cite:<id>)`.
+//
+// A citation's quote is the quoted words nearest before its link (`"…"` or
+// `“…”`, several joined by an ellipsis), else the claim's text — VERBATIM is
+// the model's rule, and compose checks it against the cues. Its span is the
+// cited second widened through the record's cues when the caller can read
+// them (`ctx.cuesOf`), else the second plus `ctx.spanSeconds`. A post on its
+// platform is cited only when `ctx.postChannelOf` finds the archive channel
+// that keeps it; otherwise its link stays a plain link, with a warning.
+
+import {
+ CitationRegistry,
+ emptyReport,
+ finish,
+ reportIdOf,
+ structureMarkdown,
+ type ConvertContext,
+ type Converted,
+} from "./convertShared";
+
+export type SweepConvertOptions = ConvertContext & {
+ // The report's id; default: the title's slug.
+ id?: string;
+ // The report's title; default: the markdown's first `#` heading.
+ title?: string;
+ // The one section of a sweep with no section headings.
+ sectionTitle?: string;
+};
+
+export async function sweepToReport(md: string, opts: SweepConvertOptions = {}): Promise<Converted> {
+ const warnings: string[] = [];
+ const registry = new CitationRegistry(opts, warnings);
+ const doc = await structureMarkdown(md, (link) => registry.link(link.href, link.label, link.quote), {
+ defaultSection: opts.sectionTitle ?? "Findings",
+ });
+ const title = opts.title ?? doc.title ?? "Sweep";
+ const report = emptyReport(reportIdOf(opts.id, title), "sweep", title);
+ if (doc.summary) report.summary = doc.summary;
+ report.citations = registry.citations;
+ report.sections = doc.sections;
+ if (!Object.keys(registry.citations).length) warnings.push("the markdown cites nothing the citation model knows");
+ return finish(report, warnings);
+}