// THE CONVERTERS' SHARED HALF — what bringing a cited document from elsewhere // into the citation model takes, whatever the document was: // // - reading a link a report cites with (`parseCitationHref`): an archive // viewer moment (`/?v=%2F&t=`, what /sweep and // /ask write), an archive post (`…&vm=post`), a moment page // (`/m//`), or a post on its own platform (x.com, bsky.app); // - turning ONE cited second into a span (`spanAt`): a record's cues, when // the caller can read them, widened to whole sentences by the one widening // (lib/cueWiden.mjs); else the second plus a default span; // - the markdown engine (`structureMarkdown`): a document's headings into // sections, its list items and paragraphs that carry a citation into // claims, everything else into the sections' bodies, and every citing link // rewritten to `[label](cite:)`; // - ids, and the finished report checked by the report validator. // // The converters themselves are ./convertSweep.ts, ./convertAsk.ts and // ./convertManifest.ts. Nothing here reads the disk: the cues and the channel // that keeps a post arrive as callbacks (./convert-server.ts has the disk // ones), so every converter runs in a test on literals. import { widen } from "../cueWiden.mjs"; import { MAX_CITATION_SPAN_SECONDS, REF_ID_RE, type Citation } from "../citations/schema"; import { citeHref, extractCiteRefs } from "../citations/inline"; import { parseMomentPath, roundMomentSeconds } from "../citations/moments"; import { isPartialDate, type Problem } from "../citations/validate"; import { REPORT_FORMAT, REPORT_ID_RE, REPORT_VERSION, type Claim, type Report, type Section } from "./schema"; import { parseReport } from "./validate"; export type Cue = { start: number; end: number; text: string }; // A record's cues, sorted by start, or null when they cannot be read. export type CuesOf = (channel: string, id: string) => Promise; export type PostPlatformName = "x" | "bluesky"; // The archive channel that keeps a post, or null. export type PostChannelOf = (post: { platform: PostPlatformName; id: string; handle?: string }) => Promise; export type ConvertContext = { cuesOf?: CuesOf; postChannelOf?: PostChannelOf; // A cited second with no cues to read becomes [second, second + this]. spanSeconds?: number; }; // What a converter hands back: the report, what it had to guess or leave out // (warnings), and the report validator's problems — a report with problems is // not to be written. export type Converted = { report: Report; warnings: string[]; problems: Problem[] }; // A cited second with no cues is this long. A /sweep or /ask citation carries // one second, never an end; ten seconds is about a spoken sentence. export const DEFAULT_SPAN_SECONDS = 10; // ─── Links ─── export type ParsedCitationHref = // An archive viewer moment, or a moment page: `seconds` is where it starts; // a moment page also carries its end. | { kind: "span"; channel: string; id: string; seconds: number; end?: number } // An archived post on the archive. | { kind: "post"; channel: string; id: string } // A post on its own platform: the archive channel is not in the link. | { kind: "post-original"; platform: PostPlatformName; id: string; handle?: string }; const X_HOSTS = new Set(["x.com", "twitter.com", "mobile.twitter.com", "www.x.com", "www.twitter.com", "mobile.x.com"]); const BSKY_HOSTS = new Set(["bsky.app", "www.bsky.app"]); // What a link cites, or null for a link that cites nothing the model knows (a // platform's video page, an article, a relative link). export function parseCitationHref(href: string): ParsedCitationHref | null { let u: URL; try { u = new URL(href.trim()); } catch { return null; } if (u.protocol !== "http:" && u.protocol !== "https:") return null; const host = u.hostname.toLowerCase(); if (X_HOSTS.has(host)) { const m = /^\/([^/]+)\/status(?:es)?\/(\d+)(?:\/|$)/.exec(u.pathname); if (!m) return null; return { kind: "post-original", platform: "x", id: m[2], ...(m[1] !== "i" ? { handle: m[1] } : {}) }; } if (BSKY_HOSTS.has(host)) { const m = /^\/profile\/([^/]+)\/post\/([A-Za-z0-9]+)\/?$/.exec(u.pathname); if (!m) return null; return { kind: "post-original", platform: "bluesky", id: m[2], handle: m[1] }; } const moment = parseMomentPath(u.pathname); if (moment) { return moment.kind === "span" ? { kind: "span", channel: moment.channel, id: moment.id, seconds: moment.start, end: moment.end } : { kind: "post", channel: moment.channel, id: moment.id }; } // The viewer: `v` is `/` (URL-encoded or not), `t` whole seconds. const v = u.searchParams.get("v"); if (!v) return null; const slash = v.indexOf("/"); if (slash <= 0 || slash === v.length - 1) return null; const channel = v.slice(0, slash); const id = v.slice(slash + 1); if (u.searchParams.get("vm") === "post") return { kind: "post", channel, id }; const t = u.searchParams.get("t"); const seconds = t && /^\d+(\.\d+)?$/.test(t) ? Number(t) : 0; return { kind: "span", channel, id, seconds }; } // ─── Spans ─── const EPS = 0.02; const wordCount = (s: string) => s.split(/\s+/).filter((w) => /[\p{L}\p{N}]/u.test(w)).length; const round2 = (n: number) => roundMomentSeconds(n); // The cue a cited second points at: the viewer's `t` is a cue's start, // floored, so the cue that starts in [t, t + 1) is the one cited; else the cue // that holds t; else the nearest one after it. function cueIndexAt(cues: readonly Cue[], t: number): number { const starting = cues.findIndex((c) => c.start >= t - EPS && c.start < t + 1); if (starting >= 0) return starting; const holding = cues.findIndex((c) => c.start <= t + EPS && c.end > t); if (holding >= 0) return holding; const after = cues.findIndex((c) => c.start > t); return after >= 0 ? after : cues.length - 1; } export type Span = { start: number; end: number; from: "cues" | "default" | "link" }; // One cited second as a span. With the record's cues: the cue it cites, run // on cue by cue until it holds as many words as the quote, then widened to // whole sentences — and if the widened span is longer than a citation may be, // the unwidened cue run, and failing that the default. Without cues (or past // their end): [second, second + spanSeconds]. An `end` the link already gave // (a moment page) is kept as it is. export async function spanAt( channel: string, id: string, seconds: number, quote: string, ctx: ConvertContext, end?: number, ): Promise { if (end !== undefined && end > seconds) return { start: round2(seconds), end: round2(end), from: "link" }; const fallback: Span = { start: round2(seconds), end: round2(seconds + (ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS)), from: "default", }; const cues = ctx.cuesOf ? await ctx.cuesOf(channel, id) : null; if (!cues || cues.length === 0 || seconds > cues[cues.length - 1].end) return fallback; const i = cueIndexAt(cues, seconds); const want = Math.max(1, wordCount(quote)); let j = i; let have = wordCount(cues[i].text); while (have < want && j + 1 < cues.length && cues[j + 1].end - cues[i].start <= MAX_CITATION_SPAN_SECONDS) { j += 1; have += wordCount(cues[j].text); } const fits = (s: number, e: number) => e > s && e - s <= MAX_CITATION_SPAN_SECONDS && round2(e) > round2(s); const w = widen(cues, cues[i].start, cues[j].end); if (fits(w.start, w.end)) return { start: round2(w.start), end: round2(w.end), from: "cues" }; if (fits(cues[i].start, cues[j].end)) return { start: round2(cues[i].start), end: round2(cues[j].end), from: "cues" }; return fallback; } // ─── Ids ─── // Unique ids in one namespace, each a reference id (lib/citations/schema.ts // REF_ID_RE): a wanted id is kept when it is free and sound, else made sound // and suffixed `-2`, `-3`, …. export class IdAllocator { private readonly taken = new Set(); private readonly counters = new Map(); constructor(private readonly charset: RegExp = /[^A-Za-z0-9_.:-]+/g) {} has(id: string): boolean { return this.taken.has(id); } claim(wanted: string, fallback = "x"): string { let base = wanted.replace(this.charset, "-").replace(/^[^A-Za-z0-9]+/, "").slice(0, 56); if (!base) base = fallback; let id = base; for (let n = 2; this.taken.has(id) || !REF_ID_RE.test(id); n++) id = `${base}-${n}`; this.taken.add(id); return id; } // The next free `01`, `02`, …. next(prefix: string): string { let n = this.counters.get(prefix) ?? 0; let id: string; do { n += 1; id = `${prefix}${String(n).padStart(2, "0")}`; } while (this.taken.has(id)); this.counters.set(prefix, n); this.taken.add(id); return id; } } // A heading or a title as an id: lowercase words joined by `-`. export function slugOf(text: string, max = 48): string { return text .normalize("NFKD") .replace(/[̀-ͯ]/g, "") .toLowerCase() .replace(/[^a-z0-9]+/g, "-") .replace(/^-+|-+$/g, "") .slice(0, max) .replace(/-+$/, ""); } // A report id (a lowercase slug, REPORT_ID_RE) from what the caller asked for, // else from the title. export function reportIdOf(wanted: string | undefined, title: string): string { if (wanted !== undefined) return wanted; const slug = slugOf(title, 64); return REPORT_ID_RE.test(slug) ? slug : "report"; } // ─── Markdown ─── // `[label](href)`: a label may hold one level of brackets (a title with // `[live]` in it); the href runs to the first space or `)`, optionally in // `<…>`, optionally followed by a quoted title. const LINK_RE = /\[((?:[^[\]]|\[[^[\]]*\])*)\]\(\s*]+)>?(?:\s+"[^"]*")?\s*\)/g; const CODE_SPAN_RE = /(?/; const RULE_RE = /^ {0,3}([-*_])(\s*\1){2,}\s*$/; type MdHeading = { kind: "heading"; level: number; text: string }; type MdBlock = { kind: "block"; md: string; code: boolean }; type MdPart = MdHeading | MdBlock; // The markdown as headings and blocks: a fenced block, a top-level list item // (with its continuation and nested lines), a run of quote lines (and the // lines that lazily continue it), or a paragraph. Blank lines and rules // separate them. function partsOf(md: string): MdPart[] { const out: MdPart[] = []; const lines = md.replace(/\r\n?/g, "\n").split("\n"); let cur: string[] = []; let curQuote = false; const flush = () => { if (cur.length) out.push({ kind: "block", md: cur.join("\n"), code: false }); cur = []; curQuote = false; }; for (let i = 0; i < lines.length; i++) { const line = lines[i]; const fence = FENCE_RE.exec(line); if (fence) { flush(); const close = new RegExp(`^ {0,3}${fence[1][0] === "`" ? "`" : "~"}{${fence[1].length},}\\s*$`); const body = [line]; while (++i < lines.length) { body.push(lines[i]); if (close.test(lines[i])) break; } out.push({ kind: "block", md: body.join("\n"), code: true }); continue; } const heading = HEADING_RE.exec(line); if (heading) { flush(); out.push({ kind: "heading", level: heading[1].length, text: heading[2] }); continue; } if (!/\S/.test(line) || RULE_RE.test(line)) { flush(); continue; } if (LIST_ITEM_RE.test(line)) { flush(); cur.push(line); continue; } const quote = QUOTE_LINE_RE.test(line); if (quote && cur.length && !curQuote) flush(); if (quote) curQuote = true; cur.push(line); } flush(); return out; } // Markdown as the plain text a claim is: no list or quote markers, no // emphasis or code ticks, a link as its label, one line. export function plainText(md: string): string { return md .split("\n") .map((l) => l.replace(LIST_ITEM_RE, "").replace(/^ {0,3}(>\s?)+/, "")) .join(" ") .replace(LINK_RE, (_m, label: string) => label) .replace(/(\*\*|__|\*|_|`)(?=\S)([\s\S]*?\S)\1/g, "$2") .replace(/\s+/g, " ") .trim(); } // The quoted words in a stretch of text: every “…” or "…" run, joined by an // ellipsis (a report quotes one passage as `"A" … "B"`). export function quotedIn(text: string): string | null { const runs = [...text.matchAll(/“([^”]+)”|"([^"]+)"/g)] .map((m) => (m[1] ?? m[2]).trim()) .filter((q) => wordCount(q) > 0); return runs.length ? runs.join(" … ") : null; } const TRIM_SEP_RE = /^[\s—–\-:;,|]+|[\s—–\-:;,|(]+$/g; // What a citing link resolves to: the id of the citation it now names, or null // to leave the link as it is. export type CiteResolver = (link: { href: string; label: string; // The quoted words nearest before the link in its block, else in the block // before it (a quote line followed by its citation line), else the block's // plain text. quote: string; }) => Promise; export type StructuredMarkdown = { title: string | null; summary: string; sections: Section[]; }; type Blk = { md: string; text: string; ids: string[]; interleaved: boolean }; function stripMarkers(md: string): string { const lines = md.split("\n"); const allQuoted = lines.every((l) => QUOTE_LINE_RE.test(l) || !/\S/.test(l)); return lines .map((l, i) => { let s = i === 0 ? l.replace(LIST_ITEM_RE, "") : l.replace(/^ {2,4}/, ""); if (allQuoted) s = s.replace(/^ {0,3}>\s?/, ""); return s; }) .join("\n") .trim(); } // One block with its citing links rewritten. A link inside a code span is // text, as it is to lib/citations/inline.ts. async function rewriteBlock(md: string, previousText: string, resolve: CiteResolver): Promise { const scan = md.replace(CODE_SPAN_RE, (s) => s.replace(/[^\n]/g, " ")); const links = [...scan.matchAll(LINK_RE)]; const ownText = plainText(md.replace(LINK_RE, "")).replace(TRIM_SEP_RE, ""); const ids: string[] = []; let out = ""; let last = 0; const withoutLinks: string[] = []; let between = 0; for (const m of links) { const start = m.index; const end = start + m[0].length; const label = md.slice(start + 1, start + 1 + m[1].length); const href = m[2]; const before = plainText(md.slice(last, start)); const quote = quotedIn(md.slice(last, start)) ?? quotedIn(md.slice(0, start)) ?? quotedIn(md.slice(end)) ?? (ownText || quotedIn(previousText) || previousText.replace(TRIM_SEP_RE, "") || label); const id = await resolve({ href, label, quote }); out += md.slice(last, start); withoutLinks.push(md.slice(last, start)); if (id) { if (ids.length && before.replace(TRIM_SEP_RE, "")) between += 1; ids.push(id); out += `[${label}](${citeHref(id)})`; } else { out += md.slice(start, end); withoutLinks.push(label); } last = end; } out += md.slice(last); withoutLinks.push(md.slice(last)); const text = plainText(withoutLinks.join("")).replace(/\s+([.,;:!?])/g, "$1").replace(TRIM_SEP_RE, ""); return { md: out, text, ids: [...new Set(ids)], interleaved: between > 0 }; } // The engine. `sectionLevel` headings (the shallowest below the title) open // sections; the first `#` heading, when it comes before any section, is the // title; deeper headings stay in the body. In a section, a block that cites // becomes a claim — a block that is ONLY citations (a `— [title @ 1:02](…)` // line under a quote) lends them to the block before it, which becomes the // claim — and every other block joins the section's body. Before the first // section, everything is the summary (its links rewritten, no claims). With no // section headings at all, everything after the title is one section, // `defaultSection`. export async function structureMarkdown( md: string, resolve: CiteResolver, { defaultSection }: { defaultSection: string }, ): Promise { const parts = partsOf(md); let title: string | null = null; const firstHeading = parts.findIndex((p) => p.kind === "heading"); const firstBlock = parts.findIndex((p) => p.kind === "block"); if (firstHeading >= 0 && (parts[firstHeading] as MdHeading).level === 1 && (firstBlock < 0 || firstHeading < firstBlock)) { title = plainText((parts[firstHeading] as MdHeading).text); parts.splice(firstHeading, 1); } const levels = parts.filter((p): p is MdHeading => p.kind === "heading").map((p) => p.level); const sectionLevel = levels.length ? Math.min(...levels) : null; const anchors = new IdAllocator(/[^a-z0-9-]+/g); const summary: string[] = []; const sections: { section: Section; body: string[] }[] = []; const open = (heading: string) => { const id = anchors.claim(slugOf(heading) || "section", "section"); sections.push({ section: { id, title: heading, claims: [] }, body: [] }); }; if (sectionLevel === null) open(defaultSection); // The block before, while it is still a candidate to take a citation-only // block's citations: its rewritten markdown, its text, and where it went. let prev: { blk: Blk; into: "body" | "claim"; at: number } | null = null; let prevText = ""; for (const part of parts) { if (part.kind === "heading") { prev = null; prevText = ""; if (part.level === sectionLevel) { open(plainText(part.text)); continue; } const line = `${"#".repeat(part.level)} ${part.text}`; if (sections.length) sections[sections.length - 1].body.push(line); else summary.push(line); continue; } if (part.code) { if (sections.length) sections[sections.length - 1].body.push(part.md); else summary.push(part.md); prev = null; prevText = ""; continue; } const blk = await rewriteBlock(part.md, prevText, resolve); prevText = plainText(part.md.replace(LINK_RE, "")); if (!sections.length) { summary.push(blk.md); continue; } const cur = sections[sections.length - 1]; const claims = cur.section.claims!; if (!blk.ids.length) { cur.body.push(blk.md); prev = { blk, into: "body", at: cur.body.length - 1 }; continue; } if (!blk.text && prev) { // Citations alone: they cite the block before. if (prev.into === "body") { cur.body.splice(prev.at, 1); claims.push(claimOf(anchors, { ...prev.blk, md: `${prev.blk.md}\n${blk.md}`, ids: blk.ids, interleaved: false })); } else { const claim = claims[prev.at]; claim.citations = [...new Set([...(claim.citations ?? []), ...blk.ids])]; } prev = null; continue; } claims.push(claimOf(anchors, blk)); prev = { blk, into: "claim", at: claims.length - 1 }; } return { title, summary: summary.join("\n\n").trim(), sections: sections.map(({ section, body }) => { const out: Section = { id: section.id, title: section.title }; const text = body.join("\n\n").trim(); if (text) out.body = text; if (section.claims!.length) out.claims = section.claims; return out; }), }; } function claimOf(anchors: IdAllocator, blk: Blk): Claim { const claim: Claim = { id: anchors.next("k"), text: blk.text || plainText(blk.md), citations: blk.ids }; if (blk.interleaved) claim.findings = stripMarkers(blk.md); return claim; } // ─── The finished report ─── export function emptyReport(id: string, kind: Report["kind"], title: string): Report { return { format: REPORT_FORMAT, version: REPORT_VERSION, id, kind, title, sections: [] }; } // The report with empty optional maps left out, checked. export function finish(report: Report, warnings: string[]): Converted { const out: Report = { ...report }; if (out.citations && Object.keys(out.citations).length === 0) delete out.citations; if (out.sources && Object.keys(out.sources).length === 0) delete out.sources; if (out.summary !== undefined && !/\S/.test(out.summary)) delete out.summary; const parsed = parseReport(out); return { report: out, warnings, problems: parsed.problems }; } // Every `cite:` id a markdown names, in order. export function citedIds(md: string | undefined): string[] { return extractCiteRefs(md).map((r) => r.id); } // A citation map entry, typed for the converters that build one. export type CitationMap = Record; // ─── The citations a converter collects ─── const CLOCK_SUFFIX_RE = /\s*@\s*\d{1,2}(?::\d{2}){1,2}\s*$/; const POST_LABEL_DATE_RE = /,\s*(\d{4}-\d{2}-\d{2}(?:T[^\s,]*)?)\s*$/; // A citing link's label as the citation's display name: one line, without the // `@ mm:ss` the moment already carries. export function labelOf(label: string): string | undefined { const s = plainText(label).replace(CLOCK_SUFFIX_RE, "").trim(); return s || undefined; } // The citations one document cites, each once: a span by its record and // second, a post by its channel and id. Ids are `c01`, `c02`, … for spans and // `p01`, … for posts, in order of first citing. export class CitationRegistry { readonly citations: CitationMap = {}; private readonly ids = new IdAllocator(); private readonly byKey = new Map(); private readonly warnedCues = new Set(); constructor( private readonly ctx: ConvertContext, private readonly warnings: string[], ) {} async span( channel: string, id: string, seconds: number, quote: string, opts: { label?: string; end?: number; kind?: "video" | "audio"; date?: string } = {}, ): Promise { const key = `span ${channel}/${id} ${seconds} ${opts.end ?? ""}`; const known = this.byKey.get(key); if (known) return known; const span = await spanAt(channel, id, seconds, quote, this.ctx, opts.end); if (span.from === "default" && this.ctx.cuesOf && !this.warnedCues.has(`${channel}/${id}`)) { this.warnedCues.add(`${channel}/${id}`); this.warnings.push( `${channel}/${id}: no cues to widen from — its citations end ${this.ctx.spanSeconds ?? DEFAULT_SPAN_SECONDS} s after the cited second`, ); } const cid = this.ids.next("c"); this.citations[cid] = { kind: opts.kind ?? "video", channel, id, start: span.start, end: span.end, quote, ...(opts.label ? { label: opts.label } : {}), ...(opts.date ? { date: opts.date } : {}), }; this.byKey.set(key, cid); return cid; } post(channel: string, id: string, quote: string, opts: { label?: string; date?: string } = {}): string { const key = `post ${channel}/${id}`; const known = this.byKey.get(key); if (known) return known; const cid = this.ids.next("p"); this.citations[cid] = { kind: "post", channel, id, quote, ...(opts.label ? { label: opts.label } : {}), ...(opts.date ? { date: opts.date } : {}), }; this.byKey.set(key, cid); return cid; } // A citing link (an archive moment or post, a moment page, a post on its // platform) as a citation's id; null, with a warning when it looked like a // citation, for a link to leave as it is. async link(href: string, label: string, quote: string): Promise { const parsed = parseCitationHref(href); if (!parsed) return null; if (parsed.kind === "span") { return this.span(parsed.channel, parsed.id, parsed.seconds, quote, { label: labelOf(label), end: parsed.end }); } const date = POST_LABEL_DATE_RE.exec(label)?.[1]; const postOpts = { label: labelOf(label), ...(date && isPartialDate(date) ? { date } : {}) }; if (parsed.kind === "post") return this.post(parsed.channel, parsed.id, quote, postOpts); const channel = this.ctx.postChannelOf ? await this.ctx.postChannelOf({ platform: parsed.platform, id: parsed.id, handle: parsed.handle }) : null; if (!channel) { this.warnings.push( `${href}: no archive channel keeps this post${this.ctx.postChannelOf ? "" : " (give a channels dir to look it up)"} — left as a plain link`, ); return null; } return this.post(channel, parsed.id, quote, postOpts); } }