commit 1362716c5c8b15293d2542cf549602f5c2c151ec
parent d7f670b2677fbaa04f4b37396cc920e3968cc808
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 04:41:34 -0400
Merge report-r4b-page-size (a report page holds each source once, archive links once (collapsed) plus per-claim matches, one shared citation map for inline previews; a size test)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
15 files changed, 580 insertions(+), 223 deletions(-)
diff --git a/common/components/Markdown.tsx b/common/components/Markdown.tsx
@@ -73,22 +73,18 @@ const OVERRIDES = {
// spacing is consistent regardless of content shape. `linkComponent` optionally
// overrides how `<a>` renders (e.g. to make in-page citation anchors scroll
// instead of opening a new tab); when omitted, links use the default external
-// styling above. `linkProps` are extra props every `<a>` override receives
-// beside the link's own (e.g. a citation map), so the link component can stay
-// one stable component instead of a closure made per render.
-export function Markdown<P extends object = object>({
+// styling above.
+export function Markdown({
children,
className,
linkComponent,
- linkProps,
}: {
children: string;
className?: string;
- linkComponent?: ComponentType<AnchorHTMLAttributes<HTMLAnchorElement> & P>;
- linkProps?: P;
+ linkComponent?: ComponentType<AnchorHTMLAttributes<HTMLAnchorElement>>;
}) {
const overrides = linkComponent
- ? { ...OVERRIDES, a: { component: linkComponent, ...(linkProps ? { props: linkProps } : {}) } }
+ ? { ...OVERRIDES, a: { component: linkComponent } }
: OVERRIDES;
return (
<div className={className}>
diff --git a/common/components/citations/CitationCard.tsx b/common/components/citations/CitationCard.tsx
@@ -2,6 +2,7 @@ import { Archive, ExternalLink, FileText, Globe, MessageSquareQuote, Mic, Play,
import type { CitationKind } from "../../lib/citations/schema";
import {
CITATION_KIND_LABELS,
+ sourceAnchor,
spanLabel,
type CitationView,
type PageCitationView,
@@ -20,8 +21,9 @@ import { VerificationBadge } from "./VerificationBadge";
// a play affordance that opens the moment page (its clip)
// post the author, date, quote (and the post's text when it says
// more), the captured shot when the view has one
-// source the document's sentence: its still, the quote, and the
-// document's archive links in context
+// source the document's sentence: its still, the quote, and
+// "from <title>" linking to the document's block on the
+// page (sourceAnchor), where its archive links are — once
// page the page's title and URL, its archived copy
//
// Three densities: `card` (a claim's evidence list), `preview` (InlineCite's
@@ -80,8 +82,8 @@ function heading(c: CitationView): { title: string; meta: string[] } {
};
case "source":
return {
- title: c.source.title,
- meta: [c.source.publisher, c.source.author, c.date ?? c.source.date].filter((x): x is string => !!x),
+ title: c.sourceTitle,
+ meta: [c.date].filter((x): x is string => !!x),
};
case "page":
return { title: c.title ?? hostOf(c.href), meta: [hostOf(c.href), c.date].filter((x): x is string => !!x) };
@@ -119,7 +121,7 @@ function Picture({ c, variant }: { c: CitationView; variant: CitationCardVariant
return (
<img
src={c.image}
- alt={`The sentence as ${c.source.title} shows it: ${c.quote}`}
+ alt={`The sentence as ${c.sourceTitle} shows it: ${c.quote}`}
loading="lazy"
className={cn("w-full rounded-md border border-border bg-white object-contain object-left-top", size)}
/>
@@ -158,15 +160,9 @@ function PostLinks({ c }: { c: PostCitationView }) {
function SourceLinks({ c }: { c: SourceCitationView }) {
return (
- <>
- {c.href && <ExternalA href={c.href}>{c.source.publisher ?? hostOf(c.href)}</ExternalA>}
- {c.source.archives.map((a) => (
- <ExternalA key={a.url} href={a.url} title={a.context}>
- <Archive className="size-3 shrink-0" aria-hidden />
- {a.label}
- </ExternalA>
- ))}
- </>
+ <a href={`#${sourceAnchor(c.sourceId)}`} className={linkClass}>
+ from {c.sourceTitle}
+ </a>
);
}
diff --git a/common/components/citations/CitationsContext.tsx b/common/components/citations/CitationsContext.tsx
@@ -0,0 +1,31 @@
+"use client";
+
+import { createContext, useContext, type ReactNode } from "react";
+import type { CitationView } from "../../lib/report/views";
+
+// A PAGE'S CITATIONS, ONCE. Every inline marker (InlineCite) on a report page
+// needs its citation's view — the label's href, the number, the preview card.
+// Handing each marker its own copy as a prop would serialize that view into the
+// page's RSC payload once per marker; a page with hundreds of markers paid for
+// it hundreds of times. Instead the page wraps its content in one
+// CitationsProvider, whose map is serialized once, and each marker carries only
+// its id and reads the view here at render time (on the server for the HTML,
+// in the browser for the preview).
+
+const CitationsContext = createContext<Readonly<Record<string, CitationView>> | null>(null);
+
+export function CitationsProvider({
+ citations,
+ children,
+}: {
+ citations: Readonly<Record<string, CitationView>>;
+ children: ReactNode;
+}) {
+ return <CitationsContext.Provider value={citations}>{children}</CitationsContext.Provider>;
+}
+
+// The view of citation `id` from the nearest provider, or undefined.
+export function useCitation(id: string | undefined): CitationView | undefined {
+ const map = useContext(CitationsContext);
+ return id !== undefined && map && Object.hasOwn(map, id) ? map[id] : undefined;
+}
diff --git a/common/components/citations/CitedMarkdown.tsx b/common/components/citations/CitedMarkdown.tsx
@@ -2,16 +2,19 @@ import type { AnchorHTMLAttributes } from "react";
import { CITE_SCHEME } from "../../lib/citations/inline";
import type { CitationView } from "../../lib/report/views";
import { Markdown } from "../Markdown";
+import { CitationsProvider } from "./CitationsContext";
import { InlineCite } from "./InlineCite";
// Report markdown with its citations live: every `[label](cite:<id>)` link
-// (lib/citations/inline.ts) becomes an InlineCite for the citation's view;
-// every other link renders as Markdown's own external link. The house markdown
-// setup (Markdown.tsx — markdown-to-jsx, raw HTML escaped, token-styled) is
+// (lib/citations/inline.ts) becomes an InlineCite for that id; every other
+// link renders as Markdown's own external link. The house markdown setup
+// (Markdown.tsx — markdown-to-jsx, raw HTML escaped, token-styled) is
// unchanged; this only swaps its link component. A `cite:` link inside code is
// code, never a citation: markdown-to-jsx makes no link there.
//
-// Pure: the citations come in as props, keyed by id.
+// The views come from the page's one CitationsProvider (a page with many
+// markdown blocks serializes its citation map once). A block rendered on its
+// own may pass `citations`, and gets a provider of its own.
const EXTERNAL_LINK_CLASS = "text-brand underline decoration-brand/40 hover:decoration-brand";
@@ -21,18 +24,10 @@ export function citeIdOf(href: string | undefined): string | null {
type Citations = Readonly<Record<string, CitationView>>;
-// The `<a>` of cited markdown: one stable component, the citations passed in
-// as a prop (Markdown `linkProps`).
-function CiteLink({
- href,
- children: label,
- citations,
- ...rest
-}: AnchorHTMLAttributes<HTMLAnchorElement> & { citations: Citations }) {
+// The `<a>` of cited markdown: one stable component.
+function CiteLink({ href, children: label, ...rest }: AnchorHTMLAttributes<HTMLAnchorElement>) {
const id = citeIdOf(href);
- if (id !== null) {
- return <InlineCite citation={Object.hasOwn(citations, id) ? citations[id] : undefined}>{label}</InlineCite>;
- }
+ if (id !== null) return <InlineCite id={id}>{label}</InlineCite>;
return (
<a {...rest} href={href} target="_blank" rel="noopener noreferrer" className={EXTERNAL_LINK_CLASS}>
{label}
@@ -46,12 +41,13 @@ export function CitedMarkdown({
className,
}: {
children: string;
- citations: Citations;
+ citations?: Citations;
className?: string;
}) {
- return (
- <Markdown className={className} linkComponent={CiteLink} linkProps={{ citations }}>
+ const md = (
+ <Markdown className={className} linkComponent={CiteLink}>
{children}
</Markdown>
);
+ return citations ? <CitationsProvider citations={citations}>{md}</CitationsProvider> : md;
}
diff --git a/common/components/citations/InlineCite.tsx b/common/components/citations/InlineCite.tsx
@@ -5,6 +5,7 @@ import { createPortal } from "react-dom";
import { citationAnchor } from "../../lib/citations/inline";
import { CITATION_KIND_LABELS, type CitationView } from "../../lib/report/views";
import { CitationCard } from "./CitationCard";
+import { useCitation } from "./CitationsContext";
// AN INLINE CITATION — what a `[label](cite:<id>)` link in report markdown
// renders as (CitedMarkdown): the label, linked to what it cites (the moment
@@ -26,8 +27,11 @@ import { CitationCard } from "./CitationCard";
// viewport's width. The number is the keyboard's way in; its reference entry
// carries the same card with every link.
//
-// A ref the page could not resolve (no view for the id — validation keeps a
-// published report from having one) renders its label alone.
+// THE VIEW comes from the page's CitationsProvider by `id` (one map for the
+// page, never a copy per marker in the RSC payload); a caller outside a
+// provider may pass the `citation` itself. A ref neither resolves (no view for
+// the id — validation keeps a published report from having one) renders its
+// label alone.
const CLOSE_DELAY_MS = 150;
const EDGE_PX = 8;
@@ -37,7 +41,17 @@ function isExternal(c: CitationView): boolean {
return c.kind === "source" || c.kind === "page";
}
-export function InlineCite({ citation, children }: { citation: CitationView | undefined; children: ReactNode }) {
+export function InlineCite({
+ citation: given,
+ id,
+ children,
+}: {
+ citation?: CitationView;
+ id?: string;
+ children: ReactNode;
+}) {
+ const fromMap = useCitation(given ? undefined : id);
+ const citation = given ?? fromMap;
const previewId = useId();
const [open, setOpen] = useState(false);
const [pos, setPos] = useState<{ top: number; left: number } | null>(null);
diff --git a/common/components/citations/citations.test.ts b/common/components/citations/citations.test.ts
@@ -5,6 +5,7 @@ import { renderToStaticMarkup } from "react-dom/server";
import type { CitationView, PageCitationView, PostCitationView, SourceCitationView, SpanCitationView } from "../../lib/report/views";
import { CitationCard } from "./CitationCard";
import { CitedMarkdown, citeIdOf } from "./CitedMarkdown";
+import { CitationsProvider } from "./CitationsContext";
import { InlineCite } from "./InlineCite";
import { ReferenceList } from "./ReferenceList";
import { QUOTE_SCORE_TOOLTIP, VerificationBadge, quoteScorePercent } from "./VerificationBadge";
@@ -70,14 +71,8 @@ const source: SourceCitationView = {
id: "a01",
number: 4,
quote: "He opened the bridge himself.",
- source: {
- id: "s0",
- kind: "article",
- title: "An example article",
- url: "https://example.org/a",
- publisher: "Example Gazette",
- archives: [{ label: "archive.org", url: "https://web.archive.org/x", context: "as published" }],
- },
+ sourceId: "s0",
+ sourceTitle: "An example article",
image: "/reports/demo/stills/a01.png",
href: "https://example.org/a",
};
@@ -122,11 +117,12 @@ test("a post card: author, date, quote, the post's fuller text, the shot", () =>
assert.match(html, /<img src="\/media\/posts\/demo-social\/123\/shot.png"/);
});
-test("a source card: the still, the quote, the document's archive links in context", () => {
+test("a source card: the still, the quote, and from-the-document — never its archive links", () => {
const html = render(h(CitationCard, { citation: source }));
assert.match(html, /<img src="\/reports\/demo\/stills\/a01.png" alt="The sentence as An example article shows it: He opened the bridge himself."/);
- assert.match(html, /href="https:\/\/web.archive.org\/x"[^>]*title="as published"/);
- assert.ok(html.includes("Example Gazette"));
+ assert.match(html, /<a href="#source-s0"[^>]*>from An example article<\/a>/);
+ assert.ok(!html.includes("archive"));
+ assert.ok(!render(h(CitationCard, { citation: source, variant: "preview" })).includes("archive"));
});
test("a page card: title, host, the archived copy, the note", () => {
@@ -170,6 +166,23 @@ test("an inline cite: the label links to what it cites, the number to its refere
assert.match(render(h(InlineCite, { citation: page, children: "records" })), /href="https:\/\/example.org\/city\/history" target="_blank" rel="noopener noreferrer"/);
// an unresolved ref is its label alone
assert.equal(render(h(InlineCite, { citation: undefined, children: "label" })), "label");
+ assert.equal(render(h(InlineCite, { id: "c01", children: "label" })), "label", "no provider, no view");
+});
+
+test("inline cites read their view from the page's one provider, by id", () => {
+ const html = render(
+ h(CitationsProvider, {
+ citations: byId,
+ children: [
+ h(InlineCite, { key: 1, id: "c01", children: "first" }),
+ h(InlineCite, { key: 2, id: "w01", children: "second" }),
+ h(InlineCite, { key: 3, id: "missing", children: "third" }),
+ ],
+ }),
+ );
+ assert.match(html, /data-inline-cite="c01"[^]*?>first<\/a>[^]*?data-cite-number="1"/);
+ assert.match(html, /data-inline-cite="w01"[^]*?data-cite-number="5"/);
+ assert.ok(html.endsWith("third"));
});
test("cited markdown: cite links become inline cites, other links stay links, code stays code", () => {
diff --git a/common/lib/report/views.test.ts b/common/lib/report/views.test.ts
@@ -8,13 +8,18 @@ import { validateReport } from "./validate";
import {
MOMENT_PLACEHOLDER_KEY,
REPORT_PLACEHOLDER_ID,
+ CLAIM_ARCHIVES_MAX,
+ CLAIM_ARCHIVE_MIN_OVERLAP,
+ archiveOverlap,
buildReportPageView,
+ claimArchives,
citedInViews,
evidenceClipPath,
momentViewPath,
orderedCitations,
reportAssetPath,
reportIndexEntry,
+ sourceAnchor,
sectionCitations,
spanLabel,
verdictTally,
@@ -116,17 +121,67 @@ test("span and post citations link to their moment page, keyed at two decimals",
assert.equal(a1.kind === "audio" && a1.poster, undefined);
});
-test("a source citation carries its document (never the saved copy) and its published still", () => {
+test("a source citation names its document by id; the document (never its saved copy) is in sources once", () => {
const s1 = view.citations.s1;
assert.equal(s1.kind, "source");
if (s1.kind !== "source") return;
assert.equal(s1.image, "/reports/demo/stills/s1.png");
assert.equal(s1.href, "https://example.org/a");
- assert.equal("saved" in s1.source, false);
- assert.equal(view.subject?.title, "An article");
- assert.equal(view.subject?.note, "The first edition.");
- assert.equal(view.subject && "saved" in view.subject, false);
+ assert.equal(s1.sourceId, "s0");
+ assert.equal(s1.sourceTitle, "An article");
+ assert.equal("source" in s1, false);
+ assert.equal(view.subject, "s0");
+ assert.deepEqual(Object.keys(view.sources), ["s0"]);
+ assert.equal(view.sources.s0.title, "An article");
+ assert.equal(view.sources.s0.note, "The first edition.");
+ assert.equal("saved" in view.sources.s0, false);
assert.ok(!JSON.stringify(view).includes("sources/s0/page.html"));
+ // the archive list is in the view once
+ assert.equal(JSON.stringify(view).split("https://archive.example.org/a").length - 1, 1);
+});
+
+test("a claim carries the document's archive links whose context overlaps its sentence, at most five", () => {
+ const sentence = "The council approved the harbor bridge budget in March after a long debate.";
+ const archives = [
+ { label: "a", url: "https://x/a", context: "the council approved the harbor bridge budget" },
+ { label: "b", url: "https://x/b", context: "an unrelated paragraph about weather and gardens" },
+ { label: "c", url: "https://x/c", context: "harbor bridge" }, // too few words in common
+ { label: "d", url: "https://x/d" }, // no context
+ ...Array.from({ length: 7 }, (_, i) => ({ label: `m${i}`, url: `https://x/m${i}`, context: `council approved budget after debate ${i}` })),
+ ];
+ assert.ok(archiveOverlap(sentence, archives[0].context) >= CLAIM_ARCHIVE_MIN_OVERLAP);
+ assert.equal(archiveOverlap(sentence, archives[1].context), 0);
+ assert.equal(archiveOverlap(sentence, archives[2].context), 0);
+ assert.equal(archiveOverlap(sentence, undefined), 0);
+ const found = claimArchives(sentence, archives);
+ assert.equal(found.length, CLAIM_ARCHIVES_MAX);
+ assert.equal(found[0].label, "a", "kept in the document's order");
+ assert.ok(!found.some((a) => a.label === "b" || a.label === "c" || a.label === "d"));
+ assert.deepEqual(claimArchives(sentence, []), []);
+});
+
+test("the builder matches a claim's archives against its source sentence, else its text", () => {
+ const withArchives: Report = {
+ ...report,
+ sources: {
+ s0: {
+ ...report.sources!.s0,
+ archives: [
+ { label: "in four", url: "https://x/4", context: "four words about something" },
+ { label: "elsewhere", url: "https://x/e", context: "nothing like either claim at all" },
+ ],
+ },
+ },
+ citations: { ...report.citations, s1: { kind: "source", source: "s0", quote: "Four words about something said here." } },
+ sections: [
+ { id: "one", title: "One", claims: [{ id: "c1", text: "Claim one.", sourceQuote: { citation: "s1" } }] },
+ { id: "two", title: "Two", claims: [{ id: "c2", text: "Something about four words, the claim's own text." }] },
+ ],
+ };
+ const v = buildReportPageView(withArchives, { record: (c) => record(c.channel, c.id) });
+ assert.deepEqual(v.sections[0].claims[0].archives?.map((a) => a.label), ["in four"]);
+ assert.deepEqual(v.sections[1].claims[0].archives?.map((a) => a.label), ["in four"]);
+ assert.equal(view.sections[0].claims[0].archives, undefined, "no context, no match");
});
test("a page citation links out; the verdict overrides are laid over the defaults", () => {
@@ -197,6 +252,7 @@ test("paths: the clip, the moment view and a report asset", () => {
);
assert.equal(momentViewPath("demo-social/123"), "/m/demo-social/123/moment.json");
assert.equal(reportAssetPath("demo", "./stills/x.png"), "/reports/demo/stills/x.png");
+ assert.equal(sourceAnchor("s0"), "source-s0");
});
test("the placeholder params can never be a report or a moment", () => {
diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts
@@ -19,7 +19,8 @@
//
// A view carries ONLY what is cited: no `saved` source copy (never published),
// no cue beyond the moment's bounded context, no record field a page does not
-// show.
+// show — and nothing large twice: a document (with its archive links) is in a
+// report view's `sources` once, and its citations name it by id.
//
// `buildReportPageView` is the one way a report becomes its view: the numbers
// are lib/report/uses.ts's (first appearance, reading order), the hrefs are
@@ -47,9 +48,10 @@ import type {
SourceArchive,
SpanCitation,
} from "../citations/schema";
+import { quoteTokens } from "../citations/verify";
import { formatTimestamp } from "../vtt";
import type { CitedIn } from "./citedIn";
-import type { Report, ReportKind } from "./schema";
+import type { Claim, Report, ReportKind } from "./schema";
import { reportCitationNumbers } from "./uses";
import { resolveVerdicts, VERDICTS, type Verdict, type VerdictStyle } from "./verdicts";
@@ -171,9 +173,16 @@ export type PostCitationView = CitationViewCommon & {
thread?: boolean;
};
+// A sentence of a document. The document itself (its byline and its archive
+// links — a long article may carry hundreds) is in the view's `sources` map
+// ONCE, never copied into each citation of it: a report citing one article
+// eighty times would otherwise carry its archive list eighty times.
export type SourceCitationView = CitationViewCommon & {
kind: "source";
- source: SourceView;
+ // The document's id in the view's `sources`, and its title for the card's
+ // "from <title>" (sourceAnchor links to the document's block on the page).
+ sourceId: string;
+ sourceTitle: string;
// The still of the sentence, published (reportAssetPath).
image?: string;
// The document itself, when it has a URL.
@@ -198,6 +207,10 @@ export type ClaimView = {
verdict?: Verdict;
// The id of the `source` citation holding the document's own sentence.
sourceQuote?: string;
+ // The document's archive links that sit in this claim's sentence (their
+ // `context` overlaps it: claimArchives), at most CLAIM_ARCHIVES_MAX. The
+ // document's full list is shown once, with the document.
+ archives?: SourceArchive[];
findings?: string;
citations: string[];
};
@@ -219,7 +232,11 @@ export type ReportPageView = {
summary?: string;
published?: string;
updated?: string;
- subject?: SourceView;
+ // The document under review: its id in `sources`.
+ subject?: string;
+ // Every document the report's `source` citations quote, and the subject,
+ // by id — each once.
+ sources: Record<string, SourceView>;
// Every verdict's label and colour, the report's overrides applied.
verdicts: Record<Verdict, VerdictStyle>;
// Every citation the report cites, numbered; one it defines but never cites
@@ -405,7 +422,8 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
citations[id] = defined({
kind: "source" as const,
...common,
- source: sourceView(c.source, s),
+ sourceId: c.source,
+ sourceTitle: s.title,
image: c.image ? reportAssetPath(report.id, c.image) : undefined,
href: s.url,
});
@@ -422,7 +440,23 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
break;
}
}
- const subject = report.subject && sources[report.subject.source];
+ // Each document once: the subject, then every document a cited source
+ // citation quotes.
+ const sourceViews: Record<string, SourceView> = {};
+ const subjectId = report.subject && sources[report.subject.source] ? report.subject.source : undefined;
+ if (subjectId) sourceViews[subjectId] = sourceView(subjectId, sources[subjectId]);
+ for (const c of Object.values(citations)) {
+ if (c.kind === "source" && !sourceViews[c.sourceId]) sourceViews[c.sourceId] = sourceView(c.sourceId, sources[c.sourceId]);
+ }
+ // A claim's archive links: those of the document its sentence is from (the
+ // subject when it names none) whose context overlaps the sentence.
+ const archivesFor = (claim: Claim): SourceArchive[] | undefined => {
+ const sq = claim.sourceQuote ? citations[claim.sourceQuote.citation] : undefined;
+ const doc = sq?.kind === "source" ? sourceViews[sq.sourceId] : subjectId ? sourceViews[subjectId] : undefined;
+ if (!doc || doc.archives.length === 0) return undefined;
+ const found = claimArchives(sq?.quote ?? claim.text, doc.archives);
+ return found.length > 0 ? found : undefined;
+ };
return defined({
format: REPORT_PAGE_FORMAT,
version: REPORT_VIEWS_VERSION,
@@ -433,7 +467,8 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
summary: report.summary,
published: report.published,
updated: report.updated,
- subject: subject ? sourceView(report.subject!.source, subject) : undefined,
+ subject: subjectId,
+ sources: sourceViews,
verdicts: resolveVerdicts(report.verdicts),
citations,
sections: report.sections.map((s) =>
@@ -448,6 +483,7 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
text: c.text,
verdict: c.verdict,
sourceQuote: c.sourceQuote?.citation,
+ archives: archivesFor(c),
findings: c.findings,
citations: c.citations ?? [],
}),
@@ -458,8 +494,48 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver)
});
}
+// ─── A claim's archive links ───
+
+// A document's archive link belongs to a claim when the words around it (its
+// `context`, as the document had them) overlap the claim's sentence: at least
+// CLAIM_ARCHIVE_MIN_SHARED distinct words of four letters or more in common,
+// and those at least CLAIM_ARCHIVE_MIN_OVERLAP of the shorter side's. Words
+// are normalised as quote verification normalises them (verify.ts
+// quoteTokens); short words (the, and, of) are left out, they match anything.
+// The best CLAIM_ARCHIVES_MAX are kept, in the document's order.
+export const CLAIM_ARCHIVE_MIN_OVERLAP = 0.6;
+export const CLAIM_ARCHIVE_MIN_SHARED = 3;
+export const CLAIM_ARCHIVES_MAX = 5;
+
+const contentWords = (text: string) => new Set(quoteTokens(text).filter((t) => t.length >= 4));
+
+export function archiveOverlap(sentence: string, context: string | undefined): number {
+ if (!context) return 0;
+ const a = contentWords(sentence);
+ const b = contentWords(context);
+ let shared = 0;
+ for (const t of b) if (a.has(t)) shared++;
+ if (shared < CLAIM_ARCHIVE_MIN_SHARED) return 0;
+ return shared / Math.min(a.size, b.size);
+}
+
+export function claimArchives(sentence: string, archives: readonly SourceArchive[]): SourceArchive[] {
+ return archives
+ .map((a, i) => ({ a, i, score: archiveOverlap(sentence, a.context) }))
+ .filter((x) => x.score >= CLAIM_ARCHIVE_MIN_OVERLAP)
+ .sort((x, y) => y.score - x.score || x.i - y.i)
+ .slice(0, CLAIM_ARCHIVES_MAX)
+ .sort((x, y) => x.i - y.i)
+ .map((x) => x.a);
+}
+
// ─── Reading a view ───
+// The anchor (fragment id, without `#`) of a document's block on a report page.
+export function sourceAnchor(sourceId: string): string {
+ return `source-${sourceId}`;
+}
+
// The report's citations in number order.
export function orderedCitations(view: Pick<ReportPageView, "citations">): CitationView[] {
return Object.values(view.citations)
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -2,7 +2,7 @@
## [Unreleased]
- **`corpus.json` is spec 5: it names a site's reports, and a site that publishes only reports says so.** A site with reports adds `reports` to its `corpus.json` (`index`: `/reports/index.json`, the count, and how to read a report's page, its citations and its moment pages) and a Reports section to `llms.txt`; its sitemap lists the report and moment pages. A site that publishes only its reports has `"scope": "cited"` and its audience under `site`, no channels and zero totals, an `llms.txt` that lists its reports and how their citations and moment pages are read, and a `site.json` with no channels. A reader that does not know spec 5 sees an empty corpus there. Needs a rebuild and deploy of each site.
-- **A site can show cited reports, and every citation opens on a page of its own.** A site built with reports has a **Reports** link in its header and a page at `/reports/` listing them. A report's page has its title, subtitle, dates and the document under review with its archive links; a fact-check's tally of verdicts; the summary; the sections and their claims, each with its verdict, the document's own sentence as an image, the findings and the evidence cards; a numbered reference list; and links to download its citations as JSON and CSV. A citation in the text shows as its words plus a number: hovering it, focusing the number or tapping it once shows a card of the citation (the quote, who said it and when, a picture or the post's screenshot, and how closely the quote matched the transcript when it was checked); the words open what it cites and the number jumps to its reference. A cited span of a video or audio record opens at `/m/<channel>/<id>/<start>-<end>/`: a short clip of the span with a little context either side, the quote, the transcript lines around it, the record's title, channel and date, a link to the original at that time, and every report on the site that cites it. A cited post opens at `/m/<channel>/<id>/` with its screenshot and text. A site that publishes only its reports (`site.json` `publish: "cited"`) opens on the report index and has no search, Ask AI, downloads or duplicates. A site with no reports is unchanged. Needs a rebuild and deploy of each site.
+- **A site can show cited reports, and every citation opens on a page of its own.** A site built with reports has a **Reports** link in its header and a page at `/reports/` listing them. A report's page has its title, subtitle, dates and the document under review, its archive links listed once under it and folded away ("N archive links in context"); a fact-check's tally of verdicts; the summary; the sections and their claims, each with its verdict, the document's own sentence as an image with a link back to the document and only the archive links that sit in that sentence (at most five), the findings and the evidence cards; a numbered reference list; and links to download its citations as JSON and CSV. A citation in the text shows as its words plus a number: hovering it, focusing the number or tapping it once shows a card of the citation (the quote, who said it and when, a picture or the post's screenshot, and how closely the quote matched the transcript when it was checked); the words open what it cites and the number jumps to its reference. A cited span of a video or audio record opens at `/m/<channel>/<id>/<start>-<end>/`: a short clip of the span with a little context either side, the quote, the transcript lines around it, the record's title, channel and date, a link to the original at that time, and every report on the site that cites it. A cited post opens at `/m/<channel>/<id>/` with its screenshot and text. A site that publishes only its reports (`site.json` `publish: "cited"`) opens on the report index and has no search, Ask AI, downloads or duplicates. A site with no reports is unchanged. Needs a rebuild and deploy of each site.
## [0.11.1] - 2026-10-01
- **Use with AI goes to the Archilyzer site's AI and MCP doc; the page on each site is gone.** The header's, the slide-out menu's, the footer's and Ask AI's **Use with AI** keep their label and open https://archilyzer.pages.dev/docs/ai-and-mcp/ in the same tab, on every site and the hub, where one block says how to run Claude Code against any archive (the source, `pnpm install`, `claude mcp add archilyzer`, `/ask`). `/use-with-ai/` is no longer built. `corpus.json`'s `useWithAi` names the doc; `llms.txt`'s Ask AI section lists the site's `/ask/` chat and the doc; the sitemap drops `/use-with-ai`. Needs a rebuild and deploy of each site and the hub.
diff --git a/export/app/components/reports/ReportArticle.tsx b/export/app/components/reports/ReportArticle.tsx
@@ -1,17 +1,20 @@
import { ArrowDownToLine } from "lucide-react";
import { CitationCard } from "yt-dlp-transcript-common/components/citations/CitationCard";
+import { CitationsProvider } from "yt-dlp-transcript-common/components/citations/CitationsContext";
import { CitedMarkdown } from "yt-dlp-transcript-common/components/citations/CitedMarkdown";
import { ReferenceList } from "yt-dlp-transcript-common/components/citations/ReferenceList";
import { VerdictChip, VerdictTally } from "yt-dlp-transcript-common/components/report/VerdictChip";
+import type { SourceArchive } from "yt-dlp-transcript-common/lib/citations/schema";
import {
orderedCitations,
+ sourceAnchor,
verdictTally,
type CitationView,
type ClaimView,
type ReportPageView,
type SourceCitationView,
} from "yt-dlp-transcript-common/lib/report/views";
-import { ExternalLinkText, Eyebrow, SourceBlock, dateLabel, textLink } from "./parts";
+import { ArchiveList, Eyebrow, SourceBlock, dateLabel, textLink } from "./parts";
// ONE REPORT, from its view (common/lib/report/views.ts): the header (title,
// subtitle, dates, the document under review with its archive links), a
@@ -22,7 +25,7 @@ import { ExternalLinkText, Eyebrow, SourceBlock, dateLabel, textLink } from "./p
const proseClass = "text-[0.95rem] leading-relaxed text-foreground";
-function SourceSentence({ c }: { c: SourceCitationView }) {
+function SourceSentence({ c, archives }: { c: SourceCitationView; archives: readonly SourceArchive[] | undefined }) {
return (
<figure data-source-sentence={c.id} className="flex flex-col gap-1.5">
{c.image ? (
@@ -38,13 +41,13 @@ function SourceSentence({ c }: { c: SourceCitationView }) {
<q>{c.quote}</q>
</blockquote>
)}
- <figcaption className="flex flex-wrap items-baseline gap-x-3 gap-y-1 text-xs text-muted-foreground">
- <span>{c.source.title}</span>
- {c.source.archives.map((a) => (
- <ExternalLinkText key={a.url} href={a.url} title={a.context}>
- {a.label}
- </ExternalLinkText>
- ))}
+ <figcaption className="flex flex-col gap-1 text-xs text-muted-foreground">
+ <a href={`#${sourceAnchor(c.sourceId)}`} className={textLink}>
+ from {c.sourceTitle}
+ </a>
+ {/* Only the document's archive links that sit in this sentence; the
+ full list is with the document, once. */}
+ {archives && archives.length > 0 && <ArchiveList archives={archives} />}
</figcaption>
</figure>
);
@@ -83,9 +86,9 @@ function Claim({
{subjectLabel}: <q className="text-foreground">{claim.text}</q>
</p>
)}
- {sentence?.kind === "source" && <SourceSentence c={sentence} />}
+ {sentence?.kind === "source" && <SourceSentence c={sentence} archives={claim.archives} />}
{claim.findings && (
- <CitedMarkdown citations={view.citations} className={proseClass}>
+ <CitedMarkdown className={proseClass}>
{claim.findings}
</CitedMarkdown>
)}
@@ -107,97 +110,116 @@ export default function ReportArticle({ view }: { view: ReportPageView }) {
const isFactcheck = view.kind === "factcheck";
const tally = isFactcheck ? verdictTally(view) : [];
const references = orderedCitations(view);
- const subjectLabel = view.subject?.kind === "article" ? "The article says" : "The source says";
+ const subject = view.subject ? view.sources[view.subject] : undefined;
+ const subjectLabel = subject?.kind === "article" ? "The article says" : "The source says";
const dates = [
view.published ? `Published ${dateLabel(view.published)}` : null,
view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null,
].filter(Boolean);
const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0);
+ // Documents quoted that are not the subject, each shown once before the
+ // references.
+ const otherSources = Object.values(view.sources).filter((s) => s.id !== view.subject);
return (
- <article data-report={view.id} className="mx-auto flex w-full max-w-3xl flex-col gap-8">
- <header className="flex flex-col gap-3 border-b border-border pb-6">
- <Eyebrow>{isFactcheck ? "Fact-check" : "Report"}</Eyebrow>
- <h1 className="font-display text-3xl font-semibold leading-tight tracking-tight text-foreground sm:text-4xl">
- {view.title}
- </h1>
- {view.subtitle && <p className="text-lg text-muted-foreground">{view.subtitle}</p>}
- {dates.length > 0 && <p className="font-mono text-xs text-muted-foreground">{dates.join(" · ")}</p>}
- {view.subject && <SourceBlock source={view.subject} label="Under review" />}
- {tally.length > 0 && (
- <div className="flex flex-col gap-2">
- <p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground">
- {claimCount} claim{claimCount === 1 ? "" : "s"} checked
- </p>
- <VerdictTally tally={tally} styles={view.verdicts} />
- </div>
+ // One citation map for every inline marker on the page (CitationsContext):
+ // serialized once, not once per marker.
+ <CitationsProvider citations={view.citations}>
+ <article data-report={view.id} className="mx-auto flex w-full max-w-3xl flex-col gap-8">
+ <header className="flex flex-col gap-3 border-b border-border pb-6">
+ <Eyebrow>{isFactcheck ? "Fact-check" : "Report"}</Eyebrow>
+ <h1 className="font-display text-3xl font-semibold leading-tight tracking-tight text-foreground sm:text-4xl">
+ {view.title}
+ </h1>
+ {view.subtitle && <p className="text-lg text-muted-foreground">{view.subtitle}</p>}
+ {dates.length > 0 && <p className="font-mono text-xs text-muted-foreground">{dates.join(" · ")}</p>}
+ {subject && <SourceBlock source={subject} label="Under review" />}
+ {tally.length > 0 && (
+ <div className="flex flex-col gap-2">
+ <p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground">
+ {claimCount} claim{claimCount === 1 ? "" : "s"} checked
+ </p>
+ <VerdictTally tally={tally} styles={view.verdicts} />
+ </div>
+ )}
+ </header>
+
+ {view.summary && (
+ <CitedMarkdown className={proseClass}>
+ {view.summary}
+ </CitedMarkdown>
)}
- </header>
- {view.summary && (
- <CitedMarkdown citations={view.citations} className={proseClass}>
- {view.summary}
- </CitedMarkdown>
- )}
+ {view.sections.length > 1 && (
+ <nav aria-label="Sections" className="rounded-lg border border-border p-4 text-sm">
+ <ol className="flex list-decimal flex-col gap-1 pl-5 marker:text-muted-foreground">
+ {view.sections.map((s) => (
+ <li key={s.id}>
+ <a href={`#${s.id}`} className={textLink}>
+ {s.title}
+ </a>
+ {s.claims.length > 0 && (
+ <span className="ml-2 font-mono text-xs text-muted-foreground">
+ {s.claims.length} claim{s.claims.length === 1 ? "" : "s"}
+ </span>
+ )}
+ </li>
+ ))}
+ </ol>
+ </nav>
+ )}
- {view.sections.length > 1 && (
- <nav aria-label="Sections" className="rounded-lg border border-border p-4 text-sm">
- <ol className="flex list-decimal flex-col gap-1 pl-5 marker:text-muted-foreground">
- {view.sections.map((s) => (
- <li key={s.id}>
- <a href={`#${s.id}`} className={textLink}>
- {s.title}
- </a>
- {s.claims.length > 0 && (
- <span className="ml-2 font-mono text-xs text-muted-foreground">
- {s.claims.length} claim{s.claims.length === 1 ? "" : "s"}
- </span>
- )}
- </li>
+ {view.sections.map((s) => (
+ <section key={s.id} id={s.id} data-section={s.id} className="flex scroll-mt-20 flex-col gap-4">
+ <h2 className="font-display text-2xl font-semibold tracking-tight text-foreground">{s.title}</h2>
+ {s.body && (
+ <CitedMarkdown className={proseClass}>
+ {s.body}
+ </CitedMarkdown>
+ )}
+ {s.claims.map((c) => (
+ <Claim key={c.id} claim={c} view={view} subjectLabel={subjectLabel} />
))}
- </ol>
- </nav>
- )}
+ </section>
+ ))}
- {view.sections.map((s) => (
- <section key={s.id} id={s.id} data-section={s.id} className="flex scroll-mt-20 flex-col gap-4">
- <h2 className="font-display text-2xl font-semibold tracking-tight text-foreground">{s.title}</h2>
- {s.body && (
- <CitedMarkdown citations={view.citations} className={proseClass}>
- {s.body}
- </CitedMarkdown>
- )}
- {s.claims.map((c) => (
- <Claim key={c.id} claim={c} view={view} subjectLabel={subjectLabel} />
- ))}
- </section>
- ))}
+ {otherSources.length > 0 && (
+ <section aria-labelledby="sources-heading" className="flex flex-col gap-3 border-t border-border pt-6">
+ <h2 id="sources-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground">
+ Sources
+ </h2>
+ {otherSources.map((s) => (
+ <SourceBlock key={s.id} source={s} label="Quoted" />
+ ))}
+ </section>
+ )}
- {references.length > 0 && (
- <section id="references" aria-labelledby="references-heading" className="flex scroll-mt-20 flex-col gap-2 border-t border-border pt-6">
- <h2 id="references-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground">
- References
- </h2>
- <ReferenceList citations={references} />
- </section>
- )}
+ {references.length > 0 && (
+ <section id="references" aria-labelledby="references-heading" className="flex scroll-mt-20 flex-col gap-2 border-t border-border pt-6">
+ <h2 id="references-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground">
+ References
+ </h2>
+ <ReferenceList citations={references} />
+ </section>
+ )}
- {(view.downloads?.json || view.downloads?.csv) && (
- <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground">
- <ArrowDownToLine className="size-4" aria-hidden />
- Download citations:
- {view.downloads.json && (
- <a href={view.downloads.json} download className={textLink}>
- JSON
- </a>
- )}
- {view.downloads.csv && (
- <a href={view.downloads.csv} download className={textLink}>
- CSV
- </a>
- )}
- </p>
- )}
- </article>
+ {(view.downloads?.json || view.downloads?.csv) && (
+ <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground">
+ <ArrowDownToLine className="size-4" aria-hidden />
+ Download citations:
+ {view.downloads.json && (
+ <a href={view.downloads.json} download className={textLink}>
+ JSON
+ </a>
+ )}
+ {view.downloads.csv && (
+ <a href={view.downloads.csv} download className={textLink}>
+ CSV
+ </a>
+ )}
+ </p>
+ )}
+ </article>
+ </CitationsProvider>
);
}
diff --git a/export/app/components/reports/parts.tsx b/export/app/components/reports/parts.tsx
@@ -1,6 +1,7 @@
import Link from "next/link";
-import { Archive, ExternalLink } from "lucide-react";
-import type { SourceView } from "yt-dlp-transcript-common/lib/report/views";
+import { ExternalLink } from "lucide-react";
+import { sourceAnchor, type SourceView } from "yt-dlp-transcript-common/lib/report/views";
+import type { SourceArchive } from "yt-dlp-transcript-common/lib/citations/schema";
// Small pieces the report, index and moment pages share.
@@ -26,35 +27,55 @@ export function ExternalLinkText({ href, children, title }: { href: string; chil
);
}
-// A document under review: its title (linked), byline, note, and its archive
-// links in the context the document gave them.
+// A document a report quotes: its title (linked), byline, note, and its
+// archive links in the context the document gave them — collapsed, since a
+// long document may carry hundreds. The block is the document's one place on
+// the page: its anchor is what a source citation's "from <title>" links to.
export function SourceBlock({ source, label }: { source: SourceView; label: string }) {
const byline = [source.publisher, source.author, dateLabel(source.date)].filter(Boolean).join(" · ");
+ const n = source.archives.length;
return (
- <div data-subject-source={source.id} className="flex flex-col gap-1.5 rounded-lg border border-border bg-card p-4 text-sm">
+ <div
+ id={sourceAnchor(source.id)}
+ data-subject-source={source.id}
+ className="flex scroll-mt-20 flex-col gap-1.5 rounded-lg border border-border bg-card p-4 text-sm"
+ >
<p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground">{label}</p>
<p className="font-medium leading-snug text-foreground">
{source.url ? <ExternalLinkText href={source.url}>{source.title}</ExternalLinkText> : source.title}
</p>
{byline && <p className="font-mono text-xs text-muted-foreground">{byline}</p>}
{source.note && <p className="text-xs text-muted-foreground">{source.note}</p>}
- {source.archives.length > 0 && (
- <ul className="mt-1 flex flex-col gap-1 text-xs">
- {source.archives.map((a) => (
- <li key={a.url} className="flex flex-wrap items-baseline gap-x-2">
- <ExternalLinkText href={a.url}>
- <Archive className="size-3 shrink-0" aria-hidden />
- {a.label}
- </ExternalLinkText>
- {a.context && <span className="text-muted-foreground">— {a.context}</span>}
- </li>
- ))}
- </ul>
+ {n > 0 && (
+ <details data-source-archives={n} className="mt-1 text-xs">
+ <summary className="cursor-pointer text-muted-foreground hover:text-foreground">
+ {`${n} archive link${n === 1 ? "" : "s"} in context`}
+ </summary>
+ <ArchiveList archives={source.archives} className="mt-2" />
+ </details>
)}
</div>
);
}
+// Archive links, each with the words around it in the document. A document
+// may carry hundreds, so a link is text with a CSS arrow, not two inline SVG
+// icons apiece.
+export function ArchiveList({ archives, className }: { archives: readonly SourceArchive[]; className?: string }) {
+ return (
+ <ul className={`flex flex-col gap-1 ${className ?? ""}`}>
+ {archives.map((a, i) => (
+ <li key={`${i}:${a.url}`} className="flex flex-wrap items-baseline gap-x-2">
+ <a href={a.url} target="_blank" rel="noopener noreferrer" className={`${textLink} after:ml-0.5 after:content-['↗']`}>
+ {a.label}
+ </a>
+ {a.context && <span className="text-muted-foreground">— {a.context}</span>}
+ </li>
+ ))}
+ </ul>
+ );
+}
+
export function EmptyState({ title, children }: { title: string; children: React.ReactNode }) {
return (
<div data-reports-empty="" className="mx-auto flex max-w-2xl flex-col gap-3">
diff --git a/export/app/lib/reportSize.test.ts b/export/app/lib/reportSize.test.ts
@@ -0,0 +1,48 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import * as React from "react";
+import { renderToString } from "react-dom/server";
+import { syntheticReport, syntheticReportView } from "../../fixtures/report-site/synthetic";
+import { validateReport } from "yt-dlp-transcript-common/lib/report/validate";
+import { CLAIM_ARCHIVES_MAX } from "yt-dlp-transcript-common/lib/report/views";
+import ReportArticle from "../components/reports/ReportArticle";
+
+// A REPORT PAGE'S SIZE is bounded by what it cites, not by what it repeats. A
+// report citing one long document many times once carried that document — and
+// its hundreds of archive links — inside every citation of it, and rendered
+// them on every card and preview: a static host's per-file limit refused the
+// page. The synthetic report (fixtures/report-site/synthetic.ts: 400 citations,
+// 100 of them of ONE source with 300 archive links) holds the budget.
+
+(globalThis as { React?: typeof React }).React = React;
+
+const MB = 1024 * 1024;
+const report = syntheticReport();
+const view = syntheticReportView(report);
+
+test("the synthetic report is sound and as large as it says", () => {
+ assert.deepEqual(validateReport(report), []);
+ const kinds = Object.values(view.citations).reduce<Record<string, number>>((n, c) => ((n[c.kind] = (n[c.kind] ?? 0) + 1), n), {});
+ assert.deepEqual(kinds, { source: 100, video: 200, post: 100 });
+ assert.equal(view.sources.s0.archives.length, 300);
+});
+
+test("the page view is under 2 MB, and carries the document once", () => {
+ const json = JSON.stringify(view, null, 2);
+ assert.ok(Buffer.byteLength(json) < 2 * MB, `page.json is ${Buffer.byteLength(json)} bytes`);
+ // every archive URL appears once in the document's entry, and otherwise only
+ // in the claims it belongs to (at most CLAIM_ARCHIVES_MAX each)
+ const first = view.sources.s0.archives[0].url;
+ const claimsWith = view.sections.flatMap((s) => s.claims).filter((c) => c.archives?.some((a) => a.url === first)).length;
+ assert.equal(json.split(first + '"').length - 1, 1 + claimsWith);
+ for (const c of view.sections.flatMap((s) => s.claims)) assert.ok((c.archives?.length ?? 0) <= CLAIM_ARCHIVES_MAX);
+ for (const c of Object.values(view.citations)) assert.ok(!("source" in c), "a source citation names its document by id");
+});
+
+test("the rendered report page is under 5 MB", () => {
+ const html = renderToString(React.createElement(ReportArticle, { view }));
+ assert.ok(Buffer.byteLength(html) < 5 * MB, `the page renders to ${Buffer.byteLength(html)} bytes`);
+ // the document's full archive list is on the page once, collapsed
+ assert.equal(html.match(/data-source-archives="300"/g)?.length, 1);
+ assert.match(html, /<details data-source-archives="300"[^>]*><summary[^>]*>300 archive links in context<\/summary>/);
+});
diff --git a/export/app/lib/reports.test.ts b/export/app/lib/reports.test.ts
@@ -85,6 +85,10 @@ test("the report page: header, tally, sections and claims, inline cites, referen
assert.ok(html.includes("Published 2026-10-01 · Updated 2026-10-04"));
assert.match(html, /data-subject-source="s0"/);
assert.ok(html.includes("as published on the day"));
+ // the document's archive links are listed once, collapsed, under its anchor
+ assert.match(html, /<div id="source-s0"[^>]*>[^]*?<details data-source-archives="1"/);
+ // a claim's source sentence links to the document instead of repeating them
+ assert.match(html, /data-source-sentence="a01"[^]*?<a href="#source-s0"[^>]*>from An example article about a demo channel<\/a>/);
assert.match(html, /data-verdict-tally=""/);
for (const v of ["CORROBORATED", "PARTLY", "CONTRADICTED", "UNTESTABLE"]) assert.match(html, new RegExp(`data-verdict="${v}"`));
assert.ok(html.includes("Read in the edition published on the day"), "the source's note, under its byline");
diff --git a/export/fixtures/report-site/public/reports/demo-factcheck/page.json b/export/fixtures/report-site/public/reports/demo-factcheck/page.json
@@ -8,22 +8,25 @@
"summary": "The article makes four claims. The recordings support one, partly support another, and contradict a third; the fourth cannot be tested. The host said the opening date out loud [on stream](cite:c01), and later [repeated it](cite:au1).",
"published": "2026-10-01",
"updated": "2026-10-04",
- "subject": {
- "id": "s0",
- "kind": "article",
- "title": "An example article about a demo channel",
- "url": "https://example.org/articles/demo",
- "publisher": "Example Gazette",
- "author": "A. Writer",
- "date": "2026-09-20",
- "note": "Read in the edition published on the day; it has since been revised.",
- "archives": [
- {
- "label": "archive.org",
- "url": "https://web.archive.org/web/2026/https://example.org/articles/demo",
- "context": "as published on the day"
- }
- ]
+ "subject": "s0",
+ "sources": {
+ "s0": {
+ "id": "s0",
+ "kind": "article",
+ "title": "An example article about a demo channel",
+ "url": "https://example.org/articles/demo",
+ "publisher": "Example Gazette",
+ "author": "A. Writer",
+ "date": "2026-09-20",
+ "note": "Read in the edition published on the day; it has since been revised.",
+ "archives": [
+ {
+ "label": "archive.org",
+ "url": "https://web.archive.org/web/2026/https://example.org/articles/demo",
+ "context": "as published on the day"
+ }
+ ]
+ }
},
"verdicts": {
"CORROBORATED": {
@@ -102,23 +105,8 @@
"id": "a01",
"number": 3,
"quote": "He opened the bridge himself in 2018.",
- "source": {
- "id": "s0",
- "kind": "article",
- "title": "An example article about a demo channel",
- "url": "https://example.org/articles/demo",
- "publisher": "Example Gazette",
- "author": "A. Writer",
- "date": "2026-09-20",
- "note": "Read in the edition published on the day; it has since been revised.",
- "archives": [
- {
- "label": "archive.org",
- "url": "https://web.archive.org/web/2026/https://example.org/articles/demo",
- "context": "as published on the day"
- }
- ]
- },
+ "sourceId": "s0",
+ "sourceTitle": "An example article about a demo channel",
"image": "/reports/demo-factcheck/stills/a01.png",
"href": "https://example.org/articles/demo"
},
@@ -137,23 +125,8 @@
"id": "a02",
"number": 5,
"quote": "He has said many times that he would move away.",
- "source": {
- "id": "s0",
- "kind": "article",
- "title": "An example article about a demo channel",
- "url": "https://example.org/articles/demo",
- "publisher": "Example Gazette",
- "author": "A. Writer",
- "date": "2026-09-20",
- "note": "Read in the edition published on the day; it has since been revised.",
- "archives": [
- {
- "label": "archive.org",
- "url": "https://web.archive.org/web/2026/https://example.org/articles/demo",
- "context": "as published on the day"
- }
- ]
- },
+ "sourceId": "s0",
+ "sourceTitle": "An example article about a demo channel",
"href": "https://example.org/articles/demo"
},
"p01": {
diff --git a/export/fixtures/report-site/synthetic.ts b/export/fixtures/report-site/synthetic.ts
@@ -0,0 +1,111 @@
+// A SYNTHETIC LARGE REPORT, for the page-size test (app/lib/reportSize.test.ts)
+// and for measuring a built page: 400 citations — 100 `source` citations of ONE
+// source that carries 300 archive links, 200 video spans and 100 posts — over
+// 100 claims in 10 sections, each claim with its source sentence, findings
+// citing two spans and a post inline, and those listed under it.
+//
+// Neutral generated text; the shape is what matters.
+
+import type { Report } from "yt-dlp-transcript-common/lib/report/schema";
+import { buildReportPageView, type RecordView, type ReportPageView } from "yt-dlp-transcript-common/lib/report/views";
+
+export const SYNTHETIC_REPORT_ID = "synthetic-large";
+
+const WORDS = [
+ "harbor", "ledger", "council", "bridge", "orchard", "lantern", "meadow", "quarry", "signal", "timber",
+ "valley", "warden", "beacon", "canyon", "delta", "ember", "falcon", "garnet", "hollow", "island",
+];
+
+// Claim i's sentence: words chosen by i, so claims differ and archive
+// contexts can be built to overlap exactly one of them.
+export function syntheticSentence(i: number): string {
+ const w = (k: number) => WORDS[(i * 7 + k * 3) % WORDS.length];
+ return `In paragraph ${i} the article says the ${w(0)} ${w(1)} met the ${w(2)} ${w(3)} near the ${w(4)} in ${2000 + (i % 25)}.`;
+}
+
+export function syntheticReport(): Report {
+ const archives = Array.from({ length: 300 }, (_, k) => ({
+ label: `archive ${k + 1}`,
+ url: `https://web.archive.org/web/2026/https://example.org/ref/${k + 1}`,
+ // Each archive link sits in claim (k % 100)'s sentence.
+ context: syntheticSentence(k % 100),
+ }));
+ const citations: NonNullable<Report["citations"]> = {};
+ for (let i = 0; i < 100; i++) {
+ citations[`a${i}`] = { kind: "source", source: "s0", quote: syntheticSentence(i), image: `stills/a${i}.png` };
+ }
+ for (let i = 0; i < 200; i++) {
+ citations[`v${i}`] = {
+ kind: "video",
+ channel: "demo-channel",
+ id: `vid${i % 40}`,
+ start: 60 + i * 30,
+ end: 80 + i * 30,
+ pad: { before: 5, after: 5 },
+ quote: `Span ${i}: and that is when I said the ${WORDS[i % 20]} was never part of the plan, not once, not ever, and I stand by it today.`,
+ verification: { quoteScore: 0.95, quoteCheckedAt: "2026-10-04T12:00:00Z", method: "token recall v1" },
+ };
+ }
+ for (let i = 0; i < 100; i++) {
+ citations[`p${i}`] = {
+ kind: "post",
+ channel: "demo-social",
+ id: `${1000000 + i}`,
+ quote: `Post ${i}: the ${WORDS[i % 20]} thing again? I covered it already, read the thread.`,
+ };
+ }
+ const sections: Report["sections"] = Array.from({ length: 10 }, (_, s) => ({
+ id: `section-${s}`,
+ title: `Section ${s}`,
+ body: `Section ${s} looks at ten claims.`,
+ claims: Array.from({ length: 10 }, (_, c) => {
+ const i = s * 10 + c;
+ return {
+ id: `claim-${i}`,
+ title: `Claim ${i}`,
+ text: syntheticSentence(i),
+ verdict: (["CORROBORATED", "PARTLY", "CONTRADICTED", "NOT_FOUND", "UNTESTABLE"] as const)[i % 5],
+ sourceQuote: { citation: `a${i}` },
+ findings: `The recordings say otherwise [here](cite:v${2 * i}) and [here](cite:v${2 * i + 1}); a [post](cite:p${i}) agrees.`,
+ citations: [`v${2 * i}`, `v${2 * i + 1}`, `p${i}`],
+ };
+ }),
+ }));
+ return {
+ format: "archilyzer-report",
+ version: 1,
+ id: SYNTHETIC_REPORT_ID,
+ kind: "factcheck",
+ title: "A synthetic large fact-check",
+ summary: "Four hundred citations of one source, spans and posts.",
+ subject: { source: "s0" },
+ sources: {
+ s0: {
+ kind: "article",
+ title: "A long example article",
+ url: "https://example.org/long",
+ publisher: "Example Gazette",
+ archives,
+ },
+ },
+ citations,
+ sections,
+ };
+}
+
+const record = (channel: string, id: string): RecordView => ({
+ channel,
+ channelTitle: "Demo Channel",
+ id,
+ title: `Demo record ${id}`,
+ date: "2026-01-10",
+ platform: "youtube",
+ originalUrl: `https://media.example.org/${channel}/${id}`,
+});
+
+export function syntheticReportView(report = syntheticReport()): ReportPageView {
+ return buildReportPageView(report, {
+ record: (c) => record(c.channel, c.id),
+ post: (c) => ({ author: "@demo_account", text: `${c.quote}\n\nMore words after the quote.` }),
+ });
+}