Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 117b9ae0effb842fe2c6f8bd2d975a807c16607e
parent d7f670b2677fbaa04f4b37396cc920e3968cc808
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 04:40:57 -0400

report pages: a document once — sources map, source citations by id, archive links collapsed under the document and matched per claim; one citation map per page for the inline markers; a size test on a 400-citation report

The page view carries each quoted document once in `sources` (the subject is
its id); a source citation names it by `sourceId` with its title. A claim
carries only the document's archive links whose context overlaps its sentence
(claimArchives, at most five). Inline markers read their view from one
CitationsProvider by id instead of carrying it as a prop. The archive lists
are text links without inline icons. The R4 fixture JSON is regenerated.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/Markdown.tsx | 12++++--------
Mcommon/components/citations/CitationCard.tsx | 24++++++++++--------------
Acommon/components/citations/CitationsContext.tsx | 31+++++++++++++++++++++++++++++++
Mcommon/components/citations/CitedMarkdown.tsx | 32++++++++++++++------------------
Mcommon/components/citations/InlineCite.tsx | 20+++++++++++++++++---
Mcommon/components/citations/citations.test.ts | 35++++++++++++++++++++++++-----------
Mcommon/lib/report/views.test.ts | 66+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Mcommon/lib/report/views.ts | 90++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Mexport/CHANGELOG.md | 2+-
Mexport/app/components/reports/ReportArticle.tsx | 200++++++++++++++++++++++++++++++++++++++++++++-----------------------------------
Mexport/app/components/reports/parts.tsx | 55++++++++++++++++++++++++++++++++++++++-----------------
Aexport/app/lib/reportSize.test.ts | 48++++++++++++++++++++++++++++++++++++++++++++++++
Mexport/app/lib/reports.test.ts | 4++++
Mexport/fixtures/report-site/public/reports/demo-factcheck/page.json | 73+++++++++++++++++++++++--------------------------------------------------
Aexport/fixtures/report-site/synthetic.ts | 111+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
15 files changed, 580 insertions(+), 223 deletions(-)

diff --git a/common/components/Markdown.tsx b/common/components/Markdown.tsx @@ -73,22 +73,18 @@ const OVERRIDES = { // spacing is consistent regardless of content shape. `linkComponent` optionally // overrides how `<a>` renders (e.g. to make in-page citation anchors scroll // instead of opening a new tab); when omitted, links use the default external -// styling above. `linkProps` are extra props every `<a>` override receives -// beside the link's own (e.g. a citation map), so the link component can stay -// one stable component instead of a closure made per render. -export function Markdown<P extends object = object>({ +// styling above. +export function Markdown({ children, className, linkComponent, - linkProps, }: { children: string; className?: string; - linkComponent?: ComponentType<AnchorHTMLAttributes<HTMLAnchorElement> & P>; - linkProps?: P; + linkComponent?: ComponentType<AnchorHTMLAttributes<HTMLAnchorElement>>; }) { const overrides = linkComponent - ? { ...OVERRIDES, a: { component: linkComponent, ...(linkProps ? { props: linkProps } : {}) } } + ? { ...OVERRIDES, a: { component: linkComponent } } : OVERRIDES; return ( <div className={className}> diff --git a/common/components/citations/CitationCard.tsx b/common/components/citations/CitationCard.tsx @@ -2,6 +2,7 @@ import { Archive, ExternalLink, FileText, Globe, MessageSquareQuote, Mic, Play, import type { CitationKind } from "../../lib/citations/schema"; import { CITATION_KIND_LABELS, + sourceAnchor, spanLabel, type CitationView, type PageCitationView, @@ -20,8 +21,9 @@ import { VerificationBadge } from "./VerificationBadge"; // a play affordance that opens the moment page (its clip) // post the author, date, quote (and the post's text when it says // more), the captured shot when the view has one -// source the document's sentence: its still, the quote, and the -// document's archive links in context +// source the document's sentence: its still, the quote, and +// "from <title>" linking to the document's block on the +// page (sourceAnchor), where its archive links are — once // page the page's title and URL, its archived copy // // Three densities: `card` (a claim's evidence list), `preview` (InlineCite's @@ -80,8 +82,8 @@ function heading(c: CitationView): { title: string; meta: string[] } { }; case "source": return { - title: c.source.title, - meta: [c.source.publisher, c.source.author, c.date ?? c.source.date].filter((x): x is string => !!x), + title: c.sourceTitle, + meta: [c.date].filter((x): x is string => !!x), }; case "page": return { title: c.title ?? hostOf(c.href), meta: [hostOf(c.href), c.date].filter((x): x is string => !!x) }; @@ -119,7 +121,7 @@ function Picture({ c, variant }: { c: CitationView; variant: CitationCardVariant return ( <img src={c.image} - alt={`The sentence as ${c.source.title} shows it: ${c.quote}`} + alt={`The sentence as ${c.sourceTitle} shows it: ${c.quote}`} loading="lazy" className={cn("w-full rounded-md border border-border bg-white object-contain object-left-top", size)} /> @@ -158,15 +160,9 @@ function PostLinks({ c }: { c: PostCitationView }) { function SourceLinks({ c }: { c: SourceCitationView }) { return ( - <> - {c.href && <ExternalA href={c.href}>{c.source.publisher ?? hostOf(c.href)}</ExternalA>} - {c.source.archives.map((a) => ( - <ExternalA key={a.url} href={a.url} title={a.context}> - <Archive className="size-3 shrink-0" aria-hidden /> - {a.label} - </ExternalA> - ))} - </> + <a href={`#${sourceAnchor(c.sourceId)}`} className={linkClass}> + from {c.sourceTitle} + </a> ); } diff --git a/common/components/citations/CitationsContext.tsx b/common/components/citations/CitationsContext.tsx @@ -0,0 +1,31 @@ +"use client"; + +import { createContext, useContext, type ReactNode } from "react"; +import type { CitationView } from "../../lib/report/views"; + +// A PAGE'S CITATIONS, ONCE. Every inline marker (InlineCite) on a report page +// needs its citation's view — the label's href, the number, the preview card. +// Handing each marker its own copy as a prop would serialize that view into the +// page's RSC payload once per marker; a page with hundreds of markers paid for +// it hundreds of times. Instead the page wraps its content in one +// CitationsProvider, whose map is serialized once, and each marker carries only +// its id and reads the view here at render time (on the server for the HTML, +// in the browser for the preview). + +const CitationsContext = createContext<Readonly<Record<string, CitationView>> | null>(null); + +export function CitationsProvider({ + citations, + children, +}: { + citations: Readonly<Record<string, CitationView>>; + children: ReactNode; +}) { + return <CitationsContext.Provider value={citations}>{children}</CitationsContext.Provider>; +} + +// The view of citation `id` from the nearest provider, or undefined. +export function useCitation(id: string | undefined): CitationView | undefined { + const map = useContext(CitationsContext); + return id !== undefined && map && Object.hasOwn(map, id) ? map[id] : undefined; +} diff --git a/common/components/citations/CitedMarkdown.tsx b/common/components/citations/CitedMarkdown.tsx @@ -2,16 +2,19 @@ import type { AnchorHTMLAttributes } from "react"; import { CITE_SCHEME } from "../../lib/citations/inline"; import type { CitationView } from "../../lib/report/views"; import { Markdown } from "../Markdown"; +import { CitationsProvider } from "./CitationsContext"; import { InlineCite } from "./InlineCite"; // Report markdown with its citations live: every `[label](cite:<id>)` link -// (lib/citations/inline.ts) becomes an InlineCite for the citation's view; -// every other link renders as Markdown's own external link. The house markdown -// setup (Markdown.tsx — markdown-to-jsx, raw HTML escaped, token-styled) is +// (lib/citations/inline.ts) becomes an InlineCite for that id; every other +// link renders as Markdown's own external link. The house markdown setup +// (Markdown.tsx — markdown-to-jsx, raw HTML escaped, token-styled) is // unchanged; this only swaps its link component. A `cite:` link inside code is // code, never a citation: markdown-to-jsx makes no link there. // -// Pure: the citations come in as props, keyed by id. +// The views come from the page's one CitationsProvider (a page with many +// markdown blocks serializes its citation map once). A block rendered on its +// own may pass `citations`, and gets a provider of its own. const EXTERNAL_LINK_CLASS = "text-brand underline decoration-brand/40 hover:decoration-brand"; @@ -21,18 +24,10 @@ export function citeIdOf(href: string | undefined): string | null { type Citations = Readonly<Record<string, CitationView>>; -// The `<a>` of cited markdown: one stable component, the citations passed in -// as a prop (Markdown `linkProps`). -function CiteLink({ - href, - children: label, - citations, - ...rest -}: AnchorHTMLAttributes<HTMLAnchorElement> & { citations: Citations }) { +// The `<a>` of cited markdown: one stable component. +function CiteLink({ href, children: label, ...rest }: AnchorHTMLAttributes<HTMLAnchorElement>) { const id = citeIdOf(href); - if (id !== null) { - return <InlineCite citation={Object.hasOwn(citations, id) ? citations[id] : undefined}>{label}</InlineCite>; - } + if (id !== null) return <InlineCite id={id}>{label}</InlineCite>; return ( <a {...rest} href={href} target="_blank" rel="noopener noreferrer" className={EXTERNAL_LINK_CLASS}> {label} @@ -46,12 +41,13 @@ export function CitedMarkdown({ className, }: { children: string; - citations: Citations; + citations?: Citations; className?: string; }) { - return ( - <Markdown className={className} linkComponent={CiteLink} linkProps={{ citations }}> + const md = ( + <Markdown className={className} linkComponent={CiteLink}> {children} </Markdown> ); + return citations ? <CitationsProvider citations={citations}>{md}</CitationsProvider> : md; } diff --git a/common/components/citations/InlineCite.tsx b/common/components/citations/InlineCite.tsx @@ -5,6 +5,7 @@ import { createPortal } from "react-dom"; import { citationAnchor } from "../../lib/citations/inline"; import { CITATION_KIND_LABELS, type CitationView } from "../../lib/report/views"; import { CitationCard } from "./CitationCard"; +import { useCitation } from "./CitationsContext"; // AN INLINE CITATION — what a `[label](cite:<id>)` link in report markdown // renders as (CitedMarkdown): the label, linked to what it cites (the moment @@ -26,8 +27,11 @@ import { CitationCard } from "./CitationCard"; // viewport's width. The number is the keyboard's way in; its reference entry // carries the same card with every link. // -// A ref the page could not resolve (no view for the id — validation keeps a -// published report from having one) renders its label alone. +// THE VIEW comes from the page's CitationsProvider by `id` (one map for the +// page, never a copy per marker in the RSC payload); a caller outside a +// provider may pass the `citation` itself. A ref neither resolves (no view for +// the id — validation keeps a published report from having one) renders its +// label alone. const CLOSE_DELAY_MS = 150; const EDGE_PX = 8; @@ -37,7 +41,17 @@ function isExternal(c: CitationView): boolean { return c.kind === "source" || c.kind === "page"; } -export function InlineCite({ citation, children }: { citation: CitationView | undefined; children: ReactNode }) { +export function InlineCite({ + citation: given, + id, + children, +}: { + citation?: CitationView; + id?: string; + children: ReactNode; +}) { + const fromMap = useCitation(given ? undefined : id); + const citation = given ?? fromMap; const previewId = useId(); const [open, setOpen] = useState(false); const [pos, setPos] = useState<{ top: number; left: number } | null>(null); diff --git a/common/components/citations/citations.test.ts b/common/components/citations/citations.test.ts @@ -5,6 +5,7 @@ import { renderToStaticMarkup } from "react-dom/server"; import type { CitationView, PageCitationView, PostCitationView, SourceCitationView, SpanCitationView } from "../../lib/report/views"; import { CitationCard } from "./CitationCard"; import { CitedMarkdown, citeIdOf } from "./CitedMarkdown"; +import { CitationsProvider } from "./CitationsContext"; import { InlineCite } from "./InlineCite"; import { ReferenceList } from "./ReferenceList"; import { QUOTE_SCORE_TOOLTIP, VerificationBadge, quoteScorePercent } from "./VerificationBadge"; @@ -70,14 +71,8 @@ const source: SourceCitationView = { id: "a01", number: 4, quote: "He opened the bridge himself.", - source: { - id: "s0", - kind: "article", - title: "An example article", - url: "https://example.org/a", - publisher: "Example Gazette", - archives: [{ label: "archive.org", url: "https://web.archive.org/x", context: "as published" }], - }, + sourceId: "s0", + sourceTitle: "An example article", image: "/reports/demo/stills/a01.png", href: "https://example.org/a", }; @@ -122,11 +117,12 @@ test("a post card: author, date, quote, the post's fuller text, the shot", () => assert.match(html, /<img src="\/media\/posts\/demo-social\/123\/shot.png"/); }); -test("a source card: the still, the quote, the document's archive links in context", () => { +test("a source card: the still, the quote, and from-the-document — never its archive links", () => { const html = render(h(CitationCard, { citation: source })); assert.match(html, /<img src="\/reports\/demo\/stills\/a01.png" alt="The sentence as An example article shows it: He opened the bridge himself."/); - assert.match(html, /href="https:\/\/web.archive.org\/x"[^>]*title="as published"/); - assert.ok(html.includes("Example Gazette")); + assert.match(html, /<a href="#source-s0"[^>]*>from An example article<\/a>/); + assert.ok(!html.includes("archive")); + assert.ok(!render(h(CitationCard, { citation: source, variant: "preview" })).includes("archive")); }); test("a page card: title, host, the archived copy, the note", () => { @@ -170,6 +166,23 @@ test("an inline cite: the label links to what it cites, the number to its refere assert.match(render(h(InlineCite, { citation: page, children: "records" })), /href="https:\/\/example.org\/city\/history" target="_blank" rel="noopener noreferrer"/); // an unresolved ref is its label alone assert.equal(render(h(InlineCite, { citation: undefined, children: "label" })), "label"); + assert.equal(render(h(InlineCite, { id: "c01", children: "label" })), "label", "no provider, no view"); +}); + +test("inline cites read their view from the page's one provider, by id", () => { + const html = render( + h(CitationsProvider, { + citations: byId, + children: [ + h(InlineCite, { key: 1, id: "c01", children: "first" }), + h(InlineCite, { key: 2, id: "w01", children: "second" }), + h(InlineCite, { key: 3, id: "missing", children: "third" }), + ], + }), + ); + assert.match(html, /data-inline-cite="c01"[^]*?>first<\/a>[^]*?data-cite-number="1"/); + assert.match(html, /data-inline-cite="w01"[^]*?data-cite-number="5"/); + assert.ok(html.endsWith("third")); }); test("cited markdown: cite links become inline cites, other links stay links, code stays code", () => { diff --git a/common/lib/report/views.test.ts b/common/lib/report/views.test.ts @@ -8,13 +8,18 @@ import { validateReport } from "./validate"; import { MOMENT_PLACEHOLDER_KEY, REPORT_PLACEHOLDER_ID, + CLAIM_ARCHIVES_MAX, + CLAIM_ARCHIVE_MIN_OVERLAP, + archiveOverlap, buildReportPageView, + claimArchives, citedInViews, evidenceClipPath, momentViewPath, orderedCitations, reportAssetPath, reportIndexEntry, + sourceAnchor, sectionCitations, spanLabel, verdictTally, @@ -116,17 +121,67 @@ test("span and post citations link to their moment page, keyed at two decimals", assert.equal(a1.kind === "audio" && a1.poster, undefined); }); -test("a source citation carries its document (never the saved copy) and its published still", () => { +test("a source citation names its document by id; the document (never its saved copy) is in sources once", () => { const s1 = view.citations.s1; assert.equal(s1.kind, "source"); if (s1.kind !== "source") return; assert.equal(s1.image, "/reports/demo/stills/s1.png"); assert.equal(s1.href, "https://example.org/a"); - assert.equal("saved" in s1.source, false); - assert.equal(view.subject?.title, "An article"); - assert.equal(view.subject?.note, "The first edition."); - assert.equal(view.subject && "saved" in view.subject, false); + assert.equal(s1.sourceId, "s0"); + assert.equal(s1.sourceTitle, "An article"); + assert.equal("source" in s1, false); + assert.equal(view.subject, "s0"); + assert.deepEqual(Object.keys(view.sources), ["s0"]); + assert.equal(view.sources.s0.title, "An article"); + assert.equal(view.sources.s0.note, "The first edition."); + assert.equal("saved" in view.sources.s0, false); assert.ok(!JSON.stringify(view).includes("sources/s0/page.html")); + // the archive list is in the view once + assert.equal(JSON.stringify(view).split("https://archive.example.org/a").length - 1, 1); +}); + +test("a claim carries the document's archive links whose context overlaps its sentence, at most five", () => { + const sentence = "The council approved the harbor bridge budget in March after a long debate."; + const archives = [ + { label: "a", url: "https://x/a", context: "the council approved the harbor bridge budget" }, + { label: "b", url: "https://x/b", context: "an unrelated paragraph about weather and gardens" }, + { label: "c", url: "https://x/c", context: "harbor bridge" }, // too few words in common + { label: "d", url: "https://x/d" }, // no context + ...Array.from({ length: 7 }, (_, i) => ({ label: `m${i}`, url: `https://x/m${i}`, context: `council approved budget after debate ${i}` })), + ]; + assert.ok(archiveOverlap(sentence, archives[0].context) >= CLAIM_ARCHIVE_MIN_OVERLAP); + assert.equal(archiveOverlap(sentence, archives[1].context), 0); + assert.equal(archiveOverlap(sentence, archives[2].context), 0); + assert.equal(archiveOverlap(sentence, undefined), 0); + const found = claimArchives(sentence, archives); + assert.equal(found.length, CLAIM_ARCHIVES_MAX); + assert.equal(found[0].label, "a", "kept in the document's order"); + assert.ok(!found.some((a) => a.label === "b" || a.label === "c" || a.label === "d")); + assert.deepEqual(claimArchives(sentence, []), []); +}); + +test("the builder matches a claim's archives against its source sentence, else its text", () => { + const withArchives: Report = { + ...report, + sources: { + s0: { + ...report.sources!.s0, + archives: [ + { label: "in four", url: "https://x/4", context: "four words about something" }, + { label: "elsewhere", url: "https://x/e", context: "nothing like either claim at all" }, + ], + }, + }, + citations: { ...report.citations, s1: { kind: "source", source: "s0", quote: "Four words about something said here." } }, + sections: [ + { id: "one", title: "One", claims: [{ id: "c1", text: "Claim one.", sourceQuote: { citation: "s1" } }] }, + { id: "two", title: "Two", claims: [{ id: "c2", text: "Something about four words, the claim's own text." }] }, + ], + }; + const v = buildReportPageView(withArchives, { record: (c) => record(c.channel, c.id) }); + assert.deepEqual(v.sections[0].claims[0].archives?.map((a) => a.label), ["in four"]); + assert.deepEqual(v.sections[1].claims[0].archives?.map((a) => a.label), ["in four"]); + assert.equal(view.sections[0].claims[0].archives, undefined, "no context, no match"); }); test("a page citation links out; the verdict overrides are laid over the defaults", () => { @@ -197,6 +252,7 @@ test("paths: the clip, the moment view and a report asset", () => { ); assert.equal(momentViewPath("demo-social/123"), "/m/demo-social/123/moment.json"); assert.equal(reportAssetPath("demo", "./stills/x.png"), "/reports/demo/stills/x.png"); + assert.equal(sourceAnchor("s0"), "source-s0"); }); test("the placeholder params can never be a report or a moment", () => { diff --git a/common/lib/report/views.ts b/common/lib/report/views.ts @@ -19,7 +19,8 @@ // // A view carries ONLY what is cited: no `saved` source copy (never published), // no cue beyond the moment's bounded context, no record field a page does not -// show. +// show — and nothing large twice: a document (with its archive links) is in a +// report view's `sources` once, and its citations name it by id. // // `buildReportPageView` is the one way a report becomes its view: the numbers // are lib/report/uses.ts's (first appearance, reading order), the hrefs are @@ -47,9 +48,10 @@ import type { SourceArchive, SpanCitation, } from "../citations/schema"; +import { quoteTokens } from "../citations/verify"; import { formatTimestamp } from "../vtt"; import type { CitedIn } from "./citedIn"; -import type { Report, ReportKind } from "./schema"; +import type { Claim, Report, ReportKind } from "./schema"; import { reportCitationNumbers } from "./uses"; import { resolveVerdicts, VERDICTS, type Verdict, type VerdictStyle } from "./verdicts"; @@ -171,9 +173,16 @@ export type PostCitationView = CitationViewCommon & { thread?: boolean; }; +// A sentence of a document. The document itself (its byline and its archive +// links — a long article may carry hundreds) is in the view's `sources` map +// ONCE, never copied into each citation of it: a report citing one article +// eighty times would otherwise carry its archive list eighty times. export type SourceCitationView = CitationViewCommon & { kind: "source"; - source: SourceView; + // The document's id in the view's `sources`, and its title for the card's + // "from <title>" (sourceAnchor links to the document's block on the page). + sourceId: string; + sourceTitle: string; // The still of the sentence, published (reportAssetPath). image?: string; // The document itself, when it has a URL. @@ -198,6 +207,10 @@ export type ClaimView = { verdict?: Verdict; // The id of the `source` citation holding the document's own sentence. sourceQuote?: string; + // The document's archive links that sit in this claim's sentence (their + // `context` overlaps it: claimArchives), at most CLAIM_ARCHIVES_MAX. The + // document's full list is shown once, with the document. + archives?: SourceArchive[]; findings?: string; citations: string[]; }; @@ -219,7 +232,11 @@ export type ReportPageView = { summary?: string; published?: string; updated?: string; - subject?: SourceView; + // The document under review: its id in `sources`. + subject?: string; + // Every document the report's `source` citations quote, and the subject, + // by id — each once. + sources: Record<string, SourceView>; // Every verdict's label and colour, the report's overrides applied. verdicts: Record<Verdict, VerdictStyle>; // Every citation the report cites, numbered; one it defines but never cites @@ -405,7 +422,8 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver) citations[id] = defined({ kind: "source" as const, ...common, - source: sourceView(c.source, s), + sourceId: c.source, + sourceTitle: s.title, image: c.image ? reportAssetPath(report.id, c.image) : undefined, href: s.url, }); @@ -422,7 +440,23 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver) break; } } - const subject = report.subject && sources[report.subject.source]; + // Each document once: the subject, then every document a cited source + // citation quotes. + const sourceViews: Record<string, SourceView> = {}; + const subjectId = report.subject && sources[report.subject.source] ? report.subject.source : undefined; + if (subjectId) sourceViews[subjectId] = sourceView(subjectId, sources[subjectId]); + for (const c of Object.values(citations)) { + if (c.kind === "source" && !sourceViews[c.sourceId]) sourceViews[c.sourceId] = sourceView(c.sourceId, sources[c.sourceId]); + } + // A claim's archive links: those of the document its sentence is from (the + // subject when it names none) whose context overlaps the sentence. + const archivesFor = (claim: Claim): SourceArchive[] | undefined => { + const sq = claim.sourceQuote ? citations[claim.sourceQuote.citation] : undefined; + const doc = sq?.kind === "source" ? sourceViews[sq.sourceId] : subjectId ? sourceViews[subjectId] : undefined; + if (!doc || doc.archives.length === 0) return undefined; + const found = claimArchives(sq?.quote ?? claim.text, doc.archives); + return found.length > 0 ? found : undefined; + }; return defined({ format: REPORT_PAGE_FORMAT, version: REPORT_VIEWS_VERSION, @@ -433,7 +467,8 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver) summary: report.summary, published: report.published, updated: report.updated, - subject: subject ? sourceView(report.subject!.source, subject) : undefined, + subject: subjectId, + sources: sourceViews, verdicts: resolveVerdicts(report.verdicts), citations, sections: report.sections.map((s) => @@ -448,6 +483,7 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver) text: c.text, verdict: c.verdict, sourceQuote: c.sourceQuote?.citation, + archives: archivesFor(c), findings: c.findings, citations: c.citations ?? [], }), @@ -458,8 +494,48 @@ export function buildReportPageView(report: Report, resolve: ReportViewResolver) }); } +// ─── A claim's archive links ─── + +// A document's archive link belongs to a claim when the words around it (its +// `context`, as the document had them) overlap the claim's sentence: at least +// CLAIM_ARCHIVE_MIN_SHARED distinct words of four letters or more in common, +// and those at least CLAIM_ARCHIVE_MIN_OVERLAP of the shorter side's. Words +// are normalised as quote verification normalises them (verify.ts +// quoteTokens); short words (the, and, of) are left out, they match anything. +// The best CLAIM_ARCHIVES_MAX are kept, in the document's order. +export const CLAIM_ARCHIVE_MIN_OVERLAP = 0.6; +export const CLAIM_ARCHIVE_MIN_SHARED = 3; +export const CLAIM_ARCHIVES_MAX = 5; + +const contentWords = (text: string) => new Set(quoteTokens(text).filter((t) => t.length >= 4)); + +export function archiveOverlap(sentence: string, context: string | undefined): number { + if (!context) return 0; + const a = contentWords(sentence); + const b = contentWords(context); + let shared = 0; + for (const t of b) if (a.has(t)) shared++; + if (shared < CLAIM_ARCHIVE_MIN_SHARED) return 0; + return shared / Math.min(a.size, b.size); +} + +export function claimArchives(sentence: string, archives: readonly SourceArchive[]): SourceArchive[] { + return archives + .map((a, i) => ({ a, i, score: archiveOverlap(sentence, a.context) })) + .filter((x) => x.score >= CLAIM_ARCHIVE_MIN_OVERLAP) + .sort((x, y) => y.score - x.score || x.i - y.i) + .slice(0, CLAIM_ARCHIVES_MAX) + .sort((x, y) => x.i - y.i) + .map((x) => x.a); +} + // ─── Reading a view ─── +// The anchor (fragment id, without `#`) of a document's block on a report page. +export function sourceAnchor(sourceId: string): string { + return `source-${sourceId}`; +} + // The report's citations in number order. export function orderedCitations(view: Pick<ReportPageView, "citations">): CitationView[] { return Object.values(view.citations) diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md @@ -2,7 +2,7 @@ ## [Unreleased] - **`corpus.json` is spec 5: it names a site's reports, and a site that publishes only reports says so.** A site with reports adds `reports` to its `corpus.json` (`index`: `/reports/index.json`, the count, and how to read a report's page, its citations and its moment pages) and a Reports section to `llms.txt`; its sitemap lists the report and moment pages. A site that publishes only its reports has `"scope": "cited"` and its audience under `site`, no channels and zero totals, an `llms.txt` that lists its reports and how their citations and moment pages are read, and a `site.json` with no channels. A reader that does not know spec 5 sees an empty corpus there. Needs a rebuild and deploy of each site. -- **A site can show cited reports, and every citation opens on a page of its own.** A site built with reports has a **Reports** link in its header and a page at `/reports/` listing them. A report's page has its title, subtitle, dates and the document under review with its archive links; a fact-check's tally of verdicts; the summary; the sections and their claims, each with its verdict, the document's own sentence as an image, the findings and the evidence cards; a numbered reference list; and links to download its citations as JSON and CSV. A citation in the text shows as its words plus a number: hovering it, focusing the number or tapping it once shows a card of the citation (the quote, who said it and when, a picture or the post's screenshot, and how closely the quote matched the transcript when it was checked); the words open what it cites and the number jumps to its reference. A cited span of a video or audio record opens at `/m/<channel>/<id>/<start>-<end>/`: a short clip of the span with a little context either side, the quote, the transcript lines around it, the record's title, channel and date, a link to the original at that time, and every report on the site that cites it. A cited post opens at `/m/<channel>/<id>/` with its screenshot and text. A site that publishes only its reports (`site.json` `publish: "cited"`) opens on the report index and has no search, Ask AI, downloads or duplicates. A site with no reports is unchanged. Needs a rebuild and deploy of each site. +- **A site can show cited reports, and every citation opens on a page of its own.** A site built with reports has a **Reports** link in its header and a page at `/reports/` listing them. A report's page has its title, subtitle, dates and the document under review, its archive links listed once under it and folded away ("N archive links in context"); a fact-check's tally of verdicts; the summary; the sections and their claims, each with its verdict, the document's own sentence as an image with a link back to the document and only the archive links that sit in that sentence (at most five), the findings and the evidence cards; a numbered reference list; and links to download its citations as JSON and CSV. A citation in the text shows as its words plus a number: hovering it, focusing the number or tapping it once shows a card of the citation (the quote, who said it and when, a picture or the post's screenshot, and how closely the quote matched the transcript when it was checked); the words open what it cites and the number jumps to its reference. A cited span of a video or audio record opens at `/m/<channel>/<id>/<start>-<end>/`: a short clip of the span with a little context either side, the quote, the transcript lines around it, the record's title, channel and date, a link to the original at that time, and every report on the site that cites it. A cited post opens at `/m/<channel>/<id>/` with its screenshot and text. A site that publishes only its reports (`site.json` `publish: "cited"`) opens on the report index and has no search, Ask AI, downloads or duplicates. A site with no reports is unchanged. Needs a rebuild and deploy of each site. ## [0.11.1] - 2026-10-01 - **Use with AI goes to the Archilyzer site's AI and MCP doc; the page on each site is gone.** The header's, the slide-out menu's, the footer's and Ask AI's **Use with AI** keep their label and open https://archilyzer.pages.dev/docs/ai-and-mcp/ in the same tab, on every site and the hub, where one block says how to run Claude Code against any archive (the source, `pnpm install`, `claude mcp add archilyzer`, `/ask`). `/use-with-ai/` is no longer built. `corpus.json`'s `useWithAi` names the doc; `llms.txt`'s Ask AI section lists the site's `/ask/` chat and the doc; the sitemap drops `/use-with-ai`. Needs a rebuild and deploy of each site and the hub. diff --git a/export/app/components/reports/ReportArticle.tsx b/export/app/components/reports/ReportArticle.tsx @@ -1,17 +1,20 @@ import { ArrowDownToLine } from "lucide-react"; import { CitationCard } from "yt-dlp-transcript-common/components/citations/CitationCard"; +import { CitationsProvider } from "yt-dlp-transcript-common/components/citations/CitationsContext"; import { CitedMarkdown } from "yt-dlp-transcript-common/components/citations/CitedMarkdown"; import { ReferenceList } from "yt-dlp-transcript-common/components/citations/ReferenceList"; import { VerdictChip, VerdictTally } from "yt-dlp-transcript-common/components/report/VerdictChip"; +import type { SourceArchive } from "yt-dlp-transcript-common/lib/citations/schema"; import { orderedCitations, + sourceAnchor, verdictTally, type CitationView, type ClaimView, type ReportPageView, type SourceCitationView, } from "yt-dlp-transcript-common/lib/report/views"; -import { ExternalLinkText, Eyebrow, SourceBlock, dateLabel, textLink } from "./parts"; +import { ArchiveList, Eyebrow, SourceBlock, dateLabel, textLink } from "./parts"; // ONE REPORT, from its view (common/lib/report/views.ts): the header (title, // subtitle, dates, the document under review with its archive links), a @@ -22,7 +25,7 @@ import { ExternalLinkText, Eyebrow, SourceBlock, dateLabel, textLink } from "./p const proseClass = "text-[0.95rem] leading-relaxed text-foreground"; -function SourceSentence({ c }: { c: SourceCitationView }) { +function SourceSentence({ c, archives }: { c: SourceCitationView; archives: readonly SourceArchive[] | undefined }) { return ( <figure data-source-sentence={c.id} className="flex flex-col gap-1.5"> {c.image ? ( @@ -38,13 +41,13 @@ function SourceSentence({ c }: { c: SourceCitationView }) { <q>{c.quote}</q> </blockquote> )} - <figcaption className="flex flex-wrap items-baseline gap-x-3 gap-y-1 text-xs text-muted-foreground"> - <span>{c.source.title}</span> - {c.source.archives.map((a) => ( - <ExternalLinkText key={a.url} href={a.url} title={a.context}> - {a.label} - </ExternalLinkText> - ))} + <figcaption className="flex flex-col gap-1 text-xs text-muted-foreground"> + <a href={`#${sourceAnchor(c.sourceId)}`} className={textLink}> + from {c.sourceTitle} + </a> + {/* Only the document's archive links that sit in this sentence; the + full list is with the document, once. */} + {archives && archives.length > 0 && <ArchiveList archives={archives} />} </figcaption> </figure> ); @@ -83,9 +86,9 @@ function Claim({ {subjectLabel}: <q className="text-foreground">{claim.text}</q> </p> )} - {sentence?.kind === "source" && <SourceSentence c={sentence} />} + {sentence?.kind === "source" && <SourceSentence c={sentence} archives={claim.archives} />} {claim.findings && ( - <CitedMarkdown citations={view.citations} className={proseClass}> + <CitedMarkdown className={proseClass}> {claim.findings} </CitedMarkdown> )} @@ -107,97 +110,116 @@ export default function ReportArticle({ view }: { view: ReportPageView }) { const isFactcheck = view.kind === "factcheck"; const tally = isFactcheck ? verdictTally(view) : []; const references = orderedCitations(view); - const subjectLabel = view.subject?.kind === "article" ? "The article says" : "The source says"; + const subject = view.subject ? view.sources[view.subject] : undefined; + const subjectLabel = subject?.kind === "article" ? "The article says" : "The source says"; const dates = [ view.published ? `Published ${dateLabel(view.published)}` : null, view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null, ].filter(Boolean); const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0); + // Documents quoted that are not the subject, each shown once before the + // references. + const otherSources = Object.values(view.sources).filter((s) => s.id !== view.subject); return ( - <article data-report={view.id} className="mx-auto flex w-full max-w-3xl flex-col gap-8"> - <header className="flex flex-col gap-3 border-b border-border pb-6"> - <Eyebrow>{isFactcheck ? "Fact-check" : "Report"}</Eyebrow> - <h1 className="font-display text-3xl font-semibold leading-tight tracking-tight text-foreground sm:text-4xl"> - {view.title} - </h1> - {view.subtitle && <p className="text-lg text-muted-foreground">{view.subtitle}</p>} - {dates.length > 0 && <p className="font-mono text-xs text-muted-foreground">{dates.join(" · ")}</p>} - {view.subject && <SourceBlock source={view.subject} label="Under review" />} - {tally.length > 0 && ( - <div className="flex flex-col gap-2"> - <p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground"> - {claimCount} claim{claimCount === 1 ? "" : "s"} checked - </p> - <VerdictTally tally={tally} styles={view.verdicts} /> - </div> + // One citation map for every inline marker on the page (CitationsContext): + // serialized once, not once per marker. + <CitationsProvider citations={view.citations}> + <article data-report={view.id} className="mx-auto flex w-full max-w-3xl flex-col gap-8"> + <header className="flex flex-col gap-3 border-b border-border pb-6"> + <Eyebrow>{isFactcheck ? "Fact-check" : "Report"}</Eyebrow> + <h1 className="font-display text-3xl font-semibold leading-tight tracking-tight text-foreground sm:text-4xl"> + {view.title} + </h1> + {view.subtitle && <p className="text-lg text-muted-foreground">{view.subtitle}</p>} + {dates.length > 0 && <p className="font-mono text-xs text-muted-foreground">{dates.join(" · ")}</p>} + {subject && <SourceBlock source={subject} label="Under review" />} + {tally.length > 0 && ( + <div className="flex flex-col gap-2"> + <p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground"> + {claimCount} claim{claimCount === 1 ? "" : "s"} checked + </p> + <VerdictTally tally={tally} styles={view.verdicts} /> + </div> + )} + </header> + + {view.summary && ( + <CitedMarkdown className={proseClass}> + {view.summary} + </CitedMarkdown> )} - </header> - {view.summary && ( - <CitedMarkdown citations={view.citations} className={proseClass}> - {view.summary} - </CitedMarkdown> - )} + {view.sections.length > 1 && ( + <nav aria-label="Sections" className="rounded-lg border border-border p-4 text-sm"> + <ol className="flex list-decimal flex-col gap-1 pl-5 marker:text-muted-foreground"> + {view.sections.map((s) => ( + <li key={s.id}> + <a href={`#${s.id}`} className={textLink}> + {s.title} + </a> + {s.claims.length > 0 && ( + <span className="ml-2 font-mono text-xs text-muted-foreground"> + {s.claims.length} claim{s.claims.length === 1 ? "" : "s"} + </span> + )} + </li> + ))} + </ol> + </nav> + )} - {view.sections.length > 1 && ( - <nav aria-label="Sections" className="rounded-lg border border-border p-4 text-sm"> - <ol className="flex list-decimal flex-col gap-1 pl-5 marker:text-muted-foreground"> - {view.sections.map((s) => ( - <li key={s.id}> - <a href={`#${s.id}`} className={textLink}> - {s.title} - </a> - {s.claims.length > 0 && ( - <span className="ml-2 font-mono text-xs text-muted-foreground"> - {s.claims.length} claim{s.claims.length === 1 ? "" : "s"} - </span> - )} - </li> + {view.sections.map((s) => ( + <section key={s.id} id={s.id} data-section={s.id} className="flex scroll-mt-20 flex-col gap-4"> + <h2 className="font-display text-2xl font-semibold tracking-tight text-foreground">{s.title}</h2> + {s.body && ( + <CitedMarkdown className={proseClass}> + {s.body} + </CitedMarkdown> + )} + {s.claims.map((c) => ( + <Claim key={c.id} claim={c} view={view} subjectLabel={subjectLabel} /> ))} - </ol> - </nav> - )} + </section> + ))} - {view.sections.map((s) => ( - <section key={s.id} id={s.id} data-section={s.id} className="flex scroll-mt-20 flex-col gap-4"> - <h2 className="font-display text-2xl font-semibold tracking-tight text-foreground">{s.title}</h2> - {s.body && ( - <CitedMarkdown citations={view.citations} className={proseClass}> - {s.body} - </CitedMarkdown> - )} - {s.claims.map((c) => ( - <Claim key={c.id} claim={c} view={view} subjectLabel={subjectLabel} /> - ))} - </section> - ))} + {otherSources.length > 0 && ( + <section aria-labelledby="sources-heading" className="flex flex-col gap-3 border-t border-border pt-6"> + <h2 id="sources-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground"> + Sources + </h2> + {otherSources.map((s) => ( + <SourceBlock key={s.id} source={s} label="Quoted" /> + ))} + </section> + )} - {references.length > 0 && ( - <section id="references" aria-labelledby="references-heading" className="flex scroll-mt-20 flex-col gap-2 border-t border-border pt-6"> - <h2 id="references-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground"> - References - </h2> - <ReferenceList citations={references} /> - </section> - )} + {references.length > 0 && ( + <section id="references" aria-labelledby="references-heading" className="flex scroll-mt-20 flex-col gap-2 border-t border-border pt-6"> + <h2 id="references-heading" className="font-display text-2xl font-semibold tracking-tight text-foreground"> + References + </h2> + <ReferenceList citations={references} /> + </section> + )} - {(view.downloads?.json || view.downloads?.csv) && ( - <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground"> - <ArrowDownToLine className="size-4" aria-hidden /> - Download citations: - {view.downloads.json && ( - <a href={view.downloads.json} download className={textLink}> - JSON - </a> - )} - {view.downloads.csv && ( - <a href={view.downloads.csv} download className={textLink}> - CSV - </a> - )} - </p> - )} - </article> + {(view.downloads?.json || view.downloads?.csv) && ( + <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground"> + <ArrowDownToLine className="size-4" aria-hidden /> + Download citations: + {view.downloads.json && ( + <a href={view.downloads.json} download className={textLink}> + JSON + </a> + )} + {view.downloads.csv && ( + <a href={view.downloads.csv} download className={textLink}> + CSV + </a> + )} + </p> + )} + </article> + </CitationsProvider> ); } diff --git a/export/app/components/reports/parts.tsx b/export/app/components/reports/parts.tsx @@ -1,6 +1,7 @@ import Link from "next/link"; -import { Archive, ExternalLink } from "lucide-react"; -import type { SourceView } from "yt-dlp-transcript-common/lib/report/views"; +import { ExternalLink } from "lucide-react"; +import { sourceAnchor, type SourceView } from "yt-dlp-transcript-common/lib/report/views"; +import type { SourceArchive } from "yt-dlp-transcript-common/lib/citations/schema"; // Small pieces the report, index and moment pages share. @@ -26,35 +27,55 @@ export function ExternalLinkText({ href, children, title }: { href: string; chil ); } -// A document under review: its title (linked), byline, note, and its archive -// links in the context the document gave them. +// A document a report quotes: its title (linked), byline, note, and its +// archive links in the context the document gave them — collapsed, since a +// long document may carry hundreds. The block is the document's one place on +// the page: its anchor is what a source citation's "from <title>" links to. export function SourceBlock({ source, label }: { source: SourceView; label: string }) { const byline = [source.publisher, source.author, dateLabel(source.date)].filter(Boolean).join(" · "); + const n = source.archives.length; return ( - <div data-subject-source={source.id} className="flex flex-col gap-1.5 rounded-lg border border-border bg-card p-4 text-sm"> + <div + id={sourceAnchor(source.id)} + data-subject-source={source.id} + className="flex scroll-mt-20 flex-col gap-1.5 rounded-lg border border-border bg-card p-4 text-sm" + > <p className="font-mono text-[10px] uppercase tracking-[0.16em] text-muted-foreground">{label}</p> <p className="font-medium leading-snug text-foreground"> {source.url ? <ExternalLinkText href={source.url}>{source.title}</ExternalLinkText> : source.title} </p> {byline && <p className="font-mono text-xs text-muted-foreground">{byline}</p>} {source.note && <p className="text-xs text-muted-foreground">{source.note}</p>} - {source.archives.length > 0 && ( - <ul className="mt-1 flex flex-col gap-1 text-xs"> - {source.archives.map((a) => ( - <li key={a.url} className="flex flex-wrap items-baseline gap-x-2"> - <ExternalLinkText href={a.url}> - <Archive className="size-3 shrink-0" aria-hidden /> - {a.label} - </ExternalLinkText> - {a.context && <span className="text-muted-foreground">— {a.context}</span>} - </li> - ))} - </ul> + {n > 0 && ( + <details data-source-archives={n} className="mt-1 text-xs"> + <summary className="cursor-pointer text-muted-foreground hover:text-foreground"> + {`${n} archive link${n === 1 ? "" : "s"} in context`} + </summary> + <ArchiveList archives={source.archives} className="mt-2" /> + </details> )} </div> ); } +// Archive links, each with the words around it in the document. A document +// may carry hundreds, so a link is text with a CSS arrow, not two inline SVG +// icons apiece. +export function ArchiveList({ archives, className }: { archives: readonly SourceArchive[]; className?: string }) { + return ( + <ul className={`flex flex-col gap-1 ${className ?? ""}`}> + {archives.map((a, i) => ( + <li key={`${i}:${a.url}`} className="flex flex-wrap items-baseline gap-x-2"> + <a href={a.url} target="_blank" rel="noopener noreferrer" className={`${textLink} after:ml-0.5 after:content-['↗']`}> + {a.label} + </a> + {a.context && <span className="text-muted-foreground">— {a.context}</span>} + </li> + ))} + </ul> + ); +} + export function EmptyState({ title, children }: { title: string; children: React.ReactNode }) { return ( <div data-reports-empty="" className="mx-auto flex max-w-2xl flex-col gap-3"> diff --git a/export/app/lib/reportSize.test.ts b/export/app/lib/reportSize.test.ts @@ -0,0 +1,48 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import * as React from "react"; +import { renderToString } from "react-dom/server"; +import { syntheticReport, syntheticReportView } from "../../fixtures/report-site/synthetic"; +import { validateReport } from "yt-dlp-transcript-common/lib/report/validate"; +import { CLAIM_ARCHIVES_MAX } from "yt-dlp-transcript-common/lib/report/views"; +import ReportArticle from "../components/reports/ReportArticle"; + +// A REPORT PAGE'S SIZE is bounded by what it cites, not by what it repeats. A +// report citing one long document many times once carried that document — and +// its hundreds of archive links — inside every citation of it, and rendered +// them on every card and preview: a static host's per-file limit refused the +// page. The synthetic report (fixtures/report-site/synthetic.ts: 400 citations, +// 100 of them of ONE source with 300 archive links) holds the budget. + +(globalThis as { React?: typeof React }).React = React; + +const MB = 1024 * 1024; +const report = syntheticReport(); +const view = syntheticReportView(report); + +test("the synthetic report is sound and as large as it says", () => { + assert.deepEqual(validateReport(report), []); + const kinds = Object.values(view.citations).reduce<Record<string, number>>((n, c) => ((n[c.kind] = (n[c.kind] ?? 0) + 1), n), {}); + assert.deepEqual(kinds, { source: 100, video: 200, post: 100 }); + assert.equal(view.sources.s0.archives.length, 300); +}); + +test("the page view is under 2 MB, and carries the document once", () => { + const json = JSON.stringify(view, null, 2); + assert.ok(Buffer.byteLength(json) < 2 * MB, `page.json is ${Buffer.byteLength(json)} bytes`); + // every archive URL appears once in the document's entry, and otherwise only + // in the claims it belongs to (at most CLAIM_ARCHIVES_MAX each) + const first = view.sources.s0.archives[0].url; + const claimsWith = view.sections.flatMap((s) => s.claims).filter((c) => c.archives?.some((a) => a.url === first)).length; + assert.equal(json.split(first + '"').length - 1, 1 + claimsWith); + for (const c of view.sections.flatMap((s) => s.claims)) assert.ok((c.archives?.length ?? 0) <= CLAIM_ARCHIVES_MAX); + for (const c of Object.values(view.citations)) assert.ok(!("source" in c), "a source citation names its document by id"); +}); + +test("the rendered report page is under 5 MB", () => { + const html = renderToString(React.createElement(ReportArticle, { view })); + assert.ok(Buffer.byteLength(html) < 5 * MB, `the page renders to ${Buffer.byteLength(html)} bytes`); + // the document's full archive list is on the page once, collapsed + assert.equal(html.match(/data-source-archives="300"/g)?.length, 1); + assert.match(html, /<details data-source-archives="300"[^>]*><summary[^>]*>300 archive links in context<\/summary>/); +}); diff --git a/export/app/lib/reports.test.ts b/export/app/lib/reports.test.ts @@ -85,6 +85,10 @@ test("the report page: header, tally, sections and claims, inline cites, referen assert.ok(html.includes("Published 2026-10-01 · Updated 2026-10-04")); assert.match(html, /data-subject-source="s0"/); assert.ok(html.includes("as published on the day")); + // the document's archive links are listed once, collapsed, under its anchor + assert.match(html, /<div id="source-s0"[^>]*>[^]*?<details data-source-archives="1"/); + // a claim's source sentence links to the document instead of repeating them + assert.match(html, /data-source-sentence="a01"[^]*?<a href="#source-s0"[^>]*>from An example article about a demo channel<\/a>/); assert.match(html, /data-verdict-tally=""/); for (const v of ["CORROBORATED", "PARTLY", "CONTRADICTED", "UNTESTABLE"]) assert.match(html, new RegExp(`data-verdict="${v}"`)); assert.ok(html.includes("Read in the edition published on the day"), "the source's note, under its byline"); diff --git a/export/fixtures/report-site/public/reports/demo-factcheck/page.json b/export/fixtures/report-site/public/reports/demo-factcheck/page.json @@ -8,22 +8,25 @@ "summary": "The article makes four claims. The recordings support one, partly support another, and contradict a third; the fourth cannot be tested. The host said the opening date out loud [on stream](cite:c01), and later [repeated it](cite:au1).", "published": "2026-10-01", "updated": "2026-10-04", - "subject": { - "id": "s0", - "kind": "article", - "title": "An example article about a demo channel", - "url": "https://example.org/articles/demo", - "publisher": "Example Gazette", - "author": "A. Writer", - "date": "2026-09-20", - "note": "Read in the edition published on the day; it has since been revised.", - "archives": [ - { - "label": "archive.org", - "url": "https://web.archive.org/web/2026/https://example.org/articles/demo", - "context": "as published on the day" - } - ] + "subject": "s0", + "sources": { + "s0": { + "id": "s0", + "kind": "article", + "title": "An example article about a demo channel", + "url": "https://example.org/articles/demo", + "publisher": "Example Gazette", + "author": "A. Writer", + "date": "2026-09-20", + "note": "Read in the edition published on the day; it has since been revised.", + "archives": [ + { + "label": "archive.org", + "url": "https://web.archive.org/web/2026/https://example.org/articles/demo", + "context": "as published on the day" + } + ] + } }, "verdicts": { "CORROBORATED": { @@ -102,23 +105,8 @@ "id": "a01", "number": 3, "quote": "He opened the bridge himself in 2018.", - "source": { - "id": "s0", - "kind": "article", - "title": "An example article about a demo channel", - "url": "https://example.org/articles/demo", - "publisher": "Example Gazette", - "author": "A. Writer", - "date": "2026-09-20", - "note": "Read in the edition published on the day; it has since been revised.", - "archives": [ - { - "label": "archive.org", - "url": "https://web.archive.org/web/2026/https://example.org/articles/demo", - "context": "as published on the day" - } - ] - }, + "sourceId": "s0", + "sourceTitle": "An example article about a demo channel", "image": "/reports/demo-factcheck/stills/a01.png", "href": "https://example.org/articles/demo" }, @@ -137,23 +125,8 @@ "id": "a02", "number": 5, "quote": "He has said many times that he would move away.", - "source": { - "id": "s0", - "kind": "article", - "title": "An example article about a demo channel", - "url": "https://example.org/articles/demo", - "publisher": "Example Gazette", - "author": "A. Writer", - "date": "2026-09-20", - "note": "Read in the edition published on the day; it has since been revised.", - "archives": [ - { - "label": "archive.org", - "url": "https://web.archive.org/web/2026/https://example.org/articles/demo", - "context": "as published on the day" - } - ] - }, + "sourceId": "s0", + "sourceTitle": "An example article about a demo channel", "href": "https://example.org/articles/demo" }, "p01": { diff --git a/export/fixtures/report-site/synthetic.ts b/export/fixtures/report-site/synthetic.ts @@ -0,0 +1,111 @@ +// A SYNTHETIC LARGE REPORT, for the page-size test (app/lib/reportSize.test.ts) +// and for measuring a built page: 400 citations — 100 `source` citations of ONE +// source that carries 300 archive links, 200 video spans and 100 posts — over +// 100 claims in 10 sections, each claim with its source sentence, findings +// citing two spans and a post inline, and those listed under it. +// +// Neutral generated text; the shape is what matters. + +import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; +import { buildReportPageView, type RecordView, type ReportPageView } from "yt-dlp-transcript-common/lib/report/views"; + +export const SYNTHETIC_REPORT_ID = "synthetic-large"; + +const WORDS = [ + "harbor", "ledger", "council", "bridge", "orchard", "lantern", "meadow", "quarry", "signal", "timber", + "valley", "warden", "beacon", "canyon", "delta", "ember", "falcon", "garnet", "hollow", "island", +]; + +// Claim i's sentence: words chosen by i, so claims differ and archive +// contexts can be built to overlap exactly one of them. +export function syntheticSentence(i: number): string { + const w = (k: number) => WORDS[(i * 7 + k * 3) % WORDS.length]; + return `In paragraph ${i} the article says the ${w(0)} ${w(1)} met the ${w(2)} ${w(3)} near the ${w(4)} in ${2000 + (i % 25)}.`; +} + +export function syntheticReport(): Report { + const archives = Array.from({ length: 300 }, (_, k) => ({ + label: `archive ${k + 1}`, + url: `https://web.archive.org/web/2026/https://example.org/ref/${k + 1}`, + // Each archive link sits in claim (k % 100)'s sentence. + context: syntheticSentence(k % 100), + })); + const citations: NonNullable<Report["citations"]> = {}; + for (let i = 0; i < 100; i++) { + citations[`a${i}`] = { kind: "source", source: "s0", quote: syntheticSentence(i), image: `stills/a${i}.png` }; + } + for (let i = 0; i < 200; i++) { + citations[`v${i}`] = { + kind: "video", + channel: "demo-channel", + id: `vid${i % 40}`, + start: 60 + i * 30, + end: 80 + i * 30, + pad: { before: 5, after: 5 }, + quote: `Span ${i}: and that is when I said the ${WORDS[i % 20]} was never part of the plan, not once, not ever, and I stand by it today.`, + verification: { quoteScore: 0.95, quoteCheckedAt: "2026-10-04T12:00:00Z", method: "token recall v1" }, + }; + } + for (let i = 0; i < 100; i++) { + citations[`p${i}`] = { + kind: "post", + channel: "demo-social", + id: `${1000000 + i}`, + quote: `Post ${i}: the ${WORDS[i % 20]} thing again? I covered it already, read the thread.`, + }; + } + const sections: Report["sections"] = Array.from({ length: 10 }, (_, s) => ({ + id: `section-${s}`, + title: `Section ${s}`, + body: `Section ${s} looks at ten claims.`, + claims: Array.from({ length: 10 }, (_, c) => { + const i = s * 10 + c; + return { + id: `claim-${i}`, + title: `Claim ${i}`, + text: syntheticSentence(i), + verdict: (["CORROBORATED", "PARTLY", "CONTRADICTED", "NOT_FOUND", "UNTESTABLE"] as const)[i % 5], + sourceQuote: { citation: `a${i}` }, + findings: `The recordings say otherwise [here](cite:v${2 * i}) and [here](cite:v${2 * i + 1}); a [post](cite:p${i}) agrees.`, + citations: [`v${2 * i}`, `v${2 * i + 1}`, `p${i}`], + }; + }), + })); + return { + format: "archilyzer-report", + version: 1, + id: SYNTHETIC_REPORT_ID, + kind: "factcheck", + title: "A synthetic large fact-check", + summary: "Four hundred citations of one source, spans and posts.", + subject: { source: "s0" }, + sources: { + s0: { + kind: "article", + title: "A long example article", + url: "https://example.org/long", + publisher: "Example Gazette", + archives, + }, + }, + citations, + sections, + }; +} + +const record = (channel: string, id: string): RecordView => ({ + channel, + channelTitle: "Demo Channel", + id, + title: `Demo record ${id}`, + date: "2026-01-10", + platform: "youtube", + originalUrl: `https://media.example.org/${channel}/${id}`, +}); + +export function syntheticReportView(report = syntheticReport()): ReportPageView { + return buildReportPageView(report, { + record: (c) => record(c.channel, c.id), + post: (c) => ({ author: "@demo_account", text: `${c.quote}\n\nMore words after the quote.` }), + }); +}