commit 688e17cceb9f4d63108977c7c11a0207de5833c1 parent c5c067b5259e16380334d8a123a6f712f9fcbfe3 Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st> Date: Mon, 5 Oct 2026 14:02:47 -0400 reports: the reports-export job, the Reports tab's Exports, and the downloads line A `reports-export` job (POST /api/ops/reports-export {siteId, reportId?, formats?}, `pnpm ops reports-export`, replayable) runs the export on prepare's queue, since it reads the media cache a prepare writes and prunes; the prepare job ends by exporting when nothing is missing. The Reports tab gains an Exports section: the Export reports button, the last export job, and per published report the files of its last export with their sizes, which the next build publishes, which stay local, and why. A report page's download line lists what compose published: HTML, PDF, Markdown, Evidence pack, Citations JSON, CSV. The report-site fixture lists every export, the e2e stage writes them with the host step's own writer, and serve-out serves .pdf. Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com> Diffstat:
22 files changed, 577 insertions(+), 45 deletions(-)
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -86,6 +86,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { "persist-videos": { label: "Persist videos", drainable: true }, // A report site's evidence media: one pass, cancelled rather than drained. "reports-prepare": { label: "Prepare report media", drainable: false }, + // A report site's exports: one pass, cancelled rather than drained. + "reports-export": { label: "Export reports", drainable: false }, }; test("added kinds carry their pinned label and drainability", () => { diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -666,6 +666,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = { queueKeyStrategy: "custom", needsMedia: true, }, + // A REPORT SITE'S EXPORTS (publish/reportExports.ts): each published report + // as report.html, report.pdf, report.md and an evidence pack, into the + // site's report-exports staging, for compose to publish. It reads the + // reports' records (text) and the prepared media cache — never a channel's + // big files — so `needsText`; like prepare it spans channels with no + // `channelSlug`, and compose's own check reports an unreadable channel. On + // prepare's queue (`reports-prepare`): it reads the cache prepare writes and + // prunes. Replayable: the spec is the site (and the report and formats). + "reports-export": { + kind: "reports-export", + label: "Export reports", + drainable: false, + replayable: true, + queueKeyStrategy: "custom", + needsMedia: false, + needsText: true, + }, // THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM). // Each runs with a channelSlug and writes under `data/<id>/` — a single // video's transcription (the video page's two Transcribe buttons, and its diff --git a/common/publish/reportExports.ts b/common/publish/reportExports.ts @@ -251,7 +251,7 @@ function viewImagePaths(view: ReportPageView): string[] { // ─── Where a view's files are on this host ─── -type ReportFiles = { +export type ReportFiles = { // A site-root path the view names → the file on disk. local: (sitePath: string) => string | undefined; // A span citation's prepared clip: its site-root path and file. @@ -320,7 +320,7 @@ async function buildEvidencePack(o: { report: Report; view: ReportPageView; files: ReportFiles; - site: Site; + site: Pick<Site, "siteId" | "siteUrl" | "siteTitle">; footer: ReportExportFooter; markdown: string; outFile: string; @@ -392,11 +392,41 @@ async function exportOneReport(o: { signal?: AbortSignal; }): Promise<{ result?: ReportExportResult; problems: ReportExportProblem[] }> { const { paths, site, report, view } = o; - const problems: ReportExportProblem[] = []; const sha = await reportFileSha256(paths, site.siteId, report.id); if (!sha) return { problems: [{ report: report.id, message: "its report.json cannot be read" }] }; + return writeReportExports({ + ...o, + dir: reportExportDir(paths, site.siteId, report.id), + reportSha256: sha, + files: reportFiles(paths, site.siteId, view, o.resolved), + ffmpegBin: paths.ffmpegBin, + }); +} + +// ONE REPORT'S EXPORTS, from its view and where its files are: every format +// asked for and the manifest, into `dir` (emptied first). The site run above +// calls it for each report; the report-site e2e stage calls it for its +// fixture. `report` is the document as resolved (its verification +// computed): the evidence pack's citations.json is its citation set. +export async function writeReportExports(o: { + dir: string; + site: Pick<Site, "siteId" | "siteUrl" | "siteTitle">; + report: Report; + view: ReportPageView; + reportSha256: string; + files: ReportFiles; + ffmpegBin: string; + formats: readonly ReportExportFormat[]; + printer: () => Promise<PdfPrinter | { missing: string }>; + zipBin: string; + now: Date; + log: (line: string) => void; + signal?: AbortSignal; +}): Promise<{ result: ReportExportResult; problems: ReportExportProblem[] }> { + const { site, report, view, files } = o; + const problems: ReportExportProblem[] = []; + const sha = o.reportSha256; const footer = exportFooterFor(view, sha); - const files = reportFiles(paths, site.siteId, view, o.resolved); const siteUrl = site.siteUrl || undefined; const siteTitle = site.siteTitle || undefined; @@ -404,7 +434,7 @@ async function exportOneReport(o: { const dataUris = new Map<string, string>(); for (const p of viewImagePaths(view)) { const local = files.local(p); - const uri = local ? await exportImageDataUri(local, paths.ffmpegBin, o.signal) : null; + const uri = local ? await exportImageDataUri(local, o.ffmpegBin, o.signal) : null; if (uri) dataUris.set(p, uri); else problems.push({ report: report.id, format: "html", message: `the image ${p} cannot be read` }); } @@ -421,7 +451,7 @@ async function exportOneReport(o: { }); const markdown = reportExportMarkdown(view, { siteUrl, siteTitle, footer }); - const dir = reportExportDir(paths, site.siteId, report.id); + const dir = o.dir; await rm(dir, { recursive: true, force: true }); await mkdir(dir, { recursive: true }); const entries: ReportExportManifest["files"] = {}; diff --git a/editor/app/api/ops/reports-export/route.test.ts b/editor/app/api/ops/reports-export/route.test.ts @@ -0,0 +1,55 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, rm } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/reports-export/route.test.ts" +// +// The body's shape and the refusals that come before any job, answered from an +// empty temp corpus: no job is queued and nothing is exported. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "reports-export-route-")); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/reports-export", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("siteId is required and a site id; reportId and formats are the other keys", async () => { + assert.match((await post({})).error, /"siteId" is required/); + const bad = await post({ siteId: "../x" }); + assert.equal(bad.status, 400); + assert.match(bad.error, /not a valid site id/); + const unknown = await post({ siteId: "demo-site", pages: 1 }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): pages — this route accepts siteId, reportId, formats/); +}); + +test("a format not an export's is refused, naming it", async () => { + const r = await post({ siteId: "demo-site", formats: ["html", "docx"] }); + assert.equal(r.status, 400); + assert.match(r.error, /1 of "formats" not in the export formats: docx/); +}); + +test("a site that does not exist is refused before any job", async () => { + const r = await post({ siteId: "demo-site", reportId: "r1", formats: ["md"] }); + assert.equal(r.status, 400); + assert.match(r.error, /No site "demo-site"/); +}); diff --git a/editor/app/api/ops/reports-export/route.ts b/editor/app/api/ops/reports-export/route.ts @@ -0,0 +1,33 @@ +import { reportsExportAction } from "../../../sites/lib/reportsExportAction"; +import { jobResponse, OpsInputError, ops, optString, optSubset, reqString } from "../_lib"; +import { isValidSiteId } from "yt-dlp-transcript-common/lib/site"; +import { REPORT_EXPORT_FORMATS } from "yt-dlp-transcript-common/lib/report/views"; + +export const dynamic = "force-dynamic"; + +// POST { siteId: string, reportId?: string, formats?: ("html"|"pdf"|"md"|"zip")[] } +// -> { ok: true, jobId } +// +// An adapter: one call to the action the site's Reports tab posts. Writes each +// published report (or the one named) as report.html, report.pdf, report.md +// and an evidence pack (or the formats named) into the site's report-exports +// staging, as one `reports-export` job on prepare's queue. The job fails, +// naming each, when an export cannot be written. +export async function POST(request: Request) { + return ops(request, ["siteId", "reportId", "formats"], async (body) => { + const siteId = reqString(body, "siteId"); + if (!isValidSiteId(siteId)) { + throw new OpsInputError( + `"${siteId}" is not a valid site id (lowercase letters, digits and "-"; must start with a letter or digit)`, + ); + } + const reportId = optString(body, "reportId"); + const formats = optSubset(body, "formats", REPORT_EXPORT_FORMATS, "the export formats"); + return jobResponse( + await reportsExportAction(siteId, { + ...(reportId !== undefined ? { reportId } : {}), + ...(formats ? { formats } : {}), + }), + ); + }); +} diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -55,6 +55,7 @@ import { } from "../channels/[slug]/incompleteTranscriptActions"; import { replayFetchWindowAction } from "../channels/[slug]/videos/[id]/videoActions"; import { reportsPrepareAction } from "../sites/lib/reportsPrepareAction"; +import { reportsExportAction } from "../sites/lib/reportsExportAction"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -97,6 +98,17 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { const { p } = params(spec); return reportsPrepareAction(str(p.siteId) ?? spec.slug); }, + // A report site's exports: the same site, report and formats, from the + // reports as they are now. + "reports-export": (spec) => { + const { p } = params(spec); + const reportId = str(p.reportId); + const formats = strings(p.formats); + return reportsExportAction(str(p.siteId) ?? spec.slug, { + ...(reportId ? { reportId } : {}), + ...(formats ? { formats } : {}), + }); + }, // A clip window sourced for another tool. Replay RE-DERIVES from disk like // every bucket job does: if the window (or a wider one covering it) has // arrived since, the retry says so neutrally rather than paying twice. diff --git a/editor/app/sites/[siteId]/reports/page.tsx b/editor/app/sites/[siteId]/reports/page.tsx @@ -6,6 +6,11 @@ import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { isCitedSite } from "yt-dlp-transcript-common/lib/site"; import { readReportMediaIndex } from "yt-dlp-transcript-common/publish/reportMedia"; import { PrepareReportMediaButton } from "../../components/PrepareReportMediaButton"; +import { ExportReportsButton } from "../../components/ExportReportsButton"; +import { + REPORT_EXPORT_FILENAMES, + REPORT_EXPORT_FORMATS, +} from "yt-dlp-transcript-common/lib/report/views"; import { ReportRowActions } from "../../components/ReportRowActions"; import { publishRefusal, @@ -13,7 +18,12 @@ import { type ReportMediaSummary, type ReportRow, } from "../../lib/reportList"; -import { latestReportsPrepareJob, readSiteReportRows } from "../../lib/reportListServer"; +import { + latestSiteReportsJob, + readSiteReportExports, + readSiteReportRows, + type ReportExportsRow, +} from "../../lib/reportListServer"; import { getSiteCached } from "../lib/siteCache"; export const dynamic = "force-dynamic"; @@ -25,7 +35,8 @@ export const metadata: Metadata = { title: "Reports" }; // draft, and what is wrong with it — the report's own checker, so this lists // what prepare and the build would refuse. Publish, unpublish and reorder write // that one key (lib/reportsActions.ts). The evidence media the published -// reports cite is prepared here too, before a build. +// reports cite is prepared here too, before a build, and the reports exported +// as files (HTML, PDF, Markdown, an evidence pack) for the build to publish. // // The documents themselves are written elsewhere (a converter, or by hand); // this tab never edits a report.json. @@ -38,10 +49,12 @@ export default async function SiteReportsPage({ const site = getSiteCached(siteId); if (!site) notFound(); const paths = getPaths(); - const [rows, mediaIndex, lastJob] = await Promise.all([ + const [rows, mediaIndex, lastJob, exportRows, lastExportJob] = await Promise.all([ readSiteReportRows(paths, site), readReportMediaIndex(paths, siteId), - latestReportsPrepareJob(paths, siteId), + latestSiteReportsJob(paths, siteId, "reports-prepare"), + readSiteReportExports(paths, site), + latestSiteReportsJob(paths, siteId, "reports-export"), ]); const publishedCount = rows.filter((r) => r.position !== null).length; const cited = isCitedSite(site); @@ -137,6 +150,44 @@ export default async function SiteReportsPage({ </p> )} </section> + + <section + className="flex flex-col gap-3 border-t border-border pt-6" + aria-label="Exports" + > + <div> + <h2 className="text-lg font-semibold">Exports</h2> + <p className="text-sm text-muted-foreground"> + Writes each published report as files a reader can save and host + again — one self-contained HTML page, a PDF of it, Markdown, and an + evidence pack (the page with its clips, stills and screenshots) — + for the site's next build to publish beside the report. + Preparing the evidence media exports too. A file over 24 MiB stays + on this machine. + </p> + </div> + <ExportReportsButton siteId={siteId} /> + <p className="text-sm text-muted-foreground"> + Last export job:{" "} + {lastExportJob ? ( + <Link href={`/jobs/${lastExportJob.id}`} className="underline font-mono"> + {lastExportJob.id} + </Link> + ) : ( + "none among recent jobs" + )} + {lastExportJob && <> ({lastExportJob.status})</>} + </p> + {exportRows.length === 0 ? ( + <p className="text-sm text-muted-foreground">No published report to export.</p> + ) : ( + <ul className="flex flex-col gap-2" aria-label="Report exports"> + {exportRows.map((row) => ( + <ExportSummary key={row.reportId} row={row} /> + ))} + </ul> + )} + </section> </div> ); } @@ -298,3 +349,45 @@ function MediaSummary({ summary }: { summary: ReportMediaSummary }) { </div> ); } + +function ExportSummary({ row }: { row: ReportExportsRow }) { + const { manifest, publishable } = row; + return ( + <li + data-testid="report-export" + data-report={row.reportId} + className="flex flex-col gap-1 rounded-md border border-border bg-card px-4 py-3 text-sm" + > + <p> + <code>{row.reportId}</code>{" "} + {manifest ? ( + <span className="text-muted-foreground"> + exported {new Date(manifest.exportedAt).toLocaleString()} + </span> + ) : ( + <span className="text-muted-foreground">not exported yet</span> + )} + </p> + {manifest && ( + <ul className="flex flex-wrap gap-x-4 gap-y-1 text-xs"> + {REPORT_EXPORT_FORMATS.filter((f) => manifest.files[f]).map((f) => ( + <li key={f}> + <code>{REPORT_EXPORT_FILENAMES[f]}</code>{" "} + {formatBytes(manifest.files[f]!.bytes)}{" "} + {publishable.files[f] ? ( + <span className="text-success">published on build</span> + ) : ( + <span className="text-warning">local only</span> + )} + </li> + ))} + </ul> + )} + {[...(manifest?.notes ?? []), ...publishable.notes].map((note, i) => ( + <p key={i} className="text-xs text-muted-foreground"> + {note} + </p> + ))} + </li> + ); +} diff --git a/editor/app/sites/components/ExportReportsButton.tsx b/editor/app/sites/components/ExportReportsButton.tsx @@ -0,0 +1,25 @@ +"use client"; + +import { useRouter } from "next/navigation"; +import { StreamActionLog } from "yt-dlp-transcript-common/components/StreamActionLog"; +import { reportsExportAction } from "../lib/reportsExportAction"; +import { cancelJobAction } from "../../jobs/actions"; + +// "Export reports": queues the site's reports-export job (every published +// report as HTML, PDF, Markdown and an evidence pack) and streams its log. +// When the job settles the tab is re-rendered, so the export list beside it +// shows what the job just wrote. +export function ExportReportsButton({ siteId }: { siteId: string }) { + const router = useRouter(); + return ( + <StreamActionLog + trigger={() => reportsExportAction(siteId)} + cancelAction={cancelJobAction} + buttonLabel="Export reports" + runningLabel="Exporting reports…" + onSettled={(started) => { + if (started) router.refresh(); + }} + /> + ); +} diff --git a/editor/app/sites/lib/reportListServer.ts b/editor/app/sites/lib/reportListServer.ts @@ -7,6 +7,12 @@ import { readJsonFile } from "yt-dlp-transcript-common/lib/jsonFile-server"; import { isReportId } from "yt-dlp-transcript-common/lib/report/schema"; import { siteDir, type Site } from "yt-dlp-transcript-common/lib/site"; import { siteReportFile } from "yt-dlp-transcript-common/publish/reportMedia"; +import { + publishableReportExports, + readReportExportManifest, + type PublishableReportExports, + type ReportExportManifest, +} from "yt-dlp-transcript-common/publish/reportExportFiles"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { listAllJobs, type JobListEntry } from "yt-dlp-transcript-common/jobs/listJobs"; import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta"; @@ -43,19 +49,38 @@ export async function readSiteReportRows(paths: Paths, site: Site): Promise<Repo // as /jobs lists them. A prepare older than that is not linked. const PREPARE_JOB_LOOKBACK = 100; -// The newest `reports-prepare` job for this site (its spec's slug is the site -// id — reportsPrepareAction.ts), live or from an earlier server lifetime, or -// null. -export async function latestReportsPrepareJob( +// The newest job of `kind` for this site (a reports job's spec's slug is the +// site id — reportsPrepareAction.ts, reportsExportAction.ts), live or from an +// earlier server lifetime, or null. +export async function latestSiteReportsJob( paths: Paths, siteId: string, + kind: "reports-prepare" | "reports-export", ): Promise<JobListEntry | null> { const page = await listAllJobs(paths, { limit: PREPARE_JOB_LOOKBACK }); const registry = getRegistry(); for (const entry of page.entries) { - if (entry.kind !== "reports-prepare") continue; + if (entry.kind !== kind) continue; const spec = registry.get(entry.id)?.spec ?? (await readJobMeta(paths, entry.id))?.spec; if (spec?.slug === siteId) return entry; } return null; } + +// One published report's exports, as the tab shows them: the manifest of the +// last export (or null), and which files compose would publish now. +export type ReportExportsRow = { + reportId: string; + manifest: ReportExportManifest | null; + publishable: PublishableReportExports; +}; + +export async function readSiteReportExports(paths: Paths, site: Site): Promise<ReportExportsRow[]> { + return Promise.all( + (site.reports ?? []).map(async (reportId) => ({ + reportId, + manifest: await readReportExportManifest(paths, site.siteId, reportId), + publishable: await publishableReportExports(paths, site.siteId, reportId), + })), + ); +} diff --git a/editor/app/sites/lib/reportsExportAction.ts b/editor/app/sites/lib/reportsExportAction.ts @@ -0,0 +1,90 @@ +"use server"; + +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { isValidSiteId, listSiteIds } from "yt-dlp-transcript-common/lib/site"; +import { isReportId } from "yt-dlp-transcript-common/lib/report/schema"; +import { + REPORT_EXPORT_FORMATS, + type ReportExportFormat, +} from "yt-dlp-transcript-common/lib/report/views"; +import { + runManagedFunction, + type StreamActionResult, +} from "yt-dlp-transcript-common/jobs/streamCommand"; +import { + exportSiteReports, + formatReportExportProblems, +} from "yt-dlp-transcript-common/publish/reportExports"; +import { REPORTS_PREPARE_QUEUE } from "./reportsQueue"; + +// EXPORT A REPORT SITE'S REPORTS as a job: each published report as +// report.html, report.pdf, report.md and an evidence pack, into +// `.export-index/sites/<siteId>/report-exports/<reportId>/`, where compose +// publishes them from. The work is publish/reportExports.ts's, the same +// `archilyzer reports export` runs. The job FAILS when an export cannot be +// written (the log names each); a PDF skipped for want of a browser is a +// note in the log, not a failure. +// +// On prepare's queue: it reads the media cache a prepare writes and prunes. +// The spec's `slug` is the SITE id; the replay handler reads `params`. +export async function reportsExportAction( + siteId: string, + opts: { reportId?: string; formats?: string[] } = {}, +): Promise<StreamActionResult> { + const paths = getPaths(); + const id = siteId.trim(); + if (!isValidSiteId(id)) { + return { ok: false, error: `"${id}" is not a valid site id` }; + } + if (!listSiteIds(paths).includes(id)) { + return { ok: false, error: `No site "${id}"` }; + } + if (opts.reportId !== undefined && !isReportId(opts.reportId)) { + return { ok: false, error: `"${opts.reportId}" is not a report id` }; + } + let formats: ReportExportFormat[] | undefined; + if (opts.formats !== undefined) { + const bad = opts.formats.filter( + (f) => !(REPORT_EXPORT_FORMATS as readonly string[]).includes(f), + ); + if (bad.length > 0 || opts.formats.length === 0) { + return { + ok: false, + error: `formats are some of ${REPORT_EXPORT_FORMATS.join(", ")}${bad.length ? ` (not ${bad.join(", ")})` : ""}`, + }; + } + formats = REPORT_EXPORT_FORMATS.filter((f) => opts.formats!.includes(f)); + } + return runManagedFunction({ + kind: "reports-export", + queueKey: REPORTS_PREPARE_QUEUE, + paths, + spec: { + kind: "reports-export", + slug: id, + params: { + siteId: id, + ...(opts.reportId ? { reportId: opts.reportId } : {}), + ...(formats ? { formats } : {}), + }, + }, + fn: async (onLog, signal) => { + const result = await exportSiteReports({ + siteId: id, + paths, + onLog, + signal, + ...(opts.reportId ? { reportId: opts.reportId } : {}), + ...(formats ? { formats } : {}), + }); + for (const r of result.exported) { + for (const note of r.manifest.notes) onLog(`note: ${r.reportId}: ${note}`); + } + if (result.problems.length === 0) return; + for (const line of formatReportExportProblems(result.problems)) onLog(line); + throw new Error( + `${result.problems.length} problem(s): the reports were not all exported`, + ); + }, + }); +} diff --git a/editor/app/sites/lib/reportsPrepareAction.ts b/editor/app/sites/lib/reportsPrepareAction.ts @@ -10,16 +10,20 @@ import { formatReportMediaProblems, prepareReportMedia, } from "yt-dlp-transcript-common/publish/reportMedia"; - -// Its own queue: one prepare at a time, waiting on no download and no build. -const REPORTS_PREPARE_QUEUE = "reports-prepare"; +import { + exportAfterPrepare, + formatReportExportProblems, +} from "yt-dlp-transcript-common/publish/reportExports"; +import { REPORTS_PREPARE_QUEUE } from "./reportsQueue"; // PREPARE A REPORT SITE'S EVIDENCE MEDIA as a job: every clip its published // reports cite, cut from the media on disk, and every cited post capture, // copied into `.export-index/sites/<siteId>/report-media/` before its build. // The work is publish/reportMedia.ts's, the same `archilyzer reports prepare` // runs. The job FAILS when any citation lacks its media — the log names each -// one — and the manifest it writes carries the same list. +// one — and the manifest it writes carries the same list. When nothing is +// missing it ends by exporting the reports as files (reportsExportAction's +// work, publish/reportExports.ts), and fails when an export does. // // The spec's `slug` is the SITE id (a spec needs one, and this job belongs to // no channel); the replay handler reads `params.siteId`. @@ -41,10 +45,20 @@ export async function reportsPrepareAction( spec: { kind: "reports-prepare", slug: id, params: { siteId: id } }, fn: async (onLog, signal) => { const index = await prepareReportMedia({ siteId: id, paths, onLog, signal }); - if (index.problems.length === 0) return; - for (const line of formatReportMediaProblems(index.problems)) onLog(line); + if (index.problems.length > 0) { + for (const line of formatReportMediaProblems(index.problems)) onLog(line); + throw new Error( + `${index.problems.length} problem(s): the site's evidence media is not complete`, + ); + } + const exported = await exportAfterPrepare(index, { siteId: id, paths, onLog, signal }); + for (const r of exported?.exported ?? []) { + for (const note of r.manifest.notes) onLog(`note: ${r.reportId}: ${note}`); + } + if (!exported || exported.problems.length === 0) return; + for (const line of formatReportExportProblems(exported.problems)) onLog(line); throw new Error( - `${index.problems.length} problem(s): the site's evidence media is not complete`, + `${exported.problems.length} problem(s): the reports were not all exported`, ); }, }); diff --git a/editor/app/sites/lib/reportsQueue.ts b/editor/app/sites/lib/reportsQueue.ts @@ -0,0 +1,4 @@ +// The queue a report site's prepare and export jobs share: one at a time, +// waiting on no download and no build — and never both at once, since an +// export reads the media cache a prepare writes and prunes. +export const REPORTS_PREPARE_QUEUE = "reports-prepare"; diff --git a/export/app/components/reports/ReportArticle.tsx b/export/app/components/reports/ReportArticle.tsx @@ -21,10 +21,20 @@ import { ArchiveList, Eyebrow, ReportName, SourceBlock, dateLabel, textLink } fr // fact-check's tally, the summary, the sections and their claims — each claim // its verdict, the document's own sentence (its still), the findings with // their inline citations, and the evidence cards — then the numbered -// reference list the inline markers jump to, and the citations as files. +// reference list the inline markers jump to, and the downloads: the report as +// files (HTML, PDF, Markdown, the evidence pack) and its citations as data. const proseClass = "text-[0.95rem] leading-relaxed text-foreground"; +const DOWNLOADS: readonly (readonly [keyof NonNullable<ReportPageView["downloads"]>, string])[] = [ + ["html", "HTML"], + ["pdf", "PDF"], + ["md", "Markdown"], + ["zip", "Evidence pack"], + ["json", "Citations JSON"], + ["csv", "CSV"], +]; + function SourceSentence({ c, archives }: { c: SourceCitationView; archives: readonly SourceArchive[] | undefined }) { return ( <figure data-source-sentence={c.id} className="flex flex-col gap-1.5"> @@ -117,6 +127,12 @@ export default function ReportArticle({ view }: { view: ReportPageView }) { view.updated && view.updated !== view.published ? `Updated ${dateLabel(view.updated)}` : null, ].filter(Boolean); const claimCount = view.sections.reduce((n, s) => n + s.claims.length, 0); + // The report as files (its exports), then its citations as data: only what + // compose published. + const downloads = DOWNLOADS.flatMap(([key, label]) => { + const href = view.downloads?.[key]; + return href ? [[key, label, href] as const] : []; + }); // Documents quoted that are not the subject, each shown once before the // references. const otherSources = Object.values(view.sources).filter((s) => s.id !== view.subject); @@ -203,20 +219,15 @@ export default function ReportArticle({ view }: { view: ReportPageView }) { </section> )} - {(view.downloads?.json || view.downloads?.csv) && ( - <p data-citation-downloads="" className="flex flex-wrap items-center gap-3 text-sm text-muted-foreground"> + {downloads.length > 0 && ( + <p data-citation-downloads="" className="flex flex-wrap items-center gap-x-3 gap-y-1 text-sm text-muted-foreground"> <ArrowDownToLine className="size-4" aria-hidden /> - Download citations: - {view.downloads.json && ( - <a href={view.downloads.json} download className={textLink}> - JSON - </a> - )} - {view.downloads.csv && ( - <a href={view.downloads.csv} download className={textLink}> - CSV + Download: + {downloads.map(([key, label, href]) => ( + <a key={key} href={href} download data-download={key} className={textLink}> + {label} </a> - )} + ))} </p> )} </article> diff --git a/export/e2e-report/audit.spec.ts b/export/e2e-report/audit.spec.ts @@ -28,6 +28,10 @@ test("it holds the reports, the moments, the cited media and the contract", () = "reports/demo-factcheck/page.json", "reports/demo-factcheck/citations.json", "reports/demo-factcheck/citations.csv", + "reports/demo-factcheck/report.html", + "reports/demo-factcheck/report.pdf", + "reports/demo-factcheck/report.md", + "reports/demo-factcheck/evidence-pack.zip", "reports/demo-factcheck/stills/a01.png", "m/index.json", "m/demo-channel/abc123/3126.00-3151.00/index.html", diff --git a/export/e2e-report/contract.ts b/export/e2e-report/contract.ts @@ -1,6 +1,8 @@ // The cited fixture site's compose step (stage.ts runs it, in a child process // with the stage's environment): what compose's reports stage writes beside -// the views — the citation downloads — and the cited site's contract, by +// the views — the citation downloads and the report's exports (HTML, PDF, +// Markdown, evidence pack, by publish/reportExports.ts's writer) — and the +// cited site's contract, by // compose's own functions (bin/compose-site.ts emitFederationFiles / // emitAiFiles), so the stage's public/ is what compose would leave for this // site. The view JSON, stills and media are the fixture's, already in place. @@ -8,8 +10,9 @@ // Every import is dynamic: getPaths() reads the environment on first use. import fs from "node:fs"; +import os from "node:os"; import path from "node:path"; -import type { MomentIndexView, ReportIndexView, ReportPageView } from "yt-dlp-transcript-common/lib/report/views"; +import type { MomentIndexView, MomentPageView, ReportIndexView, ReportPageView } from "yt-dlp-transcript-common/lib/report/views"; const { getPaths } = await import("yt-dlp-transcript-common/lib/paths"); const { getSite } = await import("yt-dlp-transcript-common/lib/site"); @@ -18,7 +21,12 @@ const { citationSet, citationsCsv } = await import("yt-dlp-transcript-common/pub const { MOMENTS_INDEX_PATH, REPORTS_INDEX_PATH, reportCitationsDownloadPath, reportViewPath } = await import( "yt-dlp-transcript-common/lib/report/views" ); -const { readFixtureReport } = await import("../fixtures/report-site/fixture"); +const { FIXTURE_DIR, readFixtureReport } = await import("../fixtures/report-site/fixture"); +const { openPlaywrightPdfPrinter, writeReportExports } = await import("yt-dlp-transcript-common/publish/reportExports"); +const { sha256Hex } = await import("yt-dlp-transcript-common/publish/reportExportFiles"); +const { REPORT_EXPORT_FILENAMES, REPORT_EXPORT_FORMATS, momentViewPath, reportExportDownloadPath } = await import( + "yt-dlp-transcript-common/lib/report/views" +); const paths = getPaths(); const siteId = process.env.SITE_ID; @@ -39,6 +47,44 @@ for (const entry of index.reports) { `${JSON.stringify(citationSet(report, view), null, 2)}\n`, ); fs.writeFileSync(pub(reportCitationsDownloadPath(entry.id, "csv")), citationsCsv(view)); + + // The report's exports, by the host step's own writer, from the files + // already in public/ (its stills, the post's shot, the clips), published + // where compose publishes them. + const printer = openPlaywrightPdfPrinter(); + const dir = path.join(fs.mkdtempSync(path.join(os.tmpdir(), "report-site-exports-")), entry.id); + try { + const { result, problems } = await writeReportExports({ + dir, + site, + report, + view, + reportSha256: sha256Hex(fs.readFileSync(path.join(FIXTURE_DIR, "source", entry.id, "report.json"))), + files: { + local: (p) => (fs.existsSync(pub(p)) ? pub(p) : undefined), + clip: (c) => { + const m = readJson<MomentPageView>(momentViewPath(c.moment)); + return m.clip ? { sitePath: m.clip.src, file: pub(m.clip.src), kind: m.kind === "audio" ? "audio" : "video" } : undefined; + }, + }, + ffmpegBin: paths.ffmpegBin, + formats: REPORT_EXPORT_FORMATS, + printer: () => printer, + zipBin: "zip", + now: new Date(), + log: console.log, + }); + if (problems.length > 0 || result.manifest.notes.length > 0) { + throw new Error(`contract.ts: the fixture's exports: ${JSON.stringify([...problems, ...result.manifest.notes])}`); + } + for (const f of REPORT_EXPORT_FORMATS) { + fs.copyFileSync(path.join(dir, REPORT_EXPORT_FILENAMES[f]), pub(reportExportDownloadPath(entry.id, f))); + } + } finally { + const p = await printer; + if (!("missing" in p)) await p.close(); + fs.rmSync(path.dirname(dir), { recursive: true, force: true }); + } } await emitFederationFiles(site, paths); diff --git a/export/e2e-report/report-site.spec.ts b/export/e2e-report/report-site.spec.ts @@ -130,16 +130,58 @@ test("a citation's number jumps to its entry in the reference list", async ({ pa await expect(page.locator("#c-w01")).toContainText("Opened to traffic: April 2019."); }); -test("the report's citations download as JSON and CSV", async ({ page, request }) => { +test("the report downloads as HTML, PDF, Markdown and an evidence pack, its citations as JSON and CSV", async ({ + page, + request, +}) => { await page.goto(REPORT); const links = page.locator("[data-citation-downloads] a[download]"); - await expect(links).toHaveCount(2); - for (const href of await links.evaluateAll((els) => els.map((el) => el.getAttribute("href")!))) { + await expect(links).toHaveText(["HTML", "PDF", "Markdown", "Evidence pack", "Citations JSON", "CSV"]); + const hrefs = await links.evaluateAll((els) => els.map((el) => el.getAttribute("href")!)); + expect(hrefs).toEqual([ + `${REPORT}report.html`, + `${REPORT}report.pdf`, + `${REPORT}report.md`, + `${REPORT}evidence-pack.zip`, + `${REPORT}citations.json`, + `${REPORT}citations.csv`, + ]); + for (const href of hrefs) { const res = await request.get(href); expect(res.status(), href).toBe(200); - if (href.endsWith(".json")) expect((await res.json()).format).toBe("archilyzer-citations"); - else expect(await res.text()).toContain("c01"); + const body = await res.body(); + if (href.endsWith("citations.json")) expect(JSON.parse(body.toString()).format).toBe("archilyzer-citations"); + else if (href.endsWith(".csv")) expect(body.toString()).toContain("c01"); + else if (href.endsWith(".html")) expect(body.toString()).toContain("<title>Checking an example article</title>"); + else if (href.endsWith(".pdf")) expect(body.subarray(0, 5).toString()).toBe("%PDF-"); + else if (href.endsWith(".md")) expect(body.toString()).toContain("# Checking an example article"); + else expect(body.subarray(0, 4).toString("hex")).toBe("504b0304"); + } +}); + +test("report.html opens with no network: one file, no script, every image inlined", async ({ page, baseURL }) => { + const url = `${baseURL}${REPORT}report.html`; + const asked: string[] = []; + // Only the file itself may load; anything else it asked for would be refused. + await page.route("**/*", (route) => { + const u = route.request().url(); + if (u === url) return route.continue(); + asked.push(u); + return route.abort(); + }); + await page.goto(url); + await expect(page.locator("h1")).toContainText("Checking an example article"); + await expect(page.locator("article[data-claim]")).toHaveCount(5); + await expect(page.locator("script")).toHaveCount(0); + const imgs = page.locator("img"); + expect(await imgs.count()).toBeGreaterThan(0); + for (const img of await imgs.all()) { + expect(await img.getAttribute("src")).toMatch(/^data:image\//); + await expect.poll(() => img.evaluate((el) => (el as HTMLImageElement).naturalWidth)).toBeGreaterThan(0); } + await expect(page.locator("[data-reference-list] > li")).toHaveCount(6); + await expect(page.locator("[data-export-footer]")).toContainText("report sha256"); + expect(asked).toEqual([]); }); test("a video citation opens its moment: the clip, the cue lines with the span marked, where it is cited", async ({ diff --git a/export/e2e-report/stage.ts b/export/e2e-report/stage.ts @@ -19,9 +19,11 @@ // THE PUBLIC DIR is what compose's reports stage would write for the site: // - the fixture report site (export/fixtures/report-site/public): the view // JSON R4's builder made, the stills, the post's shot, the evidence clips; -// - the citation downloads and the cited site's contract (site.json, +// - the citation downloads, the report's exports (report.html, report.pdf, +// report.md, evidence-pack.zip) and the cited site's contract (site.json, // corpus.json, llms.txt, robots.txt, sitemap.xml, _headers), written by -// compose's own functions in a child process (contract.ts); +// compose's and the export step's own functions in a child process +// (contract.ts); // - export/public's checked-in assets (git-tracked files only — the // checkout's composed corpus is never read). // diff --git a/export/fixtures/report-site/fixture.ts b/export/fixtures/report-site/fixture.ts @@ -29,7 +29,9 @@ import { evidenceClipPath, momentViewPath, reportCitationsDownloadPath, + reportExportDownloadPath, reportIndexEntry, + REPORT_EXPORT_FORMATS, reportViewPath, type MomentPageView, type RecordView, @@ -101,7 +103,10 @@ export function buildFixtureReportView(report = readFixtureReport()): ReportPage text: `${c.quote}\n\nPosted to settle it.`, shot: shotFor(c), }), + // Every download compose lists when the report was exported (the e2e + // stage writes the exports themselves: e2e-report/contract.ts). downloads: { + ...Object.fromEntries(REPORT_EXPORT_FORMATS.map((f) => [f, reportExportDownloadPath(report.id, f)])), json: reportCitationsDownloadPath(report.id, "json"), csv: reportCitationsDownloadPath(report.id, "csv"), }, diff --git a/export/fixtures/report-site/public/reports/demo-factcheck/page.json b/export/fixtures/report-site/public/reports/demo-factcheck/page.json @@ -213,6 +213,10 @@ } ], "downloads": { + "html": "/reports/demo-factcheck/report.html", + "pdf": "/reports/demo-factcheck/report.pdf", + "md": "/reports/demo-factcheck/report.md", + "zip": "/reports/demo-factcheck/evidence-pack.zip", "json": "/reports/demo-factcheck/citations.json", "csv": "/reports/demo-factcheck/citations.csv" } diff --git a/export/scripts/serve-out.mjs b/export/scripts/serve-out.mjs @@ -53,6 +53,7 @@ const TYPES = { ".srt": "text/plain; charset=utf-8", ".md": "text/markdown; charset=utf-8", ".zip": "application/zip", + ".pdf": "application/pdf", ".wasm": "application/wasm", }; diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -41,6 +41,7 @@ // pnpm ops build-homepage --json '{"deploy":true}' --wait // pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait // pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait +// pnpm ops reports-export --json '{"siteId":"demo-site","formats":["html","md"]}' --wait // pnpm ops get channel the-quartering // pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}' // pnpm ops tag-videos --file ids.json @@ -132,6 +133,9 @@ const ACTIONS = [ // A report site's evidence media: cut every cited clip and copy every cited // post capture into the site's report-media cache ({siteId}). "reports-prepare", + // A report site's exports: each published report as HTML, PDF, Markdown and + // an evidence pack ({siteId, reportId?, formats?}). + "reports-export", "lane", // The curated-tag writers. `tags` edits the vocabulary (define/remove); // `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch @@ -328,6 +332,11 @@ export function usage() { 'reports-prepare cuts every clip and copies every post capture a site\'s', ' published reports cite into its report-media cache, before its build:', ' {"siteId"}. The job fails, naming each one, when a citation lacks media.', + ' When nothing is missing it then exports the reports, as reports-export.', + "", + 'reports-export writes each published report as report.html, report.pdf,', + ' report.md and evidence-pack.zip for the site\'s build to publish:', + ' {"siteId", "reportId"?, "formats"?: ["html","pdf","md","zip"]}.', "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -418,6 +418,14 @@ test("reports-prepare is a POST to its route, named in the usage", () => { assert.match(usage(), /reports-prepare cuts every clip/); }); +test("reports-export is a POST to its route, named in the usage", () => { + const p = parseArgs(["reports-export", "--json", '{"siteId":"demo-site","formats":["md"]}']); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/reports-export"); + assert.deepEqual(p.body, { siteId: "demo-site", formats: ["md"] }); + assert.match(usage(), /reports-export writes each published report/); +}); + test("persist-videos is a POST to its route, named in the usage", () => { const p = parseArgs([ "persist-videos",