import { readFile, stat } from "node:fs/promises"; import path from "node:path"; import { buildReportPageView, REPORT_PAGE_FORMAT, REPORT_VIEWS_VERSION, type RecordView, type ReportPageView, } from "yt-dlp-transcript-common/lib/report/views"; import type { Report } from "yt-dlp-transcript-common/lib/report/schema"; import { platformMomentUrl } from "yt-dlp-transcript-common/lib/momentUrl"; import { readAllPosts } from "yt-dlp-transcript-common/lib/posts-server"; import type { Post } from "yt-dlp-transcript-common/lib/posts"; import { readCues } from "@/lib/projects/report.mjs"; import { CHANNELS_DIR } from "@/lib/paths"; // A report's PAGE VIEW, built the way the export site builds it // (common/lib/report/views.ts buildReportPageView) but resolved against what // umtool can read without the LMDB index or a compose run: each cited record's // own files on disk. compose's resolveSiteReports is not reused: it verifies, // prepares and THROWS on any problem, and half of what this page is for is // reading drafts that have problems. // // A record resolves from its transcript.cues.json (title, date, the uploader's // display name, webpageUrl -- through lib/projects/report.mjs readCues, which is // memoised on the file's mtime), else its metadata.info.json, else the // citation's own label, speaker and date. A post resolves from the channel's // posts. Nothing here fails the page: a record that cannot be read is a card // with less on it. const isoDay = (d: unknown): string | undefined => { const s = typeof d === "string" ? d : ""; if (/^\d{8}$/.test(s)) return `${s.slice(0, 4)}-${s.slice(4, 6)}-${s.slice(6, 8)}`; if (/^\d{4}-\d{2}-\d{2}/.test(s)) return s.slice(0, 10); return undefined; }; export const recordDir = (channel: string, id: string) => path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel, "data", id); type RecordMeta = { title?: string; date?: string; channelTitle?: string; webpageUrl?: string }; const metaMemo = new Map(); /** What a record says about itself; null when its directory holds neither file. */ export async function recordMeta(channel: string, id: string): Promise { if (!/^[A-Za-z0-9_.@-]+$/.test(channel) || !/^[A-Za-z0-9_.@-]+$/.test(id)) return null; const dir = recordDir(channel, id); const cues = (await readCues(path.join(/* turbopackIgnore: true */ dir, "transcript.cues.json"))) as | { title?: string; uploadDate?: string; webpageUrl?: string; channel?: string } | null; if (cues && (cues.title || cues.webpageUrl)) { return { title: cues.title, date: isoDay(cues.uploadDate), channelTitle: cues.channel, webpageUrl: cues.webpageUrl }; } const file = path.join(/* turbopackIgnore: true */ dir, "metadata.info.json"); const st = await stat(/* turbopackIgnore: true */ file).catch(() => null); if (!st) return null; const key = `${Math.round(st.mtimeMs)}-${st.size}`; const hit = metaMemo.get(file); if (hit?.key === key) return hit.value; let value: RecordMeta | null = null; try { const m = JSON.parse(await readFile(/* turbopackIgnore: true */ file, "utf8")); value = { title: typeof m.title === "string" ? m.title : undefined, date: isoDay(m.upload_date), channelTitle: typeof m.uploader === "string" ? m.uploader : typeof m.channel === "string" ? m.channel : undefined, webpageUrl: typeof m.webpage_url === "string" ? m.webpage_url : undefined, }; } catch { value = null; } metaMemo.set(file, { key, value }); return value; } // A channel's posts, by id. A big X archive is thousands of posts, so one read // per channel per minute. const postsMemo = new Map> }>(); export function channelPosts(channel: string): Promise> { const hit = postsMemo.get(channel); if (hit && Date.now() - hit.at < 60_000) return hit.value; const value = readAllPosts(path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel)) .then((list) => new Map(list.map((p) => [p.id, p]))) .catch(() => new Map()); postsMemo.set(channel, { at: Date.now(), value }); return value; } /** The capture screenshot of a post, if the editor took one. */ export async function postShotFile(channel: string, id: string): Promise { if (!/^[A-Za-z0-9_.@-]+$/.test(channel) || !/^[A-Za-z0-9_.@-]+$/.test(id)) return null; const file = path.join(/* turbopackIgnore: true */ CHANNELS_DIR, channel, "posts-media", id, "shot.png"); return (await stat(/* turbopackIgnore: true */ file).catch(() => null))?.isFile() ? file : null; } export const corpusMediaUrl = (abs: string) => `/api/sites/media?corpus=${encodeURIComponent(abs)}`; type Cite = NonNullable[string]; async function recordViewOf( c: Cite, metaOf: (channel: string, id: string) => Promise = recordMeta, ): Promise { if (c.kind === "video" || c.kind === "audio") { const m = await metaOf(c.channel, c.id); return { channel: c.channel, id: c.id, ...(m?.channelTitle || c.speaker ? { channelTitle: m?.channelTitle ?? c.speaker } : {}), ...(m?.title || c.label ? { title: m?.title ?? c.label } : {}), ...(m?.date || c.date ? { date: m?.date ?? c.date } : {}), ...(m?.webpageUrl ? { originalUrl: platformMomentUrl(m.webpageUrl, null, c.start) ?? m.webpageUrl } : {}), }; } if (c.kind === "post") { const post = (await channelPosts(c.channel)).get(c.id); return { channel: c.channel, id: c.id, ...(post?.authorName || post?.author ? { channelTitle: post.authorName ?? post.author } : {}), ...(post?.createdAt ? { date: post.createdAt.slice(0, 10) } : c.date ? { date: c.date } : {}), ...(post?.platform ? { platform: post.platform } : {}), ...(post?.url ? { originalUrl: post.url } : {}), }; } return undefined; } /** * The page view of a report, or -- when the report's citations cannot be built * into one (a draft naming a source it does not define) -- the same view with * no citations, and the reason. */ export async function articleView(report: Report): Promise<{ view: ReportPageView; error: string | null }> { const records = new Map(); const posts = new Map(); // A fact-check cites the same few records hundreds of times: read each once, // all at the same time (recordMeta memoises on the file, not on a promise). const metas = new Map>(); const metaOf = (channel: string, id: string) => { const k = `${channel}/${id}`; if (!metas.has(k)) metas.set(k, recordMeta(channel, id)); return metas.get(k)!; }; await Promise.all( Object.values(report.citations ?? {}).map(async (c) => { const r = await recordViewOf(c, metaOf); if (r) records.set(c, r); if (c.kind === "post") { const post = (await channelPosts(c.channel)).get(c.id); const shot = await postShotFile(c.channel, c.id); posts.set(c, { ...(post ? { author: post.authorName ?? post.author, text: post.text } : {}), ...(shot ? { shot: corpusMediaUrl(shot) } : {}), }); } }), ); try { const view = buildReportPageView(report, { record: (c) => records.get(c as Cite) ?? { channel: c.channel, id: c.id }, post: (c) => posts.get(c as Cite), }); return { view, error: null }; } catch (err) { const view = { format: REPORT_PAGE_FORMAT, version: REPORT_VIEWS_VERSION, id: report.id, kind: report.kind, ...(report.series ? { series: report.series } : {}), title: report.title, ...(report.subtitle ? { subtitle: report.subtitle } : {}), ...(report.summary ? { summary: report.summary } : {}), ...(report.method ? { method: report.method } : {}), ...(report.published ? { published: report.published } : {}), ...(report.updated ? { updated: report.updated } : {}), sources: {}, verdicts: {}, citations: {}, ...(report.slides ? { slides: report.slides } : {}), sections: report.sections.map((s) => ({ id: s.id, title: s.title, ...(s.body ? { body: s.body } : {}), ...(s.slide ? { slide: s.slide } : {}), claims: (s.claims ?? []).map((cl) => ({ ...cl, sourceQuote: cl.sourceQuote?.citation, citations: cl.citations ?? [] })), })), } as unknown as ReportPageView; return { view, error: err instanceof Error ? err.message : String(err) }; } }