// THE ONE PUBLISH MODULE THAT READS A POST CAPTURE — and only the captures a // report cites. // // A post capture (`channels//posts-media//`: the post's screenshot, // its attached media, its record) is editor-only material: nothing the export // builds may carry it, and social/postCapture.test.ts holds every publish and // export source to that by name. A report site is the one exception the plan // makes (plans/report-sites.md, "Evidence media"): a cited post's moment page // shows its screenshot and media. So the exception is THIS module, named in // that guard, and it copies exactly the posts it is handed — the caller's // cited list, already narrowed by the site's channel pool and the post // visibility rule (lib/postsVisibility.ts) — and nothing else: // // - `shot.png` and the media files (`listCapturedMediaFiles`): never the // record (`capture.json`), never the article half, never a leftover; // - into `/posts///`, each post's directory rebuilt // from scratch, so a file the capture no longer has does not linger; // - and every other directory under `/posts/` is REMOVED: what is // there after a run is the cited set, whatever an earlier run cited. // // A post with no screenshot is missing (its moment page has nothing to show); // the editor's Capture posts fills it. import { open, readdir, rm, stat } from "node:fs/promises"; import path from "node:path"; import { copyFileAtomic } from "../lib/jsonFile-server"; import { fileDigest, listCapturedMediaFiles, postCaptureDir, postsMediaDir, SHOT_FILENAME, } from "../social/postCapture"; // The directory under the report-media cache that holds the copied captures. export const REPORT_POSTS_DIRNAME = "posts"; export type CitedPost = { channel: string; id: string }; export type CopiedFile = { // Relative to the report-media cache directory, `/`-separated. file: string; bytes: number; sha256: string; }; export type CopiedCapture = CitedPost & { shot: CopiedFile & { width: number | null; height: number | null }; media: CopiedFile[]; }; export type MissingCapture = CitedPost & { message: string }; // A PNG's size from its header: the IHDR chunk is always first, its width and // height big-endian at bytes 16 and 20. Null for anything that is not a PNG. export async function pngSize(file: string): Promise<{ width: number; height: number } | null> { const head = Buffer.alloc(24); try { const fh = await open(file, "r"); try { await fh.read(head, 0, 24, 0); } finally { await fh.close(); } } catch { return null; } if (head.readUInt32BE(0) !== 0x89504e47 || head.toString("latin1", 12, 16) !== "IHDR") { return null; } return { width: head.readUInt32BE(16), height: head.readUInt32BE(20) }; } // Where a post's capture is: its channel's capture directory, the id checked // (lib's postCaptureDir throws on one that is not a post id). export function citedCaptureSourceDir(channelsDir: string, channel: string, id: string): string { return postCaptureDir(postsMediaDir(path.join(channelsDir, channel)), id); } const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true; // Copy the cited posts' captures into `/posts/`, and remove every // capture there that is not cited. `channelsDir` is the corpus's // `channels/`. export async function copyCitedPostCaptures(opts: { channelsDir: string; destRoot: string; posts: readonly CitedPost[]; }): Promise<{ copied: CopiedCapture[]; missing: MissingCapture[] }> { const postsRoot = path.join(opts.destRoot, REPORT_POSTS_DIRNAME); const copied: CopiedCapture[] = []; const missing: MissingCapture[] = []; const keep = new Set(); for (const post of opts.posts) { const key = `${post.channel}/${post.id}`; if (keep.has(key)) continue; keep.add(key); let src: string; try { src = citedCaptureSourceDir(opts.channelsDir, post.channel, post.id); } catch (e) { missing.push({ ...post, message: (e as Error).message }); continue; } const dest = path.join(postsRoot, post.channel, post.id); await rm(dest, { recursive: true, force: true }); if (!(await isFile(path.join(src, SHOT_FILENAME)))) { missing.push({ ...post, message: "the post has no captured screenshot (capture it on the channel's Posts page)" }); continue; } const copy = async (name: string): Promise => { await copyFileAtomic(path.join(src, name), path.join(dest, name), { mkdir: true }); const digest = await fileDigest(path.join(dest, name)); return { file: path.posix.join(REPORT_POSTS_DIRNAME, post.channel, post.id, name), ...digest }; }; const shot = await copy(SHOT_FILENAME); const size = await pngSize(path.join(dest, SHOT_FILENAME)); const media: CopiedFile[] = []; for (const name of await listCapturedMediaFiles(src)) { // A file through its links; a directory or a dangling link is skipped. if (await isFile(path.join(src, name))) media.push(await copy(name)); } copied.push({ ...post, shot: { ...shot, width: size?.width ?? null, height: size?.height ?? null }, media, }); } // Everything else goes: an uncited post, an uncited channel. for (const channel of await readdir(postsRoot).catch(() => [] as string[])) { const channelDir = path.join(postsRoot, channel); const ids = await readdir(channelDir).catch(() => null); if (ids === null) { await rm(channelDir, { recursive: true, force: true }); continue; } let kept = 0; for (const id of ids) { if (keep.has(`${channel}/${id}`)) kept++; else await rm(path.join(channelDir, id), { recursive: true, force: true }); } if (kept === 0) await rm(channelDir, { recursive: true, force: true }); } if (keep.size === 0) await rm(postsRoot, { recursive: true, force: true }); return { copied, missing }; }