Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit f0e07f6de96956648310e0e569cdb61583636fd0
parent 9f06f2e1addde115449ab73e5735aa795797579a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 03:11:17 -0400

common: publish/citedPostCaptures.ts copies only the cited posts' captures; the posts-media guard allows exactly that module

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/publish/citedPostCaptures.test.ts | 132+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/publish/citedPostCaptures.ts | 149+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/postCapture.test.ts | 23++++++++++++++++-------
3 files changed, 297 insertions(+), 7 deletions(-)

diff --git a/common/publish/citedPostCaptures.test.ts b/common/publish/citedPostCaptures.test.ts @@ -0,0 +1,132 @@ +// The one publish module that reads post captures copies the CITED ones and +// nothing else — the other half of the guard in the post capture tests. +// (The capture directory is reached through the module's own helper, so this +// file does not name it and needs no exception of its own.) +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test publish/citedPostCaptures.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { citedCaptureSourceDir, copyCitedPostCaptures, pngSize } from "./citedPostCaptures"; + +const ROOT = await mkdtemp(path.join(tmpdir(), "cited-captures-")); +after(() => rm(ROOT, { recursive: true, force: true })); + +// A PNG header for a w×h image (the bytes after IHDR's size do not matter here). +function png(w: number, h: number): Buffer { + const b = Buffer.alloc(33); + b.writeUInt32BE(0x89504e47, 0); + b.writeUInt32BE(0x0d0a1a0a, 4); + b.writeUInt32BE(13, 8); + b.write("IHDR", 12, "latin1"); + b.writeUInt32BE(w, 16); + b.writeUInt32BE(h, 20); + return b; +} + +async function capture(channelsDir: string, channel: string, id: string, files: Record<string, string | Buffer>) { + const dir = citedCaptureSourceDir(channelsDir, channel, id); + await mkdir(dir, { recursive: true }); + for (const [name, body] of Object.entries(files)) await writeFile(path.join(dir, name), body); +} + +async function tree(dir: string): Promise<string[]> { + const out: string[] = []; + const walk = async (d: string, rel: string) => { + for (const e of await readdir(d, { withFileTypes: true }).catch(() => [])) { + const r = rel ? `${rel}/${e.name}` : e.name; + if (e.isDirectory()) await walk(path.join(d, e.name), r); + else out.push(r); + } + }; + await walk(dir, ""); + return out.sort(); +} + +test("copies only the cited posts' screenshot and media — never the record, the article or an uncited post", async () => { + const channelsDir = path.join(ROOT, "a", "channels"); + const destRoot = path.join(ROOT, "a", "report-media"); + await capture(channelsDir, "demo-x", "111", { + "shot.png": png(600, 400), + "capture.json": "{}", + "111_1.jpg": "jpeg one", + "article.md": "an article", + "article-img-1.jpg": "inline", + "111_2.mp4.part": "partial", + }); + await capture(channelsDir, "demo-x", "222", { "shot.png": png(10, 10), "222_1.jpg": "not cited" }); + await capture(channelsDir, "other-x", "333", { "shot.png": png(10, 10) }); + + const r = await copyCitedPostCaptures({ channelsDir, destRoot, posts: [{ channel: "demo-x", id: "111" }] }); + assert.deepEqual(r.missing, []); + assert.deepEqual(await tree(destRoot), ["posts/demo-x/111/111_1.jpg", "posts/demo-x/111/shot.png"]); + assert.equal(r.copied.length, 1); + const c = r.copied[0]; + assert.equal(c.shot.file, "posts/demo-x/111/shot.png"); + assert.equal(c.shot.width, 600); + assert.equal(c.shot.height, 400); + assert.deepEqual(c.media, [ + { + file: "posts/demo-x/111/111_1.jpg", + bytes: "jpeg one".length, + sha256: createHash("sha256").update("jpeg one").digest("hex"), + }, + ]); +}); + +test("a later run with a different citation set leaves exactly that set", async () => { + const channelsDir = path.join(ROOT, "b", "channels"); + const destRoot = path.join(ROOT, "b", "report-media"); + await capture(channelsDir, "demo-x", "111", { "shot.png": png(1, 1), "111_1.jpg": "x" }); + await capture(channelsDir, "demo-x", "222", { "shot.png": png(1, 1) }); + await capture(channelsDir, "other-x", "333", { "shot.png": png(1, 1) }); + await copyCitedPostCaptures({ + channelsDir, + destRoot, + posts: [ + { channel: "demo-x", id: "111" }, + { channel: "other-x", id: "333" }, + ], + }); + // The capture lost its media since; 111 is no longer cited at all. + await rm(path.join(citedCaptureSourceDir(channelsDir, "demo-x", "111"), "111_1.jpg")); + await writeFile(path.join(destRoot, "posts", "stray.txt"), "left by hand"); + const r = await copyCitedPostCaptures({ channelsDir, destRoot, posts: [{ channel: "demo-x", id: "222" }] }); + assert.equal(r.copied.length, 1); + assert.deepEqual(await tree(destRoot), ["posts/demo-x/222/shot.png"]); +}); + +test("a cited post with no screenshot, or an id that is not one, is missing — and nothing is copied for it", async () => { + const channelsDir = path.join(ROOT, "c", "channels"); + const destRoot = path.join(ROOT, "c", "report-media"); + await capture(channelsDir, "demo-x", "111", { "111_1.jpg": "media only" }); + const r = await copyCitedPostCaptures({ + channelsDir, + destRoot, + posts: [ + { channel: "demo-x", id: "111" }, + { channel: "demo-x", id: "999" }, + { channel: "demo-x", id: "../escape" }, + ], + }); + assert.deepEqual(r.copied, []); + assert.deepEqual( + r.missing.map((m) => m.id), + ["111", "999", "../escape"], + ); + assert.match(r.missing[2].message, /is not a post id/); + assert.deepEqual(await tree(destRoot), []); +}); + +test("pngSize: the IHDR size, or null for anything else", async () => { + const f = path.join(ROOT, "x.png"); + await writeFile(f, png(1280, 720)); + assert.deepEqual(await pngSize(f), { width: 1280, height: 720 }); + await writeFile(f, "not a png at all, but long enough"); + assert.equal(await pngSize(f), null); + assert.equal(await pngSize(path.join(ROOT, "absent.png")), null); +}); diff --git a/common/publish/citedPostCaptures.ts b/common/publish/citedPostCaptures.ts @@ -0,0 +1,149 @@ +// THE ONE PUBLISH MODULE THAT READS A POST CAPTURE — and only the captures a +// report cites. +// +// A post capture (`channels/<slug>/posts-media/<id>/`: the post's screenshot, +// its attached media, its record) is editor-only material: nothing the export +// builds may carry it, and social/postCapture.test.ts holds every publish and +// export source to that by name. A report site is the one exception the plan +// makes (plans/report-sites.md, "Evidence media"): a cited post's moment page +// shows its screenshot and media. So the exception is THIS module, named in +// that guard, and it copies exactly the posts it is handed — the caller's +// cited list, already narrowed by the site's channel pool and the post +// visibility rule (lib/postsVisibility.ts) — and nothing else: +// +// - `shot.png` and the media files (`listCapturedMediaFiles`): never the +// record (`capture.json`), never the article half, never a leftover; +// - into `<destRoot>/posts/<channel>/<id>/`, each post's directory rebuilt +// from scratch, so a file the capture no longer has does not linger; +// - and every other directory under `<destRoot>/posts/` is REMOVED: what is +// there after a run is the cited set, whatever an earlier run cited. +// +// A post with no screenshot is missing (its moment page has nothing to show); +// the editor's Capture posts fills it. + +import { open, readdir, rm, stat } from "node:fs/promises"; +import path from "node:path"; +import { copyFileAtomic } from "../lib/jsonFile-server"; +import { + fileDigest, + listCapturedMediaFiles, + postCaptureDir, + postsMediaDir, + SHOT_FILENAME, +} from "../social/postCapture"; + +// The directory under the report-media cache that holds the copied captures. +export const REPORT_POSTS_DIRNAME = "posts"; + +export type CitedPost = { channel: string; id: string }; + +export type CopiedFile = { + // Relative to the report-media cache directory, `/`-separated. + file: string; + bytes: number; + sha256: string; +}; + +export type CopiedCapture = CitedPost & { + shot: CopiedFile & { width: number | null; height: number | null }; + media: CopiedFile[]; +}; + +export type MissingCapture = CitedPost & { message: string }; + +// A PNG's size from its header: the IHDR chunk is always first, its width and +// height big-endian at bytes 16 and 20. Null for anything that is not a PNG. +export async function pngSize(file: string): Promise<{ width: number; height: number } | null> { + const head = Buffer.alloc(24); + try { + const fh = await open(file, "r"); + try { + await fh.read(head, 0, 24, 0); + } finally { + await fh.close(); + } + } catch { + return null; + } + if (head.readUInt32BE(0) !== 0x89504e47 || head.toString("latin1", 12, 16) !== "IHDR") { + return null; + } + return { width: head.readUInt32BE(16), height: head.readUInt32BE(20) }; +} + +// Where a post's capture is: its channel's capture directory, the id checked +// (lib's postCaptureDir throws on one that is not a post id). +export function citedCaptureSourceDir(channelsDir: string, channel: string, id: string): string { + return postCaptureDir(postsMediaDir(path.join(channelsDir, channel)), id); +} + +const isFile = async (p: string) => (await stat(p).catch(() => null))?.isFile() === true; + +// Copy the cited posts' captures into `<destRoot>/posts/`, and remove every +// capture there that is not cited. `channelsDir` is the corpus's +// `channels/`. +export async function copyCitedPostCaptures(opts: { + channelsDir: string; + destRoot: string; + posts: readonly CitedPost[]; +}): Promise<{ copied: CopiedCapture[]; missing: MissingCapture[] }> { + const postsRoot = path.join(opts.destRoot, REPORT_POSTS_DIRNAME); + const copied: CopiedCapture[] = []; + const missing: MissingCapture[] = []; + const keep = new Set<string>(); + + for (const post of opts.posts) { + const key = `${post.channel}/${post.id}`; + if (keep.has(key)) continue; + keep.add(key); + let src: string; + try { + src = citedCaptureSourceDir(opts.channelsDir, post.channel, post.id); + } catch (e) { + missing.push({ ...post, message: (e as Error).message }); + continue; + } + const dest = path.join(postsRoot, post.channel, post.id); + await rm(dest, { recursive: true, force: true }); + if (!(await isFile(path.join(src, SHOT_FILENAME)))) { + missing.push({ ...post, message: "the post has no captured screenshot (capture it on the channel's Posts page)" }); + continue; + } + const copy = async (name: string): Promise<CopiedFile> => { + await copyFileAtomic(path.join(src, name), path.join(dest, name), { mkdir: true }); + const digest = await fileDigest(path.join(dest, name)); + return { file: path.posix.join(REPORT_POSTS_DIRNAME, post.channel, post.id, name), ...digest }; + }; + const shot = await copy(SHOT_FILENAME); + const size = await pngSize(path.join(dest, SHOT_FILENAME)); + const media: CopiedFile[] = []; + for (const name of await listCapturedMediaFiles(src)) { + // A file through its links; a directory or a dangling link is skipped. + if (await isFile(path.join(src, name))) media.push(await copy(name)); + } + copied.push({ + ...post, + shot: { ...shot, width: size?.width ?? null, height: size?.height ?? null }, + media, + }); + } + + // Everything else goes: an uncited post, an uncited channel. + for (const channel of await readdir(postsRoot).catch(() => [] as string[])) { + const channelDir = path.join(postsRoot, channel); + const ids = await readdir(channelDir).catch(() => null); + if (ids === null) { + await rm(channelDir, { recursive: true, force: true }); + continue; + } + let kept = 0; + for (const id of ids) { + if (keep.has(`${channel}/${id}`)) kept++; + else await rm(path.join(channelDir, id), { recursive: true, force: true }); + } + if (kept === 0) await rm(channelDir, { recursive: true, force: true }); + } + if (keep.size === 0) await rm(postsRoot, { recursive: true, force: true }); + + return { copied, missing }; +} diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts @@ -167,12 +167,16 @@ test("article files are named by the layout, images numbered from 1", () => { assert.ok(!isArticleFile("articles.txt")); }); -// THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON -// pages and copies trees out of export/public — it never reads a channel -// directory — so nothing it builds can carry posts-media/. Held here the cheap -// way: no module of the export build, the publish layer or the export app -// names the directory. -test("no export or publish source names posts-media/", async () => { +// THE EXPORT NEVER PUBLISHES A CAPTURE — but the one a report cites. The +// export serves the index's JSON pages and copies trees out of export/public — +// it never reads a channel directory — so nothing it builds can carry +// posts-media/. Held here the cheap way: no module of the export build, the +// publish layer or the export app names the directory, except EXACTLY the one +// that copies a report site's CITED captures (publish/citedPostCaptures.ts, +// whose own test proves it copies nothing else). +const CITED_CAPTURE_COPIER = path.join("common", "publish", "citedPostCaptures.ts"); + +test("no export or publish source names posts-media/ but the cited-capture copier", async () => { const HERE = path.dirname(fileURLToPath(import.meta.url)); const repo = path.resolve(HERE, "..", ".."); const roots = [ @@ -182,6 +186,7 @@ test("no export or publish source names posts-media/", async () => { path.join(repo, "export", "lib"), ]; const offenders: string[] = []; + const allowed: string[] = []; const walk = async (dir: string): Promise<void> => { let entries; try { @@ -196,11 +201,15 @@ test("no export or publish source names posts-media/", async () => { else if (/\.(ts|tsx|mjs|js)$/.test(e.name)) { const text = await readFile(p, "utf8"); if (text.includes(POSTS_MEDIA_DIRNAME) || text.includes("POSTS_MEDIA_DIRNAME") || text.includes("social/postCapture")) { - offenders.push(path.relative(repo, p)); + const rel = path.relative(repo, p); + if (rel === CITED_CAPTURE_COPIER) allowed.push(rel); + else offenders.push(rel); } } } }; for (const r of roots) await walk(r); assert.deepEqual(offenders, []); + // The exception exists and is the only one: if it moves, this moves with it. + assert.deepEqual(allowed, [CITED_CAPTURE_COPIER]); });