Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit f911fb85f95e590fc807d9f049bb1434d7678218
parent 12ca8cf0aa2b62b91d7525c1c8c5618cc11fe91d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 20:13:50 -0400

common: tests for the X Article capture and its place in the run; the post dir exists before the images land

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/social/xArticleCapture.test.ts | 472+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xArticleCapture.ts | 3++-
2 files changed, 474 insertions(+), 1 deletion(-)

diff --git a/common/social/xArticleCapture.test.ts b/common/social/xArticleCapture.test.ts @@ -0,0 +1,472 @@ +// The X Article half of a post capture, against fakes only: a page that +// answers the capture's evaluate calls from recorded snapshots and the written +// HTML fixture, and a request context that serves images from a table. No +// browser, no network. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticleCapture.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { mkdtemp, readdir, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import type { PageLike } from "./playwrightRuntime"; +import { + ARTICLE_SHOT_MAX_HEIGHT, + captureXArticle, + classifyXArticleSnapshot, + fullSizeImageUrl, + imageExtension, + type XArticleSnapshot, +} from "./xArticleCapture"; +import { captureXPosts, type XCaptureDeps, type XPostSnapshot } from "./xPostCapture"; +import { readPostCapture, writePostCapture } from "./postCapture"; +import type { PostCaptureInput } from "./fetchers"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8"); +const sha = (b: string | Uint8Array) => createHash("sha256").update(b).digest("hex"); +const PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 7, 7]); +const NOW = () => new Date("2026-02-03T04:05:06.000Z"); +const LINK = { articleId: "999", url: "https://x.com/i/article/999" }; + +const COVER = "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small"; +const BODY_IMG = "https://pbs.twimg.com/media/BodyImg1?format=png&name=small"; +const orig = (u: string) => fullSizeImageUrl(u); + +const articleSnap = (height = 2400): XArticleSnapshot => ({ + path: "/i/article/999", + text: "A Worked Example", + root: { rect: { x: 0, y: 53.5, width: 600.2, height }, marker: '[data-testid="twitterArticleReadView"]' }, +}); +const noArticle = (text: string, p = "/i/article/999"): XArticleSnapshot => ({ path: p, text, root: null }); + +type Served = Record<string, { status: number; body?: string; type?: string }>; + +type FakeArticlePage = PageLike & { calls: string[]; shots: unknown[]; fetched: string[] }; + +// A page for articles: the snapshot script answers `snaps` in turn (the last +// repeats), the HTML script the fixture, the lazy-load walk nothing; images +// come from `served` (every full-size picture by default). +function articlePage( + snaps: XArticleSnapshot[], + opts: { html?: string; served?: Served; gotoFails?: boolean; noRequest?: boolean } = {}, +): FakeArticlePage { + let i = 0; + const calls: string[] = []; + const shots: unknown[] = []; + const fetched: string[] = []; + const served: Served = opts.served ?? { + [orig(COVER)]: { status: 200, body: "cover bytes", type: "image/jpeg" }, + [orig(BODY_IMG)]: { status: 200, body: "chart bytes", type: "image/png" }, + }; + const p: FakeArticlePage = { + calls, + shots, + fetched, + goto: async (url: string) => { + calls.push(`goto ${url}`); + if (opts.gotoFails) throw new Error("net::ERR_TIMED_OUT\nat goto"); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: async (script: string) => { + if (script.includes("outerHTML")) return opts.html ?? FIXTURE; + if (script.includes("scrollTo")) { + calls.push("load-lazy"); + return 3; + } + calls.push("snapshot"); + return snaps[Math.min(i++, snaps.length - 1)]; + }, + screenshot: async (o?: unknown) => { + shots.push(o); + return PNG; + }, + }; + if (!opts.noRequest) { + p.request = { + get: async (url: string) => { + fetched.push(url); + const s = served[url] ?? { status: 404 }; + return { + ok: () => s.status >= 200 && s.status < 300, + status: () => s.status, + headers: (): Record<string, string> => (s.type ? { "content-type": s.type } : {}), + body: async () => new TextEncoder().encode(s.body ?? ""), + }; + }, + }; + } + return p; +} + +const tmpDir = async () => path.join(await mkdtemp(path.join(os.tmpdir(), "xarticle-")), "1"); + +// --- classification and helpers ------------------------------------------------- + +test("classify: an article page with a root is captured; walls, deletions and refusals as a post's", () => { + assert.deepEqual(classifyXArticleSnapshot(articleSnap()), { state: "captured" }); + const login = classifyXArticleSnapshot(noArticle("", "/i/flow/login")); + assert.equal(login.state, "login-wall"); + assert.ok(login.stop); + assert.equal(classifyXArticleSnapshot(noArticle("Log in Sign up")).state, "login-wall"); + assert.deepEqual(classifyXArticleSnapshot(noArticle("Hmm...this page doesn’t exist.")), { state: "deleted" }); + assert.deepEqual(classifyXArticleSnapshot(noArticle("This account is suspended")), { state: "unavailable" }); + const refused = classifyXArticleSnapshot(noArticle("Something went wrong. Try reloading.")); + assert.equal(refused.state, "error"); + assert.ok(refused.stop); + assert.match(refused.error ?? "", /instead of the article/); + assert.deepEqual(classifyXArticleSnapshot(noArticle("")), { state: "error", error: "The article did not render." }); +}); + +test("images: the full-size picture of an X media URL, and an extension from the format, path or type", () => { + assert.equal(fullSizeImageUrl(COVER), "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=orig"); + assert.equal(fullSizeImageUrl("https://example.com/a.png?name=small"), "https://example.com/a.png?name=small"); + assert.equal(fullSizeImageUrl("not a url"), "not a url"); + assert.equal(imageExtension(COVER), "jpg"); + assert.equal(imageExtension("https://example.com/p/pic.JPEG"), "jpg"); + assert.equal(imageExtension("https://example.com/p/pic", "image/webp; q=1"), "webp"); + assert.equal(imageExtension("https://example.com/p/pic"), "img"); +}); + +// --- one article ------------------------------------------------------------------ + +test("article: read, shot whole, images fetched full-size through the page; every file recorded", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap(), articleSnap(2500.4)]); + const gaps: number[] = []; + const res = await captureXArticle(p, LINK, dir, { now: NOW, imageGap: async () => void gaps.push(1) }); + assert.equal(res.stop, undefined); + assert.equal(p.calls[0], "goto https://x.com/i/article/999"); + assert.ok(p.calls.includes("load-lazy")); + // The box re-read after the lazy images loaded. + assert.deepEqual(p.shots, [ + { type: "png", fullPage: true, clip: { x: 0, y: 53, width: 601, height: 2501 } }, + ]); + assert.deepEqual(p.fetched, [orig(COVER), orig(BODY_IMG)]); + assert.deepEqual(gaps, [1], "a gap between images, none before the first"); + + const r = res.record; + assert.equal(r.state, "captured"); + assert.equal(r.title, "A Worked Example & Its Notes"); + assert.equal(r.blocks, 12); + assert.equal(r.extraction, "structured"); + assert.equal(r.trimmed, undefined); + assert.equal(r.error, undefined); + assert.deepEqual( + r.files.map((f) => [f.name, f.url]), + [ + ["article.json", LINK.url], + ["article.md", LINK.url], + ["article.html", LINK.url], + ["article.png", LINK.url], + ["article-img-1.jpg", orig(COVER)], + ["article-img-2.png", orig(BODY_IMG)], + ], + ); + const img = r.files.find((f) => f.name === "article-img-2.png")!; + assert.deepEqual([img.bytes, img.sha256], ["chart bytes".length, sha("chart bytes")]); + assert.deepEqual(new Uint8Array(await readFile(path.join(dir, "article.png"))), PNG); + assert.equal(await readFile(path.join(dir, "article.html"), "utf8"), FIXTURE); + + const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")); + assert.equal(json.articleId, "999"); + assert.equal(json.url, LINK.url); + assert.equal(json.author, "Demo Author"); + assert.equal(json.handle, "@demo_author"); + assert.equal(json.publishedAt, "2026-01-02T03:04:05.000Z"); + assert.equal(json.capturedAt, "2026-02-03T04:05:06.000Z"); + assert.equal(json.blocks.length, 12); + assert.deepEqual(json.blocks[8], { + type: "image", + src: BODY_IMG, + text: "A chart of the numbers", + file: "article-img-2.png", + }); + assert.deepEqual(json.blocks[9], { + type: "embedded-post", + text: "An embedded post's own words.", + href: "https://x.com/other_user/status/4444444444", + }); + const md = await readFile(path.join(dir, "article.md"), "utf8"); + assert.match(md, /^# A Worked Example & Its Notes\n/); + assert.match(md, /!\[A chart of the numbers\]\(<article-img-2\.png>\)/); + assert.match(md, /> Embedded post: <https:\/\/x\.com\/other_user\/status\/4444444444>/); +}); + +test("article: a tall one is shot to the cap and recorded trimmed", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap(50_000)]); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.equal(res.record.trimmed, true); + assert.equal((p.shots[0] as { clip: { height: number } }).clip.height, ARTICLE_SHOT_MAX_HEIGHT); + assert.equal(JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")).trimmed, true); +}); + +test("article: a full-size picture refused falls back to the src as rendered; one not served at all is noted", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap()], { + served: { [COVER]: { status: 200, body: "small cover", type: "image/jpeg" } }, + }); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.deepEqual(p.fetched, [orig(COVER), COVER, orig(BODY_IMG), BODY_IMG]); + assert.equal(res.record.state, "captured"); + assert.deepEqual( + res.record.files.filter((f) => f.name.startsWith("article-img-")).map((f) => [f.name, f.url]), + [["article-img-1.jpg", COVER]], + ); + assert.match(res.record.error ?? "", /1 image\(s\) could not be fetched: HTTP 404/); + const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")); + assert.equal(json.blocks[8].file, undefined, "an image not saved links to its src"); +}); + +test("article: a page with no request context still saves the text, and says the images are missing", async () => { + const res = await captureXArticle(articlePage([articleSnap()], { noRequest: true }), LINK, await tmpDir(), { + now: NOW, + }); + assert.equal(res.record.state, "captured"); + assert.match(res.record.error ?? "", /no request context/); + assert.equal(res.record.files.length, 4); +}); + +test("article: a re-capture with fewer images leaves no older numbered file behind", async () => { + const dir = await tmpDir(); + await captureXArticle(articlePage([articleSnap()]), LINK, dir, { now: NOW }); + await writeFile(path.join(dir, "article-img-3.jpg"), "stale"); + await captureXArticle(articlePage([articleSnap()], { html: "<div data-testid=\"twitterArticleRichTextView\"><p>Just text</p></div>" }), LINK, dir, { now: NOW }); + const names = (await readdir(dir)).sort(); + assert.deepEqual(names, ["article.html", "article.json", "article.md", "article.png"]); +}); + +test("article: a login wall stops the run; deleted is recorded; a failed load is an error; no file is written", async () => { + for (const [p, state, stops] of [ + [articlePage([noArticle("", "/i/flow/login")]), "login-wall", true], + [articlePage([noArticle("Something went wrong. Try reloading.")]), "error", true], + [articlePage([noArticle("Hmm...this page doesn’t exist.")]), "deleted", false], + [articlePage([articleSnap()], { gotoFails: true }), "error", false], + ] as const) { + const dir = await tmpDir(); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.equal(res.record.state, state); + assert.equal(Boolean(res.stop), stops, state); + assert.deepEqual(res.record.files, []); + assert.equal(res.record.blocks, 0); + assert.equal(p.shots.length, 0); + assert.deepEqual(await readdir(dir).catch(() => []), []); + } +}); + +// --- in the run ---------------------------------------------------------------- + +const post = (articleHrefs?: string[]): XPostSnapshot => ({ + path: "/someone/status/1", + text: "the post", + article: { rect: { x: 10, y: 120, width: 598, height: 300 }, sensitive: false, ...(articleHrefs ? { articleHrefs } : {}) }, +}); + +// Post pages by id, article pages by article id, one routed page over both. +function runHarness( + postsById: Record<string, XPostSnapshot>, + articlesById: Record<string, XArticleSnapshot[]>, +): { deps: XCaptureDeps; pauses: number[]; visits: string[]; downloads: string[] } { + const visits: string[] = []; + const pauses: number[] = []; + const downloads: string[] = []; + let current: PageLike = articlePage([]); + const articlePages = Object.fromEntries( + Object.entries(articlesById).map(([id, snaps]) => [id, articlePage(snaps)]), + ); + const routed: PageLike = { + goto: async (url: string, o?: unknown) => { + visits.push(url); + const id = url.split("/").pop()!; + if (url.includes("/article/")) { + current = articlePages[id]; + } else { + const snap = postsById[id]; + current = { + ...articlePage([]), + evaluate: async (s: string) => (s.includes("imgs.length") ? 0 : snap), + screenshot: async () => PNG, + }; + } + return current.goto(url, o); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: (s: string) => current.evaluate(s), + screenshot: (o) => current.screenshot(o), + request: { get: (url, o) => current.request!.get(url, o) }, + }; + return { + visits, + pauses, + downloads, + deps: { + openPage: async () => ({ page: routed, close: async () => {} }), + downloadMedia: async ({ id }) => { + downloads.push(id); + return { ok: true, files: [] }; + }, + pauseMs: () => 5_000, + pause: async (ms) => void pauses.push(ms), + articleImagePauseMs: () => 1_000, + now: NOW, + }, + }; +} + +async function input(ids: string[], over: Partial<PostCaptureInput> = {}): Promise<PostCaptureInput> { + return { + ids, + handle: "example_user", + outDir: await mkdtemp(path.join(os.tmpdir(), "posts-media-")), + signal: new AbortController().signal, + ...over, + }; +} + +const ARCHIVED = new Map([["11", { text: "https://x.com/i/article/999" }]]); + +test("run: a post whose archived text is an article link — shot, media, then the article, each paced", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + const inp = await input(["11", "22"], { archived: ARCHIVED }); + const res = await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, [ + "https://x.com/i/status/11", + "https://x.com/i/article/999", + "https://x.com/i/status/22", + ]); + // Five contacts (two shots, two downloads, one article): four post gaps, + // and one short gap between the article's two images. + assert.deepEqual(h.pauses, [5_000, 5_000, 1_000, 5_000, 5_000]); + assert.deepEqual( + res.outcomes.map((o) => [o.id, o.state, o.files]), + [ + ["11", "captured", 7], + ["22", "captured", 1], + ], + ); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.article?.state, "captured"); + assert.equal(rec.article?.articleId, "999"); + assert.equal(rec.article?.title, "A Worked Example & Its Notes"); + assert.equal(rec.article?.blocks, 12); + assert.equal(rec.article?.files.length, 6); + assert.equal((await readPostCapture(path.join(inp.outDir, "22")))!.article, undefined); + + // Captured is settled: a second run opens nothing. + const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []); + assert.deepEqual(again.visits, []); + + // force reads it again. + const forced = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + await captureXPosts({ ...inp, ids: ["11"], force: true }, forced.deps); + assert.deepEqual(forced.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]); +}); + +test("run: an article found only on the post's card is opened too; articles: false opens none", async () => { + const h = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] }); + const inp = await input(["11"]); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/demo_author/article/999"]); + assert.equal((await readPostCapture(path.join(inp.outDir, "11")))!.article?.state, "captured"); + + const off = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] }); + const inp2 = await input(["11"], { archived: ARCHIVED, articles: false }); + await captureXPosts(inp2, off.deps); + assert.deepEqual(off.visits, ["https://x.com/i/status/11"]); + assert.equal((await readPostCapture(path.join(inp2.outDir, "11")))!.article, undefined); +}); + +test("run: a post captured before articles owes only its article — no second shot or download", async () => { + const first = runHarness({ "11": post() }, { "999": [articleSnap()] }); + const inp = await input(["11"], { articles: false }); + await captureXPosts(inp, first.deps); + + const h = runHarness({ "11": post() }, { "999": [articleSnap()] }); + const res = await captureXPosts({ ...inp, articles: true, archived: ARCHIVED }, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/article/999"]); + assert.deepEqual(h.downloads, []); + assert.equal(res.outcomes[0].availability, undefined, "an article-only pass says nothing about the post's liveness"); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.state, "captured"); + assert.ok(rec.shot, "the shot is kept"); + assert.equal(rec.article?.state, "captured"); +}); + +test("run: an article behind a login wall stops the run with needsCookies; the rest untouched", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("", "/i/flow/login")] }); + const inp = await input(["11", "22"], { archived: ARCHIVED, media: false }); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.needsCookies, true); + assert.match(res.stoppedEarly ?? "", /Stopped at 11/); + assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.state, "captured", "the post itself was shot"); + assert.equal(rec.article?.state, "login-wall"); + assert.equal(await readPostCapture(path.join(inp.outDir, "22")), null); + + // A login wall is owed again on the next run. + const next = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + await captureXPosts({ ...inp, ids: ["11"] }, next.deps); + assert.deepEqual(next.visits, ["https://x.com/i/article/999"]); +}); + +test("run: a deleted article is recorded and settled; the run goes on", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("Hmm...this page doesn’t exist.")] }); + const inp = await input(["11", "22"], { archived: ARCHIVED, media: false }); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.stoppedEarly, undefined); + assert.equal(res.outcomes.length, 2); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.article?.state, "deleted"); + const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []); + assert.deepEqual(again.visits, []); +}); + +test("run: a deleted post opens no article", async () => { + const h = runHarness( + { "11": { path: "/i/status/11", text: "This post was deleted by the post author.", article: null } }, + { "999": [articleSnap()] }, + ); + const inp = await input(["11"], { archived: ARCHIVED }); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/status/11"]); +}); + +test("run: an older record's article is not lost by a run that does not read it", async () => { + const inp = await input(["11"], { archived: ARCHIVED }); + const dir = path.join(inp.outDir, "11"); + const article = { + articleId: "999", + url: LINK.url, + capturedAt: "2026-01-01T00:00:00.000Z", + state: "captured" as const, + blocks: 3, + files: [], + }; + await writePostCapture(dir, { + version: 1, + id: "11", + url: "https://x.com/i/status/11", + capturedAt: "2026-01-01T00:00:00.000Z", + state: "captured", + shot: { name: "shot.png", bytes: 1, sha256: "x" }, + mediaState: "error", + media: [], + article, + }); + const h = runHarness({ "11": post() }, {}); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.downloads, ["11"], "only the owed media"); + assert.deepEqual(h.visits, []); + assert.deepEqual((await readPostCapture(dir))!.article, article); +}); diff --git a/common/social/xArticleCapture.ts b/common/social/xArticleCapture.ts @@ -22,7 +22,7 @@ // NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing, // tested against recorded snapshots and a written HTML fixture, never x.com. -import { readdir, rm, writeFile } from "node:fs/promises"; +import { mkdir, readdir, rm, writeFile } from "node:fs/promises"; import path from "node:path"; import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server"; import type { PageLike } from "./playwrightRuntime"; @@ -267,6 +267,7 @@ export async function captureXArticle( } // The images, each once, numbered in reading order. + await mkdir(dir, { recursive: true }); const images = await fetchArticleImages(page, content.blocks, dir, opts); if (images.errors.length) { const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`;