commit f911fb85f95e590fc807d9f049bb1434d7678218
parent 12ca8cf0aa2b62b91d7525c1c8c5618cc11fe91d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 20:13:50 -0400
common: tests for the X Article capture and its place in the run; the post dir exists before the images land
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 474 insertions(+), 1 deletion(-)
diff --git a/common/social/xArticleCapture.test.ts b/common/social/xArticleCapture.test.ts
@@ -0,0 +1,472 @@
+// The X Article half of a post capture, against fakes only: a page that
+// answers the capture's evaluate calls from recorded snapshots and the written
+// HTML fixture, and a request context that serves images from a table. No
+// browser, no network.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticleCapture.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { createHash } from "node:crypto";
+import { mkdtemp, readdir, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import type { PageLike } from "./playwrightRuntime";
+import {
+ ARTICLE_SHOT_MAX_HEIGHT,
+ captureXArticle,
+ classifyXArticleSnapshot,
+ fullSizeImageUrl,
+ imageExtension,
+ type XArticleSnapshot,
+} from "./xArticleCapture";
+import { captureXPosts, type XCaptureDeps, type XPostSnapshot } from "./xPostCapture";
+import { readPostCapture, writePostCapture } from "./postCapture";
+import type { PostCaptureInput } from "./fetchers";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8");
+const sha = (b: string | Uint8Array) => createHash("sha256").update(b).digest("hex");
+const PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 7, 7]);
+const NOW = () => new Date("2026-02-03T04:05:06.000Z");
+const LINK = { articleId: "999", url: "https://x.com/i/article/999" };
+
+const COVER = "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small";
+const BODY_IMG = "https://pbs.twimg.com/media/BodyImg1?format=png&name=small";
+const orig = (u: string) => fullSizeImageUrl(u);
+
+const articleSnap = (height = 2400): XArticleSnapshot => ({
+ path: "/i/article/999",
+ text: "A Worked Example",
+ root: { rect: { x: 0, y: 53.5, width: 600.2, height }, marker: '[data-testid="twitterArticleReadView"]' },
+});
+const noArticle = (text: string, p = "/i/article/999"): XArticleSnapshot => ({ path: p, text, root: null });
+
+type Served = Record<string, { status: number; body?: string; type?: string }>;
+
+type FakeArticlePage = PageLike & { calls: string[]; shots: unknown[]; fetched: string[] };
+
+// A page for articles: the snapshot script answers `snaps` in turn (the last
+// repeats), the HTML script the fixture, the lazy-load walk nothing; images
+// come from `served` (every full-size picture by default).
+function articlePage(
+ snaps: XArticleSnapshot[],
+ opts: { html?: string; served?: Served; gotoFails?: boolean; noRequest?: boolean } = {},
+): FakeArticlePage {
+ let i = 0;
+ const calls: string[] = [];
+ const shots: unknown[] = [];
+ const fetched: string[] = [];
+ const served: Served = opts.served ?? {
+ [orig(COVER)]: { status: 200, body: "cover bytes", type: "image/jpeg" },
+ [orig(BODY_IMG)]: { status: 200, body: "chart bytes", type: "image/png" },
+ };
+ const p: FakeArticlePage = {
+ calls,
+ shots,
+ fetched,
+ goto: async (url: string) => {
+ calls.push(`goto ${url}`);
+ if (opts.gotoFails) throw new Error("net::ERR_TIMED_OUT\nat goto");
+ },
+ waitForSelector: async () => undefined,
+ waitForTimeout: async () => undefined,
+ on: () => {},
+ evaluate: async (script: string) => {
+ if (script.includes("outerHTML")) return opts.html ?? FIXTURE;
+ if (script.includes("scrollTo")) {
+ calls.push("load-lazy");
+ return 3;
+ }
+ calls.push("snapshot");
+ return snaps[Math.min(i++, snaps.length - 1)];
+ },
+ screenshot: async (o?: unknown) => {
+ shots.push(o);
+ return PNG;
+ },
+ };
+ if (!opts.noRequest) {
+ p.request = {
+ get: async (url: string) => {
+ fetched.push(url);
+ const s = served[url] ?? { status: 404 };
+ return {
+ ok: () => s.status >= 200 && s.status < 300,
+ status: () => s.status,
+ headers: (): Record<string, string> => (s.type ? { "content-type": s.type } : {}),
+ body: async () => new TextEncoder().encode(s.body ?? ""),
+ };
+ },
+ };
+ }
+ return p;
+}
+
+const tmpDir = async () => path.join(await mkdtemp(path.join(os.tmpdir(), "xarticle-")), "1");
+
+// --- classification and helpers -------------------------------------------------
+
+test("classify: an article page with a root is captured; walls, deletions and refusals as a post's", () => {
+ assert.deepEqual(classifyXArticleSnapshot(articleSnap()), { state: "captured" });
+ const login = classifyXArticleSnapshot(noArticle("", "/i/flow/login"));
+ assert.equal(login.state, "login-wall");
+ assert.ok(login.stop);
+ assert.equal(classifyXArticleSnapshot(noArticle("Log in Sign up")).state, "login-wall");
+ assert.deepEqual(classifyXArticleSnapshot(noArticle("Hmm...this page doesn’t exist.")), { state: "deleted" });
+ assert.deepEqual(classifyXArticleSnapshot(noArticle("This account is suspended")), { state: "unavailable" });
+ const refused = classifyXArticleSnapshot(noArticle("Something went wrong. Try reloading."));
+ assert.equal(refused.state, "error");
+ assert.ok(refused.stop);
+ assert.match(refused.error ?? "", /instead of the article/);
+ assert.deepEqual(classifyXArticleSnapshot(noArticle("")), { state: "error", error: "The article did not render." });
+});
+
+test("images: the full-size picture of an X media URL, and an extension from the format, path or type", () => {
+ assert.equal(fullSizeImageUrl(COVER), "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=orig");
+ assert.equal(fullSizeImageUrl("https://example.com/a.png?name=small"), "https://example.com/a.png?name=small");
+ assert.equal(fullSizeImageUrl("not a url"), "not a url");
+ assert.equal(imageExtension(COVER), "jpg");
+ assert.equal(imageExtension("https://example.com/p/pic.JPEG"), "jpg");
+ assert.equal(imageExtension("https://example.com/p/pic", "image/webp; q=1"), "webp");
+ assert.equal(imageExtension("https://example.com/p/pic"), "img");
+});
+
+// --- one article ------------------------------------------------------------------
+
+test("article: read, shot whole, images fetched full-size through the page; every file recorded", async () => {
+ const dir = await tmpDir();
+ const p = articlePage([articleSnap(), articleSnap(2500.4)]);
+ const gaps: number[] = [];
+ const res = await captureXArticle(p, LINK, dir, { now: NOW, imageGap: async () => void gaps.push(1) });
+ assert.equal(res.stop, undefined);
+ assert.equal(p.calls[0], "goto https://x.com/i/article/999");
+ assert.ok(p.calls.includes("load-lazy"));
+ // The box re-read after the lazy images loaded.
+ assert.deepEqual(p.shots, [
+ { type: "png", fullPage: true, clip: { x: 0, y: 53, width: 601, height: 2501 } },
+ ]);
+ assert.deepEqual(p.fetched, [orig(COVER), orig(BODY_IMG)]);
+ assert.deepEqual(gaps, [1], "a gap between images, none before the first");
+
+ const r = res.record;
+ assert.equal(r.state, "captured");
+ assert.equal(r.title, "A Worked Example & Its Notes");
+ assert.equal(r.blocks, 12);
+ assert.equal(r.extraction, "structured");
+ assert.equal(r.trimmed, undefined);
+ assert.equal(r.error, undefined);
+ assert.deepEqual(
+ r.files.map((f) => [f.name, f.url]),
+ [
+ ["article.json", LINK.url],
+ ["article.md", LINK.url],
+ ["article.html", LINK.url],
+ ["article.png", LINK.url],
+ ["article-img-1.jpg", orig(COVER)],
+ ["article-img-2.png", orig(BODY_IMG)],
+ ],
+ );
+ const img = r.files.find((f) => f.name === "article-img-2.png")!;
+ assert.deepEqual([img.bytes, img.sha256], ["chart bytes".length, sha("chart bytes")]);
+ assert.deepEqual(new Uint8Array(await readFile(path.join(dir, "article.png"))), PNG);
+ assert.equal(await readFile(path.join(dir, "article.html"), "utf8"), FIXTURE);
+
+ const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8"));
+ assert.equal(json.articleId, "999");
+ assert.equal(json.url, LINK.url);
+ assert.equal(json.author, "Demo Author");
+ assert.equal(json.handle, "@demo_author");
+ assert.equal(json.publishedAt, "2026-01-02T03:04:05.000Z");
+ assert.equal(json.capturedAt, "2026-02-03T04:05:06.000Z");
+ assert.equal(json.blocks.length, 12);
+ assert.deepEqual(json.blocks[8], {
+ type: "image",
+ src: BODY_IMG,
+ text: "A chart of the numbers",
+ file: "article-img-2.png",
+ });
+ assert.deepEqual(json.blocks[9], {
+ type: "embedded-post",
+ text: "An embedded post's own words.",
+ href: "https://x.com/other_user/status/4444444444",
+ });
+ const md = await readFile(path.join(dir, "article.md"), "utf8");
+ assert.match(md, /^# A Worked Example & Its Notes\n/);
+ assert.match(md, /!\[A chart of the numbers\]\(<article-img-2\.png>\)/);
+ assert.match(md, /> Embedded post: <https:\/\/x\.com\/other_user\/status\/4444444444>/);
+});
+
+test("article: a tall one is shot to the cap and recorded trimmed", async () => {
+ const dir = await tmpDir();
+ const p = articlePage([articleSnap(50_000)]);
+ const res = await captureXArticle(p, LINK, dir, { now: NOW });
+ assert.equal(res.record.trimmed, true);
+ assert.equal((p.shots[0] as { clip: { height: number } }).clip.height, ARTICLE_SHOT_MAX_HEIGHT);
+ assert.equal(JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")).trimmed, true);
+});
+
+test("article: a full-size picture refused falls back to the src as rendered; one not served at all is noted", async () => {
+ const dir = await tmpDir();
+ const p = articlePage([articleSnap()], {
+ served: { [COVER]: { status: 200, body: "small cover", type: "image/jpeg" } },
+ });
+ const res = await captureXArticle(p, LINK, dir, { now: NOW });
+ assert.deepEqual(p.fetched, [orig(COVER), COVER, orig(BODY_IMG), BODY_IMG]);
+ assert.equal(res.record.state, "captured");
+ assert.deepEqual(
+ res.record.files.filter((f) => f.name.startsWith("article-img-")).map((f) => [f.name, f.url]),
+ [["article-img-1.jpg", COVER]],
+ );
+ assert.match(res.record.error ?? "", /1 image\(s\) could not be fetched: HTTP 404/);
+ const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8"));
+ assert.equal(json.blocks[8].file, undefined, "an image not saved links to its src");
+});
+
+test("article: a page with no request context still saves the text, and says the images are missing", async () => {
+ const res = await captureXArticle(articlePage([articleSnap()], { noRequest: true }), LINK, await tmpDir(), {
+ now: NOW,
+ });
+ assert.equal(res.record.state, "captured");
+ assert.match(res.record.error ?? "", /no request context/);
+ assert.equal(res.record.files.length, 4);
+});
+
+test("article: a re-capture with fewer images leaves no older numbered file behind", async () => {
+ const dir = await tmpDir();
+ await captureXArticle(articlePage([articleSnap()]), LINK, dir, { now: NOW });
+ await writeFile(path.join(dir, "article-img-3.jpg"), "stale");
+ await captureXArticle(articlePage([articleSnap()], { html: "<div data-testid=\"twitterArticleRichTextView\"><p>Just text</p></div>" }), LINK, dir, { now: NOW });
+ const names = (await readdir(dir)).sort();
+ assert.deepEqual(names, ["article.html", "article.json", "article.md", "article.png"]);
+});
+
+test("article: a login wall stops the run; deleted is recorded; a failed load is an error; no file is written", async () => {
+ for (const [p, state, stops] of [
+ [articlePage([noArticle("", "/i/flow/login")]), "login-wall", true],
+ [articlePage([noArticle("Something went wrong. Try reloading.")]), "error", true],
+ [articlePage([noArticle("Hmm...this page doesn’t exist.")]), "deleted", false],
+ [articlePage([articleSnap()], { gotoFails: true }), "error", false],
+ ] as const) {
+ const dir = await tmpDir();
+ const res = await captureXArticle(p, LINK, dir, { now: NOW });
+ assert.equal(res.record.state, state);
+ assert.equal(Boolean(res.stop), stops, state);
+ assert.deepEqual(res.record.files, []);
+ assert.equal(res.record.blocks, 0);
+ assert.equal(p.shots.length, 0);
+ assert.deepEqual(await readdir(dir).catch(() => []), []);
+ }
+});
+
+// --- in the run ----------------------------------------------------------------
+
+const post = (articleHrefs?: string[]): XPostSnapshot => ({
+ path: "/someone/status/1",
+ text: "the post",
+ article: { rect: { x: 10, y: 120, width: 598, height: 300 }, sensitive: false, ...(articleHrefs ? { articleHrefs } : {}) },
+});
+
+// Post pages by id, article pages by article id, one routed page over both.
+function runHarness(
+ postsById: Record<string, XPostSnapshot>,
+ articlesById: Record<string, XArticleSnapshot[]>,
+): { deps: XCaptureDeps; pauses: number[]; visits: string[]; downloads: string[] } {
+ const visits: string[] = [];
+ const pauses: number[] = [];
+ const downloads: string[] = [];
+ let current: PageLike = articlePage([]);
+ const articlePages = Object.fromEntries(
+ Object.entries(articlesById).map(([id, snaps]) => [id, articlePage(snaps)]),
+ );
+ const routed: PageLike = {
+ goto: async (url: string, o?: unknown) => {
+ visits.push(url);
+ const id = url.split("/").pop()!;
+ if (url.includes("/article/")) {
+ current = articlePages[id];
+ } else {
+ const snap = postsById[id];
+ current = {
+ ...articlePage([]),
+ evaluate: async (s: string) => (s.includes("imgs.length") ? 0 : snap),
+ screenshot: async () => PNG,
+ };
+ }
+ return current.goto(url, o);
+ },
+ waitForSelector: async () => undefined,
+ waitForTimeout: async () => undefined,
+ on: () => {},
+ evaluate: (s: string) => current.evaluate(s),
+ screenshot: (o) => current.screenshot(o),
+ request: { get: (url, o) => current.request!.get(url, o) },
+ };
+ return {
+ visits,
+ pauses,
+ downloads,
+ deps: {
+ openPage: async () => ({ page: routed, close: async () => {} }),
+ downloadMedia: async ({ id }) => {
+ downloads.push(id);
+ return { ok: true, files: [] };
+ },
+ pauseMs: () => 5_000,
+ pause: async (ms) => void pauses.push(ms),
+ articleImagePauseMs: () => 1_000,
+ now: NOW,
+ },
+ };
+}
+
+async function input(ids: string[], over: Partial<PostCaptureInput> = {}): Promise<PostCaptureInput> {
+ return {
+ ids,
+ handle: "example_user",
+ outDir: await mkdtemp(path.join(os.tmpdir(), "posts-media-")),
+ signal: new AbortController().signal,
+ ...over,
+ };
+}
+
+const ARCHIVED = new Map([["11", { text: "https://x.com/i/article/999" }]]);
+
+test("run: a post whose archived text is an article link — shot, media, then the article, each paced", async () => {
+ const h = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] });
+ const inp = await input(["11", "22"], { archived: ARCHIVED });
+ const res = await captureXPosts(inp, h.deps);
+ assert.deepEqual(h.visits, [
+ "https://x.com/i/status/11",
+ "https://x.com/i/article/999",
+ "https://x.com/i/status/22",
+ ]);
+ // Five contacts (two shots, two downloads, one article): four post gaps,
+ // and one short gap between the article's two images.
+ assert.deepEqual(h.pauses, [5_000, 5_000, 1_000, 5_000, 5_000]);
+ assert.deepEqual(
+ res.outcomes.map((o) => [o.id, o.state, o.files]),
+ [
+ ["11", "captured", 7],
+ ["22", "captured", 1],
+ ],
+ );
+ const rec = (await readPostCapture(path.join(inp.outDir, "11")))!;
+ assert.equal(rec.article?.state, "captured");
+ assert.equal(rec.article?.articleId, "999");
+ assert.equal(rec.article?.title, "A Worked Example & Its Notes");
+ assert.equal(rec.article?.blocks, 12);
+ assert.equal(rec.article?.files.length, 6);
+ assert.equal((await readPostCapture(path.join(inp.outDir, "22")))!.article, undefined);
+
+ // Captured is settled: a second run opens nothing.
+ const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] });
+ assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []);
+ assert.deepEqual(again.visits, []);
+
+ // force reads it again.
+ const forced = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] });
+ await captureXPosts({ ...inp, ids: ["11"], force: true }, forced.deps);
+ assert.deepEqual(forced.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]);
+});
+
+test("run: an article found only on the post's card is opened too; articles: false opens none", async () => {
+ const h = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] });
+ const inp = await input(["11"]);
+ await captureXPosts(inp, h.deps);
+ assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/demo_author/article/999"]);
+ assert.equal((await readPostCapture(path.join(inp.outDir, "11")))!.article?.state, "captured");
+
+ const off = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] });
+ const inp2 = await input(["11"], { archived: ARCHIVED, articles: false });
+ await captureXPosts(inp2, off.deps);
+ assert.deepEqual(off.visits, ["https://x.com/i/status/11"]);
+ assert.equal((await readPostCapture(path.join(inp2.outDir, "11")))!.article, undefined);
+});
+
+test("run: a post captured before articles owes only its article — no second shot or download", async () => {
+ const first = runHarness({ "11": post() }, { "999": [articleSnap()] });
+ const inp = await input(["11"], { articles: false });
+ await captureXPosts(inp, first.deps);
+
+ const h = runHarness({ "11": post() }, { "999": [articleSnap()] });
+ const res = await captureXPosts({ ...inp, articles: true, archived: ARCHIVED }, h.deps);
+ assert.deepEqual(h.visits, ["https://x.com/i/article/999"]);
+ assert.deepEqual(h.downloads, []);
+ assert.equal(res.outcomes[0].availability, undefined, "an article-only pass says nothing about the post's liveness");
+ const rec = (await readPostCapture(path.join(inp.outDir, "11")))!;
+ assert.equal(rec.state, "captured");
+ assert.ok(rec.shot, "the shot is kept");
+ assert.equal(rec.article?.state, "captured");
+});
+
+test("run: an article behind a login wall stops the run with needsCookies; the rest untouched", async () => {
+ const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("", "/i/flow/login")] });
+ const inp = await input(["11", "22"], { archived: ARCHIVED, media: false });
+ const res = await captureXPosts(inp, h.deps);
+ assert.equal(res.needsCookies, true);
+ assert.match(res.stoppedEarly ?? "", /Stopped at 11/);
+ assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]);
+ const rec = (await readPostCapture(path.join(inp.outDir, "11")))!;
+ assert.equal(rec.state, "captured", "the post itself was shot");
+ assert.equal(rec.article?.state, "login-wall");
+ assert.equal(await readPostCapture(path.join(inp.outDir, "22")), null);
+
+ // A login wall is owed again on the next run.
+ const next = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] });
+ await captureXPosts({ ...inp, ids: ["11"] }, next.deps);
+ assert.deepEqual(next.visits, ["https://x.com/i/article/999"]);
+});
+
+test("run: a deleted article is recorded and settled; the run goes on", async () => {
+ const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("Hmm...this page doesn’t exist.")] });
+ const inp = await input(["11", "22"], { archived: ARCHIVED, media: false });
+ const res = await captureXPosts(inp, h.deps);
+ assert.equal(res.stoppedEarly, undefined);
+ assert.equal(res.outcomes.length, 2);
+ const rec = (await readPostCapture(path.join(inp.outDir, "11")))!;
+ assert.equal(rec.article?.state, "deleted");
+ const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] });
+ assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []);
+ assert.deepEqual(again.visits, []);
+});
+
+test("run: a deleted post opens no article", async () => {
+ const h = runHarness(
+ { "11": { path: "/i/status/11", text: "This post was deleted by the post author.", article: null } },
+ { "999": [articleSnap()] },
+ );
+ const inp = await input(["11"], { archived: ARCHIVED });
+ await captureXPosts(inp, h.deps);
+ assert.deepEqual(h.visits, ["https://x.com/i/status/11"]);
+});
+
+test("run: an older record's article is not lost by a run that does not read it", async () => {
+ const inp = await input(["11"], { archived: ARCHIVED });
+ const dir = path.join(inp.outDir, "11");
+ const article = {
+ articleId: "999",
+ url: LINK.url,
+ capturedAt: "2026-01-01T00:00:00.000Z",
+ state: "captured" as const,
+ blocks: 3,
+ files: [],
+ };
+ await writePostCapture(dir, {
+ version: 1,
+ id: "11",
+ url: "https://x.com/i/status/11",
+ capturedAt: "2026-01-01T00:00:00.000Z",
+ state: "captured",
+ shot: { name: "shot.png", bytes: 1, sha256: "x" },
+ mediaState: "error",
+ media: [],
+ article,
+ });
+ const h = runHarness({ "11": post() }, {});
+ await captureXPosts(inp, h.deps);
+ assert.deepEqual(h.downloads, ["11"], "only the owed media");
+ assert.deepEqual(h.visits, []);
+ assert.deepEqual((await readPostCapture(dir))!.article, article);
+});
diff --git a/common/social/xArticleCapture.ts b/common/social/xArticleCapture.ts
@@ -22,7 +22,7 @@
// NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing,
// tested against recorded snapshots and a written HTML fixture, never x.com.
-import { readdir, rm, writeFile } from "node:fs/promises";
+import { mkdir, readdir, rm, writeFile } from "node:fs/promises";
import path from "node:path";
import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server";
import type { PageLike } from "./playwrightRuntime";
@@ -267,6 +267,7 @@ export async function captureXArticle(
}
// The images, each once, numbered in reading order.
+ await mkdir(dir, { recursive: true });
const images = await fetchArticleImages(page, content.blocks, dir, opts);
if (images.errors.length) {
const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`;