commit 12ca8cf0aa2b62b91d7525c1c8c5618cc11fe91d
parent 195dbff2fb7b2cc2a94e8f741b6de2775f7b24fa
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 20:09:08 -0400
common: capture-posts opens the X Article a post links to (articles, default on) and saves it beside the post
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
9 files changed, 736 insertions(+), 30 deletions(-)
diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts
@@ -14,7 +14,7 @@ process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
const { getPaths } = await import("../lib/paths");
const { registerSocialFetcher } = await import("../social/fetchers");
-const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server");
+const { readPostAvailability, writePostAvailability, writePosts } = await import("../lib/posts-server");
const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts");
type Input = import("../social/fetchers").PostCaptureInput;
type Result = import("../social/fetchers").PostCaptureResult;
@@ -36,6 +36,19 @@ registerSocialFetcher({
},
});
registerSocialFetcher({
+ id: "test-capture-x",
+ label: "Test capture (X)",
+ platform: "twitter",
+ fields: {},
+ detect: () => false,
+ probe: async () => ({ ok: true }),
+ fetch: async () => ({ posts: [], complete: true }),
+ captureByIds: async (input) => {
+ calls.push(input);
+ return answer;
+ },
+});
+registerSocialFetcher({
id: "test-no-capture",
label: "No-capture fetcher",
platform: "bluesky",
@@ -143,3 +156,64 @@ test("a run stopped by the source fails the job; one cancelled by the operator d
const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal });
assert.equal(cancelled.ok, true);
});
+
+test("the articles: on by default, refused alone only when not asked for by name, archived text passed (X)", async () => {
+ const root = path.join(channelsDir, "demo-x");
+ await mkdir(root, { recursive: true });
+ await writeFile(
+ path.join(root, "config.json"),
+ JSON.stringify({
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "twitter",
+ postFetcher: "test-capture-x",
+ socialHandle: "example_user",
+ name: "Example (X)",
+ url: "https://x.com/example_user",
+ }),
+ );
+ const post = (id: string, text: string, links: string[] = []) => ({
+ id,
+ slug: `demo-x/${id}`,
+ channelSlug: "demo-x",
+ author: "example_user",
+ createdAt: "2026-01-02T03:04:05.000Z",
+ uploadDate: "20260102",
+ text,
+ url: `https://x.com/example_user/status/${id}`,
+ platform: "twitter" as const,
+ isReply: false,
+ isRepost: false,
+ links,
+ });
+ await writePosts(root, [
+ post("111", "https://x.com/i/article/999"),
+ post("222", "a plain post", ["https://example.com/page"]),
+ post("333", "not asked for"),
+ ]);
+ answer = { outcomes: [] };
+
+ calls = [];
+ assert.equal((await run("demo-x", ["111", "222"])).ok, true);
+ assert.equal(calls[0].articles, undefined);
+ const archived = calls[0].archived ?? new Map();
+ assert.deepEqual([...archived.keys()].sort(), ["111", "222"]);
+ assert.deepEqual(archived.get("111"), { text: "https://x.com/i/article/999", links: [] });
+ assert.deepEqual(archived.get("222"), { text: "a plain post", links: ["https://example.com/page"] });
+
+ calls = [];
+ assert.equal((await run("demo-x", ["111"], { articles: false })).ok, true);
+ assert.equal(calls[0].articles, false);
+ assert.equal(calls[0].archived, undefined);
+
+ // Both halves off: refused, unless the articles are asked for by name.
+ calls = [];
+ assert.equal((await run("demo-x", ["111"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE);
+ assert.equal(
+ (await run("demo-x", ["111"], { shots: false, media: false, articles: false })).error,
+ NOTHING_TO_CAPTURE,
+ );
+ assert.equal(calls.length, 0);
+ assert.equal((await run("demo-x", ["111"], { shots: false, media: false, articles: true })).ok, true);
+ assert.equal(calls[0].articles, true);
+});
diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts
@@ -10,7 +10,9 @@
// nothing about the post).
//
// Only ids already in the channel's posts archive are captured: a capture is
-// of the archive, and an id from somewhere else is refused by name.
+// of the archive, and an id from somewhere else is refused by name. Each id's
+// archived text and links go to the fetcher with it, so a post that is an X
+// Article's link is known as one before its page is opened.
import path from "node:path";
import { readChannelConfig } from "./channels";
@@ -24,6 +26,7 @@ import {
} from "../lib/cookiePolicy";
import {
mergePostAvailability,
+ readAllPosts,
readPostAvailability,
readSeenPostIds,
writePostAvailability,
@@ -54,6 +57,8 @@ export type CapturePostsOptions = {
media?: boolean;
// Capture again what is already captured.
force?: boolean;
+ // Also open and save the X Article a post links to. Defaults to true.
+ articles?: boolean;
onLog?: (line: string) => void;
signal?: AbortSignal;
// The job's drain: stop between posts.
@@ -79,7 +84,17 @@ export function capturePostsProblem(
}
export const NOTHING_TO_CAPTURE =
- "Nothing to capture: both the screenshot and the media are turned off.";
+ 'Nothing to capture: both the screenshot and the media are turned off (to read only the articles, ask for "articles": true).';
+
+// Both halves off is nothing to do — unless the articles were asked for by
+// name: their default alone does not make a run.
+export function nothingToCapture(o: {
+ shots?: boolean;
+ media?: boolean;
+ articles?: boolean;
+}): boolean {
+ return o.shots === false && o.media === false && o.articles !== true;
+}
// The ids not in the channel's archive, or null when every one is.
export function strayCaptureIds(
@@ -103,7 +118,7 @@ export async function capturePosts(
const channelRoot = path.join(paths.channelsDir, slug);
const ids = [...new Set(opts.ids)];
- if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE);
+ if (nothingToCapture(opts)) return fail(NOTHING_TO_CAPTURE);
if (ids.length === 0) return fail("No post ids to capture.");
const config = await readChannelConfig(paths, slug);
@@ -121,6 +136,16 @@ export async function capturePosts(
const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot));
if (stray) return fail(strayIdsRefusal(slug, stray));
+ // The archived text of each id, for the article links in it (X only).
+ let archived: Map<string, { text: string; links: string[] }> | undefined;
+ if (opts.articles !== false && fetcher.platform === "twitter") {
+ const wanted = new Set(ids);
+ archived = new Map();
+ for (const p of await readAllPosts(channelRoot)) {
+ if (wanted.has(p.id)) archived.set(p.id, { text: p.text, links: p.links ?? [] });
+ }
+ }
+
const policy = resolveCookiePolicy(settings, config);
const xLogin =
fetcher.platform === "twitter"
@@ -132,6 +157,7 @@ export async function capturePosts(
[
opts.shots === false ? "" : " screenshot",
opts.media === false ? "" : " media",
+ opts.articles === false || fetcher.platform !== "twitter" ? "" : " articles",
].join("") +
(opts.force ? ", again where already captured" : "") +
".",
@@ -147,6 +173,8 @@ export async function capturePosts(
shots: opts.shots,
media: opts.media,
force: opts.force,
+ articles: opts.articles,
+ archived,
cookies: alwaysCookies(policy),
cookieSource: xLogin?.source,
browserCookies: xLogin?.browserSpec,
diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts
@@ -182,6 +182,13 @@ export type PostCaptureInput = Pick<
media?: boolean;
// Capture again what is already captured.
force?: boolean;
+ // Also capture the long-form article a post links to (X Articles), where it
+ // links to one. Defaults to true.
+ articles?: boolean;
+ // What the posts archive holds for each id — its text and expanded links —
+ // for the links a capture follows (an X Article's). Ids absent are read
+ // from the page alone.
+ archived?: ReadonlyMap<string, { text: string; links?: ReadonlyArray<string> }>;
// A soft stop: no new post is started once it fires, and the one in hand
// finishes (`signal` cancels outright).
drain?: AbortSignal;
diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts
@@ -48,6 +48,19 @@ export type PageLike = {
fullPage?: boolean;
clip?: { x: number; y: number; width: number; height: number };
}) => Promise<Uint8Array>;
+ // The page's own request context (the profile's cookies): an X Article's
+ // inline images are fetched through it (xArticleCapture.ts).
+ request?: {
+ get: (
+ url: string,
+ opts?: { timeout?: number; failOnStatusCode?: boolean },
+ ) => Promise<{
+ ok: () => boolean;
+ status: () => number;
+ headers: () => Record<string, string>;
+ body: () => Promise<Uint8Array>;
+ }>;
+ };
};
export type BrowserContextLike = {
diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts
@@ -11,10 +11,12 @@ import os from "node:os";
import path from "node:path";
import { fileURLToPath } from "node:url";
import {
+ articleImageFilename,
CAPTURE_FILENAME,
captureAvailability,
captureWork,
describeCapturedFile,
+ isArticleFile,
listCapturedMediaFiles,
POSTS_MEDIA_DIRNAME,
postCaptureDir,
@@ -22,6 +24,7 @@ import {
readPostCapture,
SHOT_FILENAME,
writePostCapture,
+ type ArticleCaptureRecord,
type PostCaptureRecord,
} from "./postCapture";
@@ -75,6 +78,12 @@ test("capture.json: each file's sha256 and byte size, the URLs, round-tripped",
assert.deepEqual(await readPostCapture(dir), rec);
// The record and the shot are not media.
assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]);
+
+ // Nor is the article half, whatever a later gallery-dl run finds beside it.
+ for (const name of ["article.json", "article.md", "article.png", "article.html", "article-img-1.jpg"]) {
+ await writeFile(path.join(dir, name), "x");
+ }
+ assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]);
});
test("capture.json: absent, unparseable or of another shape reads as no capture", async () => {
@@ -97,27 +106,65 @@ test("availability: only what the page said about the post — never a verdict f
test("what a capture owes: nothing for a settled post, only the missing half otherwise, everything when forced", () => {
const all = { shots: true, media: true, force: false };
- assert.deepEqual(captureWork(null, all), { shot: true, media: true });
- assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false });
+ assert.deepEqual(captureWork(null, all), { shot: true, media: true, article: false });
+ assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false, article: false });
const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" });
- assert.deepEqual(captureWork(done, all), { shot: false, media: false });
- assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false });
+ assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: false });
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false, article: false });
// The media failed (or was never asked for): only the media is owed.
- assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true });
- assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true });
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true, article: false });
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true, article: false });
// A shot that failed is owed again.
- assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false });
+ assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false, article: false });
// Deleted is settled: asking again is a request for the same answer.
assert.deepEqual(captureWork(record({ state: "deleted", mediaState: "skipped" }), all), {
shot: false,
media: false,
+ article: false,
});
// force re-takes whatever was asked for.
- assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true });
+ assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true, article: false });
assert.deepEqual(captureWork(record({ state: "deleted" }), { shots: false, media: true, force: true }), {
shot: false,
media: true,
+ article: false,
+ });
+});
+
+test("what a capture owes the article half: only when asked, settled once captured, deleted or walled", () => {
+ const all = { shots: true, media: true, force: false, articles: true };
+ const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" });
+ const article = (state: ArticleCaptureRecord["state"]): ArticleCaptureRecord => ({
+ articleId: "9",
+ url: "https://x.com/i/article/9",
+ capturedAt: "2026-01-02T03:04:05.000Z",
+ state,
+ blocks: 0,
+ files: [],
});
+ assert.equal(captureWork(null, all).article, true);
+ assert.equal(captureWork(null, { ...all, articles: false }).article, false);
+ // Not asked for by name: the older callers' shape owes no article.
+ assert.equal(captureWork(null, { shots: true, media: true, force: false }).article, false);
+ // A post shot before articles were captured owes only its article.
+ assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: true });
+ for (const state of ["captured", "deleted", "unavailable"] as const) {
+ assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, false, state);
+ }
+ for (const state of ["error", "login-wall"] as const) {
+ assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, true, state);
+ }
+ // force re-reads a captured article; a deleted post owes nothing.
+ assert.equal(captureWork(record({ ...done, article: article("captured") }), { ...all, force: true }).article, true);
+ assert.equal(captureWork(record({ state: "deleted" }), all).article, false);
+});
+
+test("article files are named by the layout, images numbered from 1", () => {
+ assert.equal(articleImageFilename(1, "jpg"), "article-img-1.jpg");
+ assert.ok(isArticleFile("article.md"));
+ assert.ok(isArticleFile("article-img-12.png"));
+ assert.ok(!isArticleFile("123_1.jpg"));
+ assert.ok(!isArticleFile("articles.txt"));
});
// THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON
diff --git a/common/social/postCapture.ts b/common/social/postCapture.ts
@@ -4,6 +4,10 @@
// channels/<slug>/posts-media/<post id>/shot.png — the post, as rendered
// channels/<slug>/posts-media/<post id>/<media…> — its attached media
// channels/<slug>/posts-media/<post id>/capture.json — this module's record
+// channels/<slug>/posts-media/<post id>/article.* — the X Article the post
+// links to, when it is one: article.json (its blocks), article.md,
+// article.png, article.html (the root as X served it) and
+// article-img-<n>.<ext> (its inline images) — xArticleCapture.ts
//
// A directory per post, unlike the posts themselves (month-sharded JSONL): a
// capture is a handful of files, and only for the posts someone asked for. It
@@ -26,6 +30,20 @@ import type { PostAvailability } from "../lib/posts";
export const POSTS_MEDIA_DIRNAME = "posts-media";
export const CAPTURE_FILENAME = "capture.json";
export const SHOT_FILENAME = "shot.png";
+export const ARTICLE_JSON_FILENAME = "article.json";
+export const ARTICLE_MD_FILENAME = "article.md";
+export const ARTICLE_SHOT_FILENAME = "article.png";
+export const ARTICLE_HTML_FILENAME = "article.html";
+
+// The n-th (1-based) inline image of an article.
+export function articleImageFilename(n: number, ext: string): string {
+ return `article-img-${n}.${ext}`;
+}
+
+// A file of the article half, never a medium of the post.
+export function isArticleFile(name: string): boolean {
+ return /^article(\.|-img-)/.test(name);
+}
export function postsMediaDir(channelRoot: string): string {
return path.join(channelRoot, POSTS_MEDIA_DIRNAME);
@@ -88,6 +106,28 @@ export type CapturedFile = {
// asked for or not attempted ("skipped"), or failed ("error").
export type CaptureMediaState = "ok" | "none" | "skipped" | "error";
+// The article half: an X Article the post links to, opened and read
+// (xArticleCapture.ts). `state` is the article page's, in the post's terms: a
+// deleted or walled article is settled, an error is owed again, a login wall
+// stopped the run.
+export type ArticleCaptureRecord = {
+ articleId: string;
+ url: string;
+ capturedAt: string;
+ state: PostCaptureState;
+ title?: string;
+ // The blocks read, for a captured article (article.json holds them).
+ blocks: number;
+ // How the body was read: by X's markers, or the root's text split at block
+ // elements because the markers were missing.
+ extraction?: "structured" | "fallback";
+ // article.json, article.md, article.png, article.html, each image.
+ files: CapturedFile[];
+ // article.png stops at a height cap; the article ran longer.
+ trimmed?: boolean;
+ error?: string;
+};
+
export type PostCaptureRecord = {
version: 1;
id: string;
@@ -102,6 +142,8 @@ export type PostCaptureRecord = {
shot?: CapturedFile;
mediaState: CaptureMediaState;
media: CapturedFile[];
+ // The X Article the post links to, when one was captured or tried.
+ article?: ArticleCaptureRecord;
error?: string;
};
@@ -128,8 +170,8 @@ export async function describeCapturedFile(
return { name, ...digest, ...(url ? { url } : {}) };
}
-// The media files in a post's directory: everything but the shot, the record
-// and a download's leftovers.
+// The media files in a post's directory: everything but the shot, the record,
+// the article half and a download's leftovers.
export async function listCapturedMediaFiles(dir: string): Promise<string[]> {
let names: string[];
try {
@@ -142,6 +184,7 @@ export async function listCapturedMediaFiles(dir: string): Promise<string[]> {
(n) =>
n !== SHOT_FILENAME &&
n !== CAPTURE_FILENAME &&
+ !isArticleFile(n) &&
!n.endsWith(".part") &&
!n.startsWith("."),
)
@@ -171,19 +214,31 @@ export async function writePostCapture(
// A deleted post is settled: the platform said so, and asking again costs a
// request for the same answer (`force` asks anyway). A shot on disk is kept; a
// media download that did not finish ("error", or never attempted) is owed.
+// The article half is owed only if the post links to one (the caller knows);
+// one captured, deleted or walled is settled, one that failed is owed.
export function captureWork(
existing: PostCaptureRecord | null,
- wanted: { shots: boolean; media: boolean; force: boolean },
-): { shot: boolean; media: boolean } {
+ wanted: { shots: boolean; media: boolean; force: boolean; articles?: boolean },
+): { shot: boolean; media: boolean; article: boolean } {
+ const articles = wanted.articles ?? false;
if (wanted.force || !existing) {
- return { shot: wanted.shots, media: wanted.media };
+ return { shot: wanted.shots, media: wanted.media, article: articles };
}
- if (existing.state === "deleted") return { shot: false, media: false };
+ if (existing.state === "deleted") return { shot: false, media: false, article: false };
return {
shot: wanted.shots && !existing.shot,
media:
wanted.media &&
existing.mediaState !== "ok" &&
existing.mediaState !== "none",
+ article: articles && !articleSettled(existing.article),
};
}
+
+export function articleSettled(article: ArticleCaptureRecord | undefined): boolean {
+ return (
+ article?.state === "captured" ||
+ article?.state === "deleted" ||
+ article?.state === "unavailable"
+ );
+}
diff --git a/common/social/xArticleCapture.ts b/common/social/xArticleCapture.ts
@@ -0,0 +1,391 @@
+// Capturing an X Article (a long-form post) beside the post that links to it:
+// the article page opened in the same logged-in profile and page as the post's
+// shot, read into blocks (xArticle.ts), and saved as
+// article.json — the blocks, title, byline, in reading order
+// article.md — the same as readable markdown
+// article.png — the article root, shot whole up to a height cap
+// article.html — the root as X served it, so a better reading later costs no
+// second visit
+// article-img-<n>.<ext> — its inline images, fetched through the page's own
+// request context (the profile's cookies), not a new tool
+//
+// PACED AS ONE CONTACT. The article load is a contact with X like the post's
+// shot: the run waits its 4–10 s gap before it (xPostCapture.ts). The images
+// are the page's own, already loaded once by the browser; they are fetched
+// again a short gap apart, not at the post gap.
+//
+// The page is read as a post page is (classifyXPostSnapshot): a login wall
+// stops the run (needsCookies), "Something went wrong" stops it, a deleted or
+// unavailable article is recorded and settled, anything else is an error a
+// later run tries again. Nothing is retried in the run that met it.
+//
+// NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing,
+// tested against recorded snapshots and a written HTML fixture, never x.com.
+
+import { readdir, rm, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server";
+import type { PageLike } from "./playwrightRuntime";
+import {
+ ARTICLE_HTML_FILENAME,
+ ARTICLE_JSON_FILENAME,
+ ARTICLE_MD_FILENAME,
+ ARTICLE_SHOT_FILENAME,
+ articleImageFilename,
+ describeCapturedFile,
+ type ArticleCaptureRecord,
+ type CapturedFile,
+} from "./postCapture";
+import {
+ extractXArticle,
+ findXArticleLink,
+ xArticleMarkdown,
+ type XArticleBlock,
+ type XArticleLink,
+} from "./xArticle";
+import { classifyXPostSnapshot, type XPostVerdict } from "./xPostCapture";
+
+// The article root, most specific first. The read view holds the title, the
+// byline and the body; the rich-text view only the body, so it is widened to
+// the article around it.
+export const ARTICLE_ROOT_MARKERS = [
+ '[data-testid="twitterArticleReadView"]',
+ '[data-testid="twitterArticleRichTextView"]',
+ '[data-testid="longformRichTextComponent"]',
+];
+const ARTICLE_FALLBACK_ROOT = '[data-testid="primaryColumn"] article, [data-testid="primaryColumn"] [role="article"]';
+
+// article.png stops here. Chromium's full-page capture is a single texture,
+// and past its 16384 px limit a taller shot repeats or blanks, so the cap sits
+// under it; a longer article is recorded `trimmed` (article.md has it all).
+export const ARTICLE_SHOT_MAX_HEIGHT = 16_000;
+
+export type XArticleSnapshot = {
+ path: string;
+ text: string;
+ root: null | {
+ rect: { x: number; y: number; width: number; height: number };
+ // Which marker found the root ("fallback": none did).
+ marker: string;
+ };
+};
+
+const ARTICLE_SNAPSHOT_SCRIPT = `(() => {
+ const markers = ${JSON.stringify(ARTICLE_ROOT_MARKERS)};
+ let root = null;
+ let marker = null;
+ for (const m of markers) {
+ const el = document.querySelector(m);
+ if (el) { root = el; marker = m; break; }
+ }
+ if (root && marker !== markers[0]) {
+ root = root.closest('article, [role="article"]') || root;
+ }
+ if (!root) {
+ root = document.querySelector(${JSON.stringify(ARTICLE_FALLBACK_ROOT)});
+ if (root) marker = "fallback";
+ }
+ const main = document.querySelector('[data-testid="primaryColumn"]') || document.body;
+ const text = ((main && main.innerText) || "").slice(0, 4000);
+ let out = null;
+ if (root) {
+ for (const old of document.querySelectorAll('[data-archilyzer-article]')) {
+ old.removeAttribute('data-archilyzer-article');
+ }
+ root.setAttribute('data-archilyzer-article', '');
+ const r = root.getBoundingClientRect();
+ out = {
+ rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height },
+ marker,
+ };
+ }
+ return { path: location.pathname, text, root: out };
+})()`;
+
+// Walk down the page so lazy images load, then back to the top for the shot.
+const LOAD_LAZY_SCRIPT = `(async () => {
+ const root = document.querySelector('[data-archilyzer-article]') || document.body;
+ for (const img of root.querySelectorAll('img[loading="lazy"]')) img.loading = "eager";
+ const wait = (ms) => new Promise((r) => setTimeout(r, ms));
+ let steps = 0;
+ for (let y = 0; y < document.documentElement.scrollHeight && steps < 200; steps++) {
+ y += Math.max(400, Math.floor(window.innerHeight * 0.8));
+ window.scrollTo(0, y);
+ await wait(250);
+ }
+ window.scrollTo(0, 0);
+ await wait(500);
+ const pending = Array.from(root.querySelectorAll("img")).filter((i) => !i.complete);
+ await Promise.all(pending.map((i) => new Promise((r) => {
+ i.addEventListener("load", r, { once: true });
+ i.addEventListener("error", r, { once: true });
+ setTimeout(r, 5000);
+ })));
+ return steps;
+})()`;
+
+const ARTICLE_HTML_SCRIPT = `(() => {
+ const root = document.querySelector('[data-archilyzer-article]');
+ return root ? root.outerHTML : "";
+})()`;
+
+// What an article page means, in a post's terms. A page with a root is
+// captured; one without is read for X's markers as a post page is.
+export function classifyXArticleSnapshot(s: XArticleSnapshot): XPostVerdict {
+ return classifyXPostSnapshot(
+ {
+ path: s.path,
+ text: s.text,
+ article: s.root ? { rect: s.root.rect, sensitive: false } : null,
+ },
+ "article",
+ );
+}
+
+// The article a post's rendered card links to: the hrefs read from it.
+export function xArticleLinkFromCard(hrefs: ReadonlyArray<string> | undefined): XArticleLink | null {
+ return hrefs?.length ? findXArticleLink(hrefs) : null;
+}
+
+// The full-size picture of an X media URL (`name=orig`); anything else as is.
+export function fullSizeImageUrl(src: string): string {
+ try {
+ const u = new URL(src);
+ if (u.hostname === "pbs.twimg.com" && u.pathname.startsWith("/media/")) {
+ u.searchParams.set("name", "orig");
+ return u.toString();
+ }
+ } catch {
+ /* not a URL: as is */
+ }
+ return src;
+}
+
+const EXT_BY_TYPE: Record<string, string> = {
+ "image/jpeg": "jpg",
+ "image/png": "png",
+ "image/gif": "gif",
+ "image/webp": "webp",
+ "image/avif": "avif",
+};
+
+// An image's extension: X's `format=` parameter, the path's own, the response's
+// type, in that order.
+export function imageExtension(url: string, contentType?: string): string {
+ try {
+ const u = new URL(url);
+ const format = u.searchParams.get("format");
+ if (format && /^[a-z0-9]{2,5}$/i.test(format)) return format.toLowerCase();
+ const m = /\.([a-z0-9]{2,5})$/i.exec(u.pathname);
+ if (m) return m[1].toLowerCase() === "jpeg" ? "jpg" : m[1].toLowerCase();
+ } catch {
+ /* fall through to the type */
+ }
+ const type = (contentType ?? "").split(";")[0].trim().toLowerCase();
+ return EXT_BY_TYPE[type] ?? "img";
+}
+
+export type XArticleCaptureResult = {
+ record: ArticleCaptureRecord;
+ // Set when the run must stop here (a login wall, X refusing pages).
+ stop?: string;
+};
+
+export type XArticleCaptureOptions = {
+ now: () => Date;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // The gap between one image fetch and the next.
+ imageGap?: () => Promise<void>;
+};
+
+// Open the article, read it, shoot it, fetch its images, write its files into
+// `dir`. The record says how it went; nothing is retried here.
+export async function captureXArticle(
+ page: PageLike,
+ link: XArticleLink,
+ dir: string,
+ opts: XArticleCaptureOptions,
+): Promise<XArticleCaptureResult> {
+ const { onLog } = opts;
+ const base = { articleId: link.articleId, url: link.url };
+ const failed = (verdict: XPostVerdict): XArticleCaptureResult => ({
+ record: {
+ ...base,
+ capturedAt: opts.now().toISOString(),
+ state: verdict.state,
+ blocks: 0,
+ files: [],
+ ...(verdict.error ? { error: verdict.error } : {}),
+ },
+ ...(verdict.stop ? { stop: verdict.stop } : {}),
+ });
+
+ try {
+ await page.goto(link.url, { waitUntil: "domcontentloaded", timeout: 60_000 });
+ } catch (err) {
+ return failed({ state: "error", error: `Could not load ${link.url}: ${firstLine(err)}` });
+ }
+ await page
+ .waitForSelector([...ARTICLE_ROOT_MARKERS, ARTICLE_FALLBACK_ROOT].join(", "), { timeout: 20_000 })
+ .catch(() => {});
+ await page.waitForTimeout(1_500);
+ let snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot;
+ const verdict = classifyXArticleSnapshot(snap);
+ if (verdict.state !== "captured") return failed(verdict);
+ if (snap.root?.marker === "fallback") {
+ onLog?.(`article ${link.articleId}: no article marker on the page — reading the page's post as the article.`);
+ }
+
+ await page.evaluate(LOAD_LAZY_SCRIPT).catch(() => {});
+ snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot;
+ const html = String((await page.evaluate(ARTICLE_HTML_SCRIPT)) ?? "");
+ const rect = snap.root?.rect;
+ if (!html || !rect || rect.width < 1 || rect.height < 1) {
+ return failed({ state: "error", error: "The article rendered with nothing to read." });
+ }
+ const content = extractXArticle(html);
+ const capturedAt = opts.now().toISOString();
+
+ // The shot: the root, whole, up to the cap.
+ const trimmed = rect.height > ARTICLE_SHOT_MAX_HEIGHT;
+ let png: Uint8Array | undefined;
+ let error: string | undefined;
+ try {
+ png = await page.screenshot({
+ type: "png",
+ fullPage: true,
+ clip: {
+ x: Math.max(0, Math.floor(rect.x)),
+ y: Math.max(0, Math.floor(rect.y)),
+ width: Math.ceil(rect.width),
+ height: Math.min(Math.ceil(rect.height), ARTICLE_SHOT_MAX_HEIGHT),
+ },
+ });
+ } catch (err) {
+ error = `The article's screenshot failed: ${firstLine(err)}`;
+ }
+
+ // The images, each once, numbered in reading order.
+ const images = await fetchArticleImages(page, content.blocks, dir, opts);
+ if (images.errors.length) {
+ const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`;
+ error = error ? `${error}; ${why}` : why;
+ }
+
+ const blocks: XArticleBlock[] = content.blocks.map((b) =>
+ b.type === "image" && b.src && images.saved.has(b.src)
+ ? { ...b, file: images.saved.get(b.src)!.name }
+ : b,
+ );
+ const article = {
+ version: 1,
+ ...base,
+ ...(content.title ? { title: content.title } : {}),
+ ...(content.author ? { author: content.author } : {}),
+ ...(content.handle ? { handle: content.handle } : {}),
+ ...(content.publishedAt ? { publishedAt: content.publishedAt } : {}),
+ capturedAt,
+ extraction: content.extraction,
+ ...(trimmed ? { trimmed: true } : {}),
+ blocks,
+ };
+ await writeJsonAtomic(path.join(dir, ARTICLE_JSON_FILENAME), article, { mkdir: true });
+ await writeFileAtomic(
+ path.join(dir, ARTICLE_MD_FILENAME),
+ xArticleMarkdown({ ...content, blocks, url: link.url }),
+ );
+ await writeFileAtomic(path.join(dir, ARTICLE_HTML_FILENAME), html);
+ if (png) await writeFile(path.join(dir, ARTICLE_SHOT_FILENAME), png);
+ await removeStaleImages(dir, new Set([...images.saved.values()].map((f) => f.name)));
+
+ const files: CapturedFile[] = [
+ await describeCapturedFile(dir, ARTICLE_JSON_FILENAME, link.url),
+ await describeCapturedFile(dir, ARTICLE_MD_FILENAME, link.url),
+ await describeCapturedFile(dir, ARTICLE_HTML_FILENAME, link.url),
+ ...(png ? [await describeCapturedFile(dir, ARTICLE_SHOT_FILENAME, link.url)] : []),
+ ...images.saved.values(),
+ ];
+ return {
+ record: {
+ ...base,
+ capturedAt,
+ state: "captured",
+ ...(content.title ? { title: content.title } : {}),
+ blocks: blocks.length,
+ extraction: content.extraction,
+ files,
+ ...(trimmed ? { trimmed: true } : {}),
+ ...(error ? { error } : {}),
+ },
+ };
+}
+
+// Each distinct image src, fetched through the page's request context into
+// `article-img-<n>.<ext>`: the full-size picture first, the src as rendered if
+// that is refused.
+async function fetchArticleImages(
+ page: PageLike,
+ blocks: ReadonlyArray<XArticleBlock>,
+ dir: string,
+ opts: XArticleCaptureOptions,
+): Promise<{ saved: Map<string, CapturedFile>; errors: string[] }> {
+ const saved = new Map<string, CapturedFile>();
+ const errors: string[] = [];
+ const srcs = [...new Set(blocks.flatMap((b) => (b.type === "image" && b.src ? [b.src] : [])))];
+ if (srcs.length === 0) return { saved, errors };
+ if (!page.request) {
+ errors.push("this browser page cannot fetch (no request context)");
+ return { saved, errors };
+ }
+ let n = 0;
+ for (const src of srcs) {
+ if (opts.signal?.aborted) {
+ errors.push("cancelled before the rest of the images");
+ break;
+ }
+ if (n > 0) await opts.imageGap?.();
+ n++;
+ const tries = [...new Set([fullSizeImageUrl(src), src])];
+ let last = "";
+ for (const url of tries) {
+ try {
+ const res = await page.request.get(url, { timeout: 30_000, failOnStatusCode: false });
+ if (!res.ok()) {
+ last = `HTTP ${res.status()} for ${url}`;
+ continue;
+ }
+ const body = await res.body();
+ const name = articleImageFilename(n, imageExtension(url, res.headers()["content-type"]));
+ await writeFile(path.join(dir, name), body);
+ saved.set(src, await describeCapturedFile(dir, name, url));
+ last = "";
+ break;
+ } catch (err) {
+ last = `${firstLine(err)} (${url})`;
+ }
+ }
+ if (last) errors.push(last);
+ }
+ if (saved.size) opts.onLog?.(`article: ${saved.size} image(s) saved.`);
+ return { saved, errors };
+}
+
+// A re-capture with fewer images leaves no older numbered file behind.
+async function removeStaleImages(dir: string, keep: ReadonlySet<string>): Promise<void> {
+ let names: string[];
+ try {
+ names = await readdir(dir);
+ } catch {
+ return;
+ }
+ for (const name of names) {
+ if (/^article-img-\d+\./.test(name) && !keep.has(name)) {
+ await rm(path.join(dir, name), { force: true });
+ }
+ }
+}
+
+function firstLine(err: unknown): string {
+ return ((err as Error)?.message ?? String(err)).split("\n")[0];
+}
diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts
@@ -41,6 +41,7 @@ import {
} from "./xSessionBroker";
import type { XCookieSource } from "./xCookieSource";
import { listCapturedMediaFiles } from "./postCapture";
+import { xArticleLinkFromArchive } from "./xArticle";
import {
captureXPosts,
xStatusUrl,
@@ -328,11 +329,17 @@ export async function captureXPostsByIds(
input: PostCaptureInput,
): Promise<PostCaptureResult> {
const paths = getPaths();
- if (input.shots ?? true) {
+ // The profile shoots the posts and opens the articles. A post is known to
+ // link to an article before any page only by its archived text; one found
+ // on a card is found by a shot, which already needs the profile.
+ const articles =
+ (input.articles ?? true) &&
+ input.ids.some((id) => xArticleLinkFromArchive(input.archived?.get(id)) !== null);
+ if ((input.shots ?? true) || articles) {
const status = await readXSessionStatus(paths);
if (!status.hasProfile) {
const why =
- "No X session profile to shoot posts with: connect an X account on /settings, then run it again.";
+ "No X session profile to shoot posts or open articles with: connect an X account on /settings, then run it again.";
input.onLog?.(`[auth] ${why}`);
return { outcomes: [], needsCookies: true, stoppedEarly: why };
}
@@ -355,9 +362,14 @@ export async function captureXPostsByIds(
},
pauseMs: () => xRequestPauseMs(),
pause,
+ articleImagePauseMs: () => ARTICLE_IMAGE_PAUSE_MS + Math.floor(Math.random() * ARTICLE_IMAGE_PAUSE_MS),
});
}
+// The gap between an article's images: 1–2 s. They are the page's own
+// pictures, already loaded once by the browser, so not the full post gap.
+const ARTICLE_IMAGE_PAUSE_MS = 1_000;
+
// ---------------------------------------------------------------------------
// Search: the older-posts backfill
// ---------------------------------------------------------------------------
diff --git a/common/social/xPostCapture.ts b/common/social/xPostCapture.ts
@@ -21,6 +21,10 @@
// - a protected, suspended or vanished account, or a withheld post:
// unavailable.
//
+// A post that is an X Article (long-form) links to it — in its archived text,
+// or on its rendered card — and the run then opens the article too, one more
+// paced contact (xArticleCapture.ts), unless `articles` is off.
+//
// NOT VERIFIED AGAINST LIVE X. The page markers below are X's as of this
// writing and are tested against recorded snapshots, never x.com; the first
// real run is the check.
@@ -41,11 +45,14 @@ import {
readPostCapture,
SHOT_FILENAME,
writePostCapture,
+ type ArticleCaptureRecord,
type CapturedFile,
type CaptureMediaState,
type PostCaptureRecord,
type PostCaptureState,
} from "./postCapture";
+import { xArticleLinkFromArchive, type XArticleLink } from "./xArticle";
+import { captureXArticle, xArticleLinkFromCard } from "./xArticleCapture";
export function xStatusUrl(id: string): string {
return `https://x.com/i/status/${id}`;
@@ -60,6 +67,8 @@ export type XPostSnapshot = {
rect: { x: number; y: number; width: number; height: number };
// A "Show" / "View" button inside the post: a sensitive-media cover.
sensitive: boolean;
+ // The hrefs inside the post that look like an X Article's: its card.
+ articleHrefs?: string[];
};
};
@@ -83,9 +92,14 @@ const SNAPSHOT_SCRIPT = (id: string) => `(() => {
const sensitive = Array.from(own.querySelectorAll('button, [role="button"]')).some(
(b) => /^(show|view)$/i.test((b.innerText || "").trim()),
);
+ const articleHrefs = Array.from(own.querySelectorAll('a[href*="/article/"]'))
+ .map((l) => l.getAttribute("href") || "")
+ .filter(Boolean)
+ .slice(0, 10);
article = {
rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height },
sensitive,
+ articleHrefs,
};
}
return { path: location.pathname, text, article };
@@ -140,7 +154,12 @@ export type XPostVerdict = {
};
// What a snapshot means. Pure, so every marker is testable without a browser.
-export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict {
+// `noun` names what the page should have shown (an article page is read the
+// same way, xArticleCapture.ts).
+export function classifyXPostSnapshot(
+ s: Pick<XPostSnapshot, "path" | "text" | "article">,
+ noun = "post",
+): XPostVerdict {
if (/^\/(i\/flow\/login|login|i\/flow\/signup)\b/.test(s.path)) {
return {
state: "login-wall",
@@ -153,7 +172,7 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict {
if (REFUSED_TEXT.test(s.text)) {
return {
state: "error",
- error: "X answered “Something went wrong” instead of the post.",
+ error: `X answered “Something went wrong” instead of the ${noun}.`,
stop: "X is refusing pages right now; stopping rather than asking again.",
};
}
@@ -162,19 +181,23 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict {
if (AGE_WALL_TEXT.test(s.text)) {
return {
state: "error",
- error: "X shows this post only to an age-verified session.",
+ error: `X shows this ${noun} only to an age-verified session.`,
};
}
if (LOGGED_OUT_TEXT.test(s.text)) {
return {
state: "login-wall",
- stop: "X showed its logged-out page instead of the post.",
+ stop: `X showed its logged-out page instead of the ${noun}.`,
};
}
- return { state: "error", error: "The post did not render." };
+ return { state: "error", error: `The ${noun} did not render.` };
}
-export type XShotResult = XPostVerdict & { shot?: CapturedFile };
+export type XShotResult = XPostVerdict & {
+ shot?: CapturedFile;
+ // The X Article the post's card links to, when it does.
+ articleLink?: XArticleLink;
+};
// One post's screenshot: load, read the page, open a sensitive cover, shoot
// the post's own article. Writes `shot.png` into `dir` only for a post that
@@ -234,7 +257,8 @@ export async function shootXPost(
await mkdir(dir, { recursive: true });
await writeFile(path.join(dir, SHOT_FILENAME), png);
const shot = await describeCapturedFile(dir, SHOT_FILENAME, url);
- return { ...verdict, shot };
+ const articleLink = xArticleLinkFromCard(snap.article?.articleHrefs);
+ return { ...verdict, shot, ...(articleLink ? { articleLink } : {}) };
}
export type MediaDownloadResult =
@@ -255,6 +279,8 @@ export type XCaptureDeps = {
// The gap before each contact with X after the first.
pauseMs: () => number;
pause: (ms: number, signal: AbortSignal) => Promise<void>;
+ // The gap between one article image and the next (none when absent).
+ articleImagePauseMs?: () => number;
now?: () => Date;
};
@@ -274,6 +300,7 @@ export async function captureXPosts(
shots: input.shots ?? true,
media: input.media ?? true,
force: input.force ?? false,
+ articles: input.articles ?? true,
};
const now = deps.now ?? (() => new Date());
const outcomes: PostCaptureOutcome[] = [];
@@ -296,7 +323,15 @@ export async function captureXPosts(
const dir = postCaptureDir(input.outDir, id);
const existing = await readPostCapture(dir);
const work = captureWork(existing, wanted);
- if (!work.shot && !work.media) {
+ // The article the post links to, as far as is known before any page:
+ // the archived text, or the last capture's record.
+ let articleLink: XArticleLink | null = wanted.articles
+ ? (xArticleLinkFromArchive(input.archived?.get(id)) ??
+ (existing?.article ? { articleId: existing.article.articleId, url: existing.article.url } : null))
+ : null;
+ const articleOwedNow =
+ work.article && !!articleLink && (!existing || work.shot || existing.state === "captured");
+ if (!work.shot && !work.media && !articleOwedNow) {
onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`);
continue;
}
@@ -318,6 +353,7 @@ export async function captureXPosts(
shot = res.shot;
error = res.error;
stop = res.stop;
+ if (wanted.articles && !articleLink && res.articleLink) articleLink = res.articleLink;
}
let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped");
@@ -347,6 +383,34 @@ export async function captureXPosts(
}
}
+ // The article: after the post and its media, one more paced load in the
+ // same page. Only for a post that is there, in a run not already
+ // stopping.
+ let article: ArticleCaptureRecord | undefined = existing?.article;
+ let articleFailed = false;
+ if (work.article && articleLink && (state === undefined || state === "captured") && !stop) {
+ await contact();
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ browser ??= await deps.openPage();
+ const got = await captureXArticle(browser.page, articleLink, dir, {
+ now,
+ onLog,
+ signal,
+ imageGap: deps.articleImagePauseMs
+ ? () => deps.pause(deps.articleImagePauseMs!(), signal)
+ : undefined,
+ });
+ article = got.record;
+ articleFailed = article.state === "error";
+ // A run that only read the article learns of the post only that it
+ // linked to a readable article.
+ state ??= article.state === "captured" ? "captured" : "error";
+ if (got.stop) {
+ stop = got.stop;
+ if (article.state === "login-wall") needsCookies = true;
+ }
+ }
+
const finalState: PostCaptureState = state ?? "error";
const record: PostCaptureRecord = {
version: 1,
@@ -358,6 +422,7 @@ export async function captureXPosts(
...(shot ? { shot } : {}),
mediaState,
media,
+ ...(article ? { article } : {}),
...(error ? { error } : {}),
};
await writePostCapture(dir, record);
@@ -369,13 +434,14 @@ export async function captureXPosts(
...(work.shot && captureAvailability(finalState)
? { availability: captureAvailability(finalState) }
: {}),
- files: (shot ? 1 : 0) + media.length,
+ files: (shot ? 1 : 0) + media.length + (article?.files.length ?? 0),
...(error ? { error } : {}),
});
onLog?.(
`${id}: ${finalState}${sensitive ? " (behind a sensitive-media cover)" : ""}` +
(shot ? ", shot" : "") +
(mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") +
+ (article && article !== existing?.article ? `, ${describeArticle(article)}` : "") +
(error ? ` — ${error}` : ""),
);
@@ -384,7 +450,7 @@ export async function captureXPosts(
needsCookies: needsCookies || finalState === "login-wall",
});
}
- errorsInARow = finalState === "error" ? errorsInARow + 1 : 0;
+ errorsInARow = finalState === "error" || articleFailed ? errorsInARow + 1 : 0;
if (errorsInARow >= STOP_AFTER_ERRORS) {
return stopped(
`${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`,
@@ -397,6 +463,19 @@ export async function captureXPosts(
}
}
+function describeArticle(a: ArticleCaptureRecord): string {
+ if (a.state !== "captured") {
+ return `article ${a.state}` + (a.error ? ` (${a.error})` : "");
+ }
+ return (
+ `article${a.title ? ` “${a.title}”` : ""} (${a.blocks} block(s), ${a.files.length} file(s)` +
+ (a.extraction === "fallback" ? ", read by the fallback" : "") +
+ (a.trimmed ? ", shot trimmed" : "") +
+ ")" +
+ (a.error ? ` — ${a.error}` : "")
+ );
+}
+
function firstLine(err: unknown): string {
return ((err as Error)?.message ?? String(err)).split("\n")[0];
}