// Capturing an X Article (a long-form post) beside the post that links to it: // the article page opened in the same logged-in profile and page as the post's // shot, read into blocks (xArticle.ts), and saved as // article.json — the blocks, title, byline, in reading order // article.md — the same as readable markdown // article.png — the article root, shot whole up to a height cap // article.html — the root as X served it, so a better reading later costs no // second visit // article-img-. — its inline images, fetched through the page's own // request context (the profile's cookies), not a new tool // // PACED AS ONE CONTACT. The article load is a contact with X like the post's // shot: the run waits its 4–10 s gap before it (xPostCapture.ts). The images // are the page's own, already loaded once by the browser; they are fetched // again a short gap apart, not at the post gap. // // The page is read as a post page is (classifyXPostSnapshot): a login wall // stops the run (needsCookies), "Something went wrong" stops it, a deleted or // unavailable article is recorded and settled, anything else is an error a // later run tries again. Nothing is retried in the run that met it. // // NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing, // tested against recorded snapshots and a written HTML fixture, never x.com. import { mkdir, readdir, rm, writeFile } from "node:fs/promises"; import path from "node:path"; import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server"; import type { PageLike } from "./playwrightRuntime"; import { ARTICLE_HTML_FILENAME, ARTICLE_JSON_FILENAME, ARTICLE_MD_FILENAME, ARTICLE_SHOT_FILENAME, articleImageFilename, describeCapturedFile, type ArticleCaptureRecord, type CapturedFile, } from "./postCapture"; import { extractXArticle, findXArticleLink, xArticleMarkdown, type XArticleBlock, type XArticleLink, } from "./xArticle"; import { classifyXPostSnapshot, type XPostVerdict } from "./xPostCapture"; // The article root, most specific first. The read view holds the title, the // byline and the body; the rich-text view only the body, so it is widened to // the article around it. export const ARTICLE_ROOT_MARKERS = [ '[data-testid="twitterArticleReadView"]', '[data-testid="twitterArticleRichTextView"]', '[data-testid="longformRichTextComponent"]', ]; const ARTICLE_FALLBACK_ROOT = '[data-testid="primaryColumn"] article, [data-testid="primaryColumn"] [role="article"]'; // article.png stops here. Chromium's full-page capture is a single texture, // and past its 16384 px limit a taller shot repeats or blanks, so the cap sits // under it; a longer article is recorded `trimmed` (article.md has it all). export const ARTICLE_SHOT_MAX_HEIGHT = 16_000; export type XArticleSnapshot = { path: string; text: string; root: null | { rect: { x: number; y: number; width: number; height: number }; // Which marker found the root ("fallback": none did). marker: string; }; }; const ARTICLE_SNAPSHOT_SCRIPT = `(() => { const markers = ${JSON.stringify(ARTICLE_ROOT_MARKERS)}; let root = null; let marker = null; for (const m of markers) { const el = document.querySelector(m); if (el) { root = el; marker = m; break; } } if (root && marker !== markers[0]) { root = root.closest('article, [role="article"]') || root; } if (!root) { root = document.querySelector(${JSON.stringify(ARTICLE_FALLBACK_ROOT)}); if (root) marker = "fallback"; } const main = document.querySelector('[data-testid="primaryColumn"]') || document.body; const text = ((main && main.innerText) || "").slice(0, 4000); let out = null; if (root) { for (const old of document.querySelectorAll('[data-archilyzer-article]')) { old.removeAttribute('data-archilyzer-article'); } root.setAttribute('data-archilyzer-article', ''); const r = root.getBoundingClientRect(); out = { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, marker, }; } return { path: location.pathname, text, root: out }; })()`; // Walk down the page so lazy images load, then back to the top for the shot. const LOAD_LAZY_SCRIPT = `(async () => { const root = document.querySelector('[data-archilyzer-article]') || document.body; for (const img of root.querySelectorAll('img[loading="lazy"]')) img.loading = "eager"; const wait = (ms) => new Promise((r) => setTimeout(r, ms)); let steps = 0; for (let y = 0; y < document.documentElement.scrollHeight && steps < 200; steps++) { y += Math.max(400, Math.floor(window.innerHeight * 0.8)); window.scrollTo(0, y); await wait(250); } window.scrollTo(0, 0); await wait(500); const pending = Array.from(root.querySelectorAll("img")).filter((i) => !i.complete); await Promise.all(pending.map((i) => new Promise((r) => { i.addEventListener("load", r, { once: true }); i.addEventListener("error", r, { once: true }); setTimeout(r, 5000); }))); return steps; })()`; const ARTICLE_HTML_SCRIPT = `(() => { const root = document.querySelector('[data-archilyzer-article]'); return root ? root.outerHTML : ""; })()`; // What an article page means, in a post's terms. A page with a root is // captured; one without is read for X's markers as a post page is. export function classifyXArticleSnapshot(s: XArticleSnapshot): XPostVerdict { return classifyXPostSnapshot( { path: s.path, text: s.text, article: s.root ? { rect: s.root.rect, sensitive: false } : null, }, "article", ); } // The article a post's rendered card links to: the hrefs read from it. export function xArticleLinkFromCard(hrefs: ReadonlyArray | undefined): XArticleLink | null { return hrefs?.length ? findXArticleLink(hrefs) : null; } // The full-size picture of an X media URL (`name=orig`); anything else as is. export function fullSizeImageUrl(src: string): string { try { const u = new URL(src); if (u.hostname === "pbs.twimg.com" && u.pathname.startsWith("/media/")) { u.searchParams.set("name", "orig"); return u.toString(); } } catch { /* not a URL: as is */ } return src; } const EXT_BY_TYPE: Record = { "image/jpeg": "jpg", "image/png": "png", "image/gif": "gif", "image/webp": "webp", "image/avif": "avif", }; // An image's extension: X's `format=` parameter, the path's own, the response's // type, in that order. export function imageExtension(url: string, contentType?: string): string { try { const u = new URL(url); const format = u.searchParams.get("format"); if (format && /^[a-z0-9]{2,5}$/i.test(format)) return format.toLowerCase(); const m = /\.([a-z0-9]{2,5})$/i.exec(u.pathname); if (m) return m[1].toLowerCase() === "jpeg" ? "jpg" : m[1].toLowerCase(); } catch { /* fall through to the type */ } const type = (contentType ?? "").split(";")[0].trim().toLowerCase(); return EXT_BY_TYPE[type] ?? "img"; } export type XArticleCaptureResult = { record: ArticleCaptureRecord; // Set when the run must stop here (a login wall, X refusing pages). stop?: string; }; export type XArticleCaptureOptions = { now: () => Date; onLog?: (line: string) => void; signal?: AbortSignal; // The gap between one image fetch and the next. imageGap?: () => Promise; }; // Open the article, read it, shoot it, fetch its images, write its files into // `dir`. The record says how it went; nothing is retried here. export async function captureXArticle( page: PageLike, link: XArticleLink, dir: string, opts: XArticleCaptureOptions, ): Promise { const { onLog } = opts; const base = { articleId: link.articleId, url: link.url }; const failed = (verdict: XPostVerdict): XArticleCaptureResult => ({ record: { ...base, capturedAt: opts.now().toISOString(), state: verdict.state, blocks: 0, files: [], ...(verdict.error ? { error: verdict.error } : {}), }, ...(verdict.stop ? { stop: verdict.stop } : {}), }); try { await page.goto(link.url, { waitUntil: "domcontentloaded", timeout: 60_000 }); } catch (err) { return failed({ state: "error", error: `Could not load ${link.url}: ${firstLine(err)}` }); } await page .waitForSelector([...ARTICLE_ROOT_MARKERS, ARTICLE_FALLBACK_ROOT].join(", "), { timeout: 20_000 }) .catch(() => {}); await page.waitForTimeout(1_500); let snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; const verdict = classifyXArticleSnapshot(snap); if (verdict.state !== "captured") return failed(verdict); if (snap.root?.marker === "fallback") { onLog?.(`article ${link.articleId}: no article marker on the page — reading the page's post as the article.`); } await page.evaluate(LOAD_LAZY_SCRIPT).catch(() => {}); snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; const html = String((await page.evaluate(ARTICLE_HTML_SCRIPT)) ?? ""); const rect = snap.root?.rect; if (!html || !rect || rect.width < 1 || rect.height < 1) { return failed({ state: "error", error: "The article rendered with nothing to read." }); } const content = extractXArticle(html); const capturedAt = opts.now().toISOString(); // The shot: the root, whole, up to the cap. const trimmed = rect.height > ARTICLE_SHOT_MAX_HEIGHT; let png: Uint8Array | undefined; let error: string | undefined; try { png = await page.screenshot({ type: "png", fullPage: true, clip: { x: Math.max(0, Math.floor(rect.x)), y: Math.max(0, Math.floor(rect.y)), width: Math.ceil(rect.width), height: Math.min(Math.ceil(rect.height), ARTICLE_SHOT_MAX_HEIGHT), }, }); } catch (err) { error = `The article's screenshot failed: ${firstLine(err)}`; } // The images, each once, numbered in reading order. await mkdir(dir, { recursive: true }); const images = await fetchArticleImages(page, content.blocks, dir, opts); if (images.errors.length) { const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`; error = error ? `${error}; ${why}` : why; } const blocks: XArticleBlock[] = content.blocks.map((b) => b.type === "image" && b.src && images.saved.has(b.src) ? { ...b, file: images.saved.get(b.src)!.name } : b, ); const article = { version: 1, ...base, ...(content.title ? { title: content.title } : {}), ...(content.author ? { author: content.author } : {}), ...(content.handle ? { handle: content.handle } : {}), ...(content.publishedAt ? { publishedAt: content.publishedAt } : {}), capturedAt, extraction: content.extraction, ...(trimmed ? { trimmed: true } : {}), blocks, }; await writeJsonAtomic(path.join(dir, ARTICLE_JSON_FILENAME), article, { mkdir: true }); await writeFileAtomic( path.join(dir, ARTICLE_MD_FILENAME), xArticleMarkdown({ ...content, blocks, url: link.url }), ); await writeFileAtomic(path.join(dir, ARTICLE_HTML_FILENAME), html); if (png) await writeFile(path.join(dir, ARTICLE_SHOT_FILENAME), png); await removeStaleImages(dir, new Set([...images.saved.values()].map((f) => f.name))); const files: CapturedFile[] = [ await describeCapturedFile(dir, ARTICLE_JSON_FILENAME, link.url), await describeCapturedFile(dir, ARTICLE_MD_FILENAME, link.url), await describeCapturedFile(dir, ARTICLE_HTML_FILENAME, link.url), ...(png ? [await describeCapturedFile(dir, ARTICLE_SHOT_FILENAME, link.url)] : []), ...images.saved.values(), ]; return { record: { ...base, capturedAt, state: "captured", ...(content.title ? { title: content.title } : {}), blocks: blocks.length, extraction: content.extraction, files, ...(trimmed ? { trimmed: true } : {}), ...(error ? { error } : {}), }, }; } // Each distinct image src, fetched through the page's request context into // `article-img-.`: the full-size picture first, the src as rendered if // that is refused. async function fetchArticleImages( page: PageLike, blocks: ReadonlyArray, dir: string, opts: XArticleCaptureOptions, ): Promise<{ saved: Map; errors: string[] }> { const saved = new Map(); const errors: string[] = []; const srcs = [...new Set(blocks.flatMap((b) => (b.type === "image" && b.src ? [b.src] : [])))]; if (srcs.length === 0) return { saved, errors }; if (!page.request) { errors.push("this browser page cannot fetch (no request context)"); return { saved, errors }; } let n = 0; for (const src of srcs) { if (opts.signal?.aborted) { errors.push("cancelled before the rest of the images"); break; } if (n > 0) await opts.imageGap?.(); n++; const tries = [...new Set([fullSizeImageUrl(src), src])]; let last = ""; for (const url of tries) { try { const res = await page.request.get(url, { timeout: 30_000, failOnStatusCode: false }); if (!res.ok()) { last = `HTTP ${res.status()} for ${url}`; continue; } const body = await res.body(); const name = articleImageFilename(n, imageExtension(url, res.headers()["content-type"])); await writeFile(path.join(dir, name), body); saved.set(src, await describeCapturedFile(dir, name, url)); last = ""; break; } catch (err) { last = `${firstLine(err)} (${url})`; } } if (last) errors.push(last); } if (saved.size) opts.onLog?.(`article: ${saved.size} image(s) saved.`); return { saved, errors }; } // A re-capture with fewer images leaves no older numbered file behind. async function removeStaleImages(dir: string, keep: ReadonlySet): Promise { let names: string[]; try { names = await readdir(dir); } catch { return; } for (const name of names) { if (/^article-img-\d+\./.test(name) && !keep.has(name)) { await rm(path.join(dir, name), { force: true }); } } } function firstLine(err: unknown): string { return ((err as Error)?.message ?? String(err)).split("\n")[0]; }