Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 12ca8cf0aa2b62b91d7525c1c8c5618cc11fe91d
parent 195dbff2fb7b2cc2a94e8f741b6de2775f7b24fa
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 20:09:08 -0400

common: capture-posts opens the X Article a post links to (articles, default on) and saves it beside the post

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/capturePosts.test.ts | 76+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/controller/capturePosts.ts | 34+++++++++++++++++++++++++++++++---
Mcommon/social/fetchers.ts | 7+++++++
Mcommon/social/playwrightRuntime.ts | 13+++++++++++++
Mcommon/social/postCapture.test.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Mcommon/social/postCapture.ts | 67+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Acommon/social/xArticleCapture.ts | 391+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.ts | 16++++++++++++++--
Mcommon/social/xPostCapture.ts | 99+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
9 files changed, 736 insertions(+), 30 deletions(-)

diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts @@ -14,7 +14,7 @@ process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); const { getPaths } = await import("../lib/paths"); const { registerSocialFetcher } = await import("../social/fetchers"); -const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server"); +const { readPostAvailability, writePostAvailability, writePosts } = await import("../lib/posts-server"); const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts"); type Input = import("../social/fetchers").PostCaptureInput; type Result = import("../social/fetchers").PostCaptureResult; @@ -36,6 +36,19 @@ registerSocialFetcher({ }, }); registerSocialFetcher({ + id: "test-capture-x", + label: "Test capture (X)", + platform: "twitter", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), + captureByIds: async (input) => { + calls.push(input); + return answer; + }, +}); +registerSocialFetcher({ id: "test-no-capture", label: "No-capture fetcher", platform: "bluesky", @@ -143,3 +156,64 @@ test("a run stopped by the source fails the job; one cancelled by the operator d const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal }); assert.equal(cancelled.ok, true); }); + +test("the articles: on by default, refused alone only when not asked for by name, archived text passed (X)", async () => { + const root = path.join(channelsDir, "demo-x"); + await mkdir(root, { recursive: true }); + await writeFile( + path.join(root, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + platform: "twitter", + postFetcher: "test-capture-x", + socialHandle: "example_user", + name: "Example (X)", + url: "https://x.com/example_user", + }), + ); + const post = (id: string, text: string, links: string[] = []) => ({ + id, + slug: `demo-x/${id}`, + channelSlug: "demo-x", + author: "example_user", + createdAt: "2026-01-02T03:04:05.000Z", + uploadDate: "20260102", + text, + url: `https://x.com/example_user/status/${id}`, + platform: "twitter" as const, + isReply: false, + isRepost: false, + links, + }); + await writePosts(root, [ + post("111", "https://x.com/i/article/999"), + post("222", "a plain post", ["https://example.com/page"]), + post("333", "not asked for"), + ]); + answer = { outcomes: [] }; + + calls = []; + assert.equal((await run("demo-x", ["111", "222"])).ok, true); + assert.equal(calls[0].articles, undefined); + const archived = calls[0].archived ?? new Map(); + assert.deepEqual([...archived.keys()].sort(), ["111", "222"]); + assert.deepEqual(archived.get("111"), { text: "https://x.com/i/article/999", links: [] }); + assert.deepEqual(archived.get("222"), { text: "a plain post", links: ["https://example.com/page"] }); + + calls = []; + assert.equal((await run("demo-x", ["111"], { articles: false })).ok, true); + assert.equal(calls[0].articles, false); + assert.equal(calls[0].archived, undefined); + + // Both halves off: refused, unless the articles are asked for by name. + calls = []; + assert.equal((await run("demo-x", ["111"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE); + assert.equal( + (await run("demo-x", ["111"], { shots: false, media: false, articles: false })).error, + NOTHING_TO_CAPTURE, + ); + assert.equal(calls.length, 0); + assert.equal((await run("demo-x", ["111"], { shots: false, media: false, articles: true })).ok, true); + assert.equal(calls[0].articles, true); +}); diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts @@ -10,7 +10,9 @@ // nothing about the post). // // Only ids already in the channel's posts archive are captured: a capture is -// of the archive, and an id from somewhere else is refused by name. +// of the archive, and an id from somewhere else is refused by name. Each id's +// archived text and links go to the fetcher with it, so a post that is an X +// Article's link is known as one before its page is opened. import path from "node:path"; import { readChannelConfig } from "./channels"; @@ -24,6 +26,7 @@ import { } from "../lib/cookiePolicy"; import { mergePostAvailability, + readAllPosts, readPostAvailability, readSeenPostIds, writePostAvailability, @@ -54,6 +57,8 @@ export type CapturePostsOptions = { media?: boolean; // Capture again what is already captured. force?: boolean; + // Also open and save the X Article a post links to. Defaults to true. + articles?: boolean; onLog?: (line: string) => void; signal?: AbortSignal; // The job's drain: stop between posts. @@ -79,7 +84,17 @@ export function capturePostsProblem( } export const NOTHING_TO_CAPTURE = - "Nothing to capture: both the screenshot and the media are turned off."; + 'Nothing to capture: both the screenshot and the media are turned off (to read only the articles, ask for "articles": true).'; + +// Both halves off is nothing to do — unless the articles were asked for by +// name: their default alone does not make a run. +export function nothingToCapture(o: { + shots?: boolean; + media?: boolean; + articles?: boolean; +}): boolean { + return o.shots === false && o.media === false && o.articles !== true; +} // The ids not in the channel's archive, or null when every one is. export function strayCaptureIds( @@ -103,7 +118,7 @@ export async function capturePosts( const channelRoot = path.join(paths.channelsDir, slug); const ids = [...new Set(opts.ids)]; - if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE); + if (nothingToCapture(opts)) return fail(NOTHING_TO_CAPTURE); if (ids.length === 0) return fail("No post ids to capture."); const config = await readChannelConfig(paths, slug); @@ -121,6 +136,16 @@ export async function capturePosts( const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot)); if (stray) return fail(strayIdsRefusal(slug, stray)); + // The archived text of each id, for the article links in it (X only). + let archived: Map<string, { text: string; links: string[] }> | undefined; + if (opts.articles !== false && fetcher.platform === "twitter") { + const wanted = new Set(ids); + archived = new Map(); + for (const p of await readAllPosts(channelRoot)) { + if (wanted.has(p.id)) archived.set(p.id, { text: p.text, links: p.links ?? [] }); + } + } + const policy = resolveCookiePolicy(settings, config); const xLogin = fetcher.platform === "twitter" @@ -132,6 +157,7 @@ export async function capturePosts( [ opts.shots === false ? "" : " screenshot", opts.media === false ? "" : " media", + opts.articles === false || fetcher.platform !== "twitter" ? "" : " articles", ].join("") + (opts.force ? ", again where already captured" : "") + ".", @@ -147,6 +173,8 @@ export async function capturePosts( shots: opts.shots, media: opts.media, force: opts.force, + articles: opts.articles, + archived, cookies: alwaysCookies(policy), cookieSource: xLogin?.source, browserCookies: xLogin?.browserSpec, diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts @@ -182,6 +182,13 @@ export type PostCaptureInput = Pick< media?: boolean; // Capture again what is already captured. force?: boolean; + // Also capture the long-form article a post links to (X Articles), where it + // links to one. Defaults to true. + articles?: boolean; + // What the posts archive holds for each id — its text and expanded links — + // for the links a capture follows (an X Article's). Ids absent are read + // from the page alone. + archived?: ReadonlyMap<string, { text: string; links?: ReadonlyArray<string> }>; // A soft stop: no new post is started once it fires, and the one in hand // finishes (`signal` cancels outright). drain?: AbortSignal; diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts @@ -48,6 +48,19 @@ export type PageLike = { fullPage?: boolean; clip?: { x: number; y: number; width: number; height: number }; }) => Promise<Uint8Array>; + // The page's own request context (the profile's cookies): an X Article's + // inline images are fetched through it (xArticleCapture.ts). + request?: { + get: ( + url: string, + opts?: { timeout?: number; failOnStatusCode?: boolean }, + ) => Promise<{ + ok: () => boolean; + status: () => number; + headers: () => Record<string, string>; + body: () => Promise<Uint8Array>; + }>; + }; }; export type BrowserContextLike = { diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts @@ -11,10 +11,12 @@ import os from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { + articleImageFilename, CAPTURE_FILENAME, captureAvailability, captureWork, describeCapturedFile, + isArticleFile, listCapturedMediaFiles, POSTS_MEDIA_DIRNAME, postCaptureDir, @@ -22,6 +24,7 @@ import { readPostCapture, SHOT_FILENAME, writePostCapture, + type ArticleCaptureRecord, type PostCaptureRecord, } from "./postCapture"; @@ -75,6 +78,12 @@ test("capture.json: each file's sha256 and byte size, the URLs, round-tripped", assert.deepEqual(await readPostCapture(dir), rec); // The record and the shot are not media. assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]); + + // Nor is the article half, whatever a later gallery-dl run finds beside it. + for (const name of ["article.json", "article.md", "article.png", "article.html", "article-img-1.jpg"]) { + await writeFile(path.join(dir, name), "x"); + } + assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]); }); test("capture.json: absent, unparseable or of another shape reads as no capture", async () => { @@ -97,27 +106,65 @@ test("availability: only what the page said about the post — never a verdict f test("what a capture owes: nothing for a settled post, only the missing half otherwise, everything when forced", () => { const all = { shots: true, media: true, force: false }; - assert.deepEqual(captureWork(null, all), { shot: true, media: true }); - assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false }); + assert.deepEqual(captureWork(null, all), { shot: true, media: true, article: false }); + assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false, article: false }); const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" }); - assert.deepEqual(captureWork(done, all), { shot: false, media: false }); - assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false }); + assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: false }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false, article: false }); // The media failed (or was never asked for): only the media is owed. - assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true }); - assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true, article: false }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true, article: false }); // A shot that failed is owed again. - assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false }); + assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false, article: false }); // Deleted is settled: asking again is a request for the same answer. assert.deepEqual(captureWork(record({ state: "deleted", mediaState: "skipped" }), all), { shot: false, media: false, + article: false, }); // force re-takes whatever was asked for. - assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true }); + assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true, article: false }); assert.deepEqual(captureWork(record({ state: "deleted" }), { shots: false, media: true, force: true }), { shot: false, media: true, + article: false, + }); +}); + +test("what a capture owes the article half: only when asked, settled once captured, deleted or walled", () => { + const all = { shots: true, media: true, force: false, articles: true }; + const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" }); + const article = (state: ArticleCaptureRecord["state"]): ArticleCaptureRecord => ({ + articleId: "9", + url: "https://x.com/i/article/9", + capturedAt: "2026-01-02T03:04:05.000Z", + state, + blocks: 0, + files: [], }); + assert.equal(captureWork(null, all).article, true); + assert.equal(captureWork(null, { ...all, articles: false }).article, false); + // Not asked for by name: the older callers' shape owes no article. + assert.equal(captureWork(null, { shots: true, media: true, force: false }).article, false); + // A post shot before articles were captured owes only its article. + assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: true }); + for (const state of ["captured", "deleted", "unavailable"] as const) { + assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, false, state); + } + for (const state of ["error", "login-wall"] as const) { + assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, true, state); + } + // force re-reads a captured article; a deleted post owes nothing. + assert.equal(captureWork(record({ ...done, article: article("captured") }), { ...all, force: true }).article, true); + assert.equal(captureWork(record({ state: "deleted" }), all).article, false); +}); + +test("article files are named by the layout, images numbered from 1", () => { + assert.equal(articleImageFilename(1, "jpg"), "article-img-1.jpg"); + assert.ok(isArticleFile("article.md")); + assert.ok(isArticleFile("article-img-12.png")); + assert.ok(!isArticleFile("123_1.jpg")); + assert.ok(!isArticleFile("articles.txt")); }); // THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON diff --git a/common/social/postCapture.ts b/common/social/postCapture.ts @@ -4,6 +4,10 @@ // channels/<slug>/posts-media/<post id>/shot.png — the post, as rendered // channels/<slug>/posts-media/<post id>/<media…> — its attached media // channels/<slug>/posts-media/<post id>/capture.json — this module's record +// channels/<slug>/posts-media/<post id>/article.* — the X Article the post +// links to, when it is one: article.json (its blocks), article.md, +// article.png, article.html (the root as X served it) and +// article-img-<n>.<ext> (its inline images) — xArticleCapture.ts // // A directory per post, unlike the posts themselves (month-sharded JSONL): a // capture is a handful of files, and only for the posts someone asked for. It @@ -26,6 +30,20 @@ import type { PostAvailability } from "../lib/posts"; export const POSTS_MEDIA_DIRNAME = "posts-media"; export const CAPTURE_FILENAME = "capture.json"; export const SHOT_FILENAME = "shot.png"; +export const ARTICLE_JSON_FILENAME = "article.json"; +export const ARTICLE_MD_FILENAME = "article.md"; +export const ARTICLE_SHOT_FILENAME = "article.png"; +export const ARTICLE_HTML_FILENAME = "article.html"; + +// The n-th (1-based) inline image of an article. +export function articleImageFilename(n: number, ext: string): string { + return `article-img-${n}.${ext}`; +} + +// A file of the article half, never a medium of the post. +export function isArticleFile(name: string): boolean { + return /^article(\.|-img-)/.test(name); +} export function postsMediaDir(channelRoot: string): string { return path.join(channelRoot, POSTS_MEDIA_DIRNAME); @@ -88,6 +106,28 @@ export type CapturedFile = { // asked for or not attempted ("skipped"), or failed ("error"). export type CaptureMediaState = "ok" | "none" | "skipped" | "error"; +// The article half: an X Article the post links to, opened and read +// (xArticleCapture.ts). `state` is the article page's, in the post's terms: a +// deleted or walled article is settled, an error is owed again, a login wall +// stopped the run. +export type ArticleCaptureRecord = { + articleId: string; + url: string; + capturedAt: string; + state: PostCaptureState; + title?: string; + // The blocks read, for a captured article (article.json holds them). + blocks: number; + // How the body was read: by X's markers, or the root's text split at block + // elements because the markers were missing. + extraction?: "structured" | "fallback"; + // article.json, article.md, article.png, article.html, each image. + files: CapturedFile[]; + // article.png stops at a height cap; the article ran longer. + trimmed?: boolean; + error?: string; +}; + export type PostCaptureRecord = { version: 1; id: string; @@ -102,6 +142,8 @@ export type PostCaptureRecord = { shot?: CapturedFile; mediaState: CaptureMediaState; media: CapturedFile[]; + // The X Article the post links to, when one was captured or tried. + article?: ArticleCaptureRecord; error?: string; }; @@ -128,8 +170,8 @@ export async function describeCapturedFile( return { name, ...digest, ...(url ? { url } : {}) }; } -// The media files in a post's directory: everything but the shot, the record -// and a download's leftovers. +// The media files in a post's directory: everything but the shot, the record, +// the article half and a download's leftovers. export async function listCapturedMediaFiles(dir: string): Promise<string[]> { let names: string[]; try { @@ -142,6 +184,7 @@ export async function listCapturedMediaFiles(dir: string): Promise<string[]> { (n) => n !== SHOT_FILENAME && n !== CAPTURE_FILENAME && + !isArticleFile(n) && !n.endsWith(".part") && !n.startsWith("."), ) @@ -171,19 +214,31 @@ export async function writePostCapture( // A deleted post is settled: the platform said so, and asking again costs a // request for the same answer (`force` asks anyway). A shot on disk is kept; a // media download that did not finish ("error", or never attempted) is owed. +// The article half is owed only if the post links to one (the caller knows); +// one captured, deleted or walled is settled, one that failed is owed. export function captureWork( existing: PostCaptureRecord | null, - wanted: { shots: boolean; media: boolean; force: boolean }, -): { shot: boolean; media: boolean } { + wanted: { shots: boolean; media: boolean; force: boolean; articles?: boolean }, +): { shot: boolean; media: boolean; article: boolean } { + const articles = wanted.articles ?? false; if (wanted.force || !existing) { - return { shot: wanted.shots, media: wanted.media }; + return { shot: wanted.shots, media: wanted.media, article: articles }; } - if (existing.state === "deleted") return { shot: false, media: false }; + if (existing.state === "deleted") return { shot: false, media: false, article: false }; return { shot: wanted.shots && !existing.shot, media: wanted.media && existing.mediaState !== "ok" && existing.mediaState !== "none", + article: articles && !articleSettled(existing.article), }; } + +export function articleSettled(article: ArticleCaptureRecord | undefined): boolean { + return ( + article?.state === "captured" || + article?.state === "deleted" || + article?.state === "unavailable" + ); +} diff --git a/common/social/xArticleCapture.ts b/common/social/xArticleCapture.ts @@ -0,0 +1,391 @@ +// Capturing an X Article (a long-form post) beside the post that links to it: +// the article page opened in the same logged-in profile and page as the post's +// shot, read into blocks (xArticle.ts), and saved as +// article.json — the blocks, title, byline, in reading order +// article.md — the same as readable markdown +// article.png — the article root, shot whole up to a height cap +// article.html — the root as X served it, so a better reading later costs no +// second visit +// article-img-<n>.<ext> — its inline images, fetched through the page's own +// request context (the profile's cookies), not a new tool +// +// PACED AS ONE CONTACT. The article load is a contact with X like the post's +// shot: the run waits its 4–10 s gap before it (xPostCapture.ts). The images +// are the page's own, already loaded once by the browser; they are fetched +// again a short gap apart, not at the post gap. +// +// The page is read as a post page is (classifyXPostSnapshot): a login wall +// stops the run (needsCookies), "Something went wrong" stops it, a deleted or +// unavailable article is recorded and settled, anything else is an error a +// later run tries again. Nothing is retried in the run that met it. +// +// NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing, +// tested against recorded snapshots and a written HTML fixture, never x.com. + +import { readdir, rm, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server"; +import type { PageLike } from "./playwrightRuntime"; +import { + ARTICLE_HTML_FILENAME, + ARTICLE_JSON_FILENAME, + ARTICLE_MD_FILENAME, + ARTICLE_SHOT_FILENAME, + articleImageFilename, + describeCapturedFile, + type ArticleCaptureRecord, + type CapturedFile, +} from "./postCapture"; +import { + extractXArticle, + findXArticleLink, + xArticleMarkdown, + type XArticleBlock, + type XArticleLink, +} from "./xArticle"; +import { classifyXPostSnapshot, type XPostVerdict } from "./xPostCapture"; + +// The article root, most specific first. The read view holds the title, the +// byline and the body; the rich-text view only the body, so it is widened to +// the article around it. +export const ARTICLE_ROOT_MARKERS = [ + '[data-testid="twitterArticleReadView"]', + '[data-testid="twitterArticleRichTextView"]', + '[data-testid="longformRichTextComponent"]', +]; +const ARTICLE_FALLBACK_ROOT = '[data-testid="primaryColumn"] article, [data-testid="primaryColumn"] [role="article"]'; + +// article.png stops here. Chromium's full-page capture is a single texture, +// and past its 16384 px limit a taller shot repeats or blanks, so the cap sits +// under it; a longer article is recorded `trimmed` (article.md has it all). +export const ARTICLE_SHOT_MAX_HEIGHT = 16_000; + +export type XArticleSnapshot = { + path: string; + text: string; + root: null | { + rect: { x: number; y: number; width: number; height: number }; + // Which marker found the root ("fallback": none did). + marker: string; + }; +}; + +const ARTICLE_SNAPSHOT_SCRIPT = `(() => { + const markers = ${JSON.stringify(ARTICLE_ROOT_MARKERS)}; + let root = null; + let marker = null; + for (const m of markers) { + const el = document.querySelector(m); + if (el) { root = el; marker = m; break; } + } + if (root && marker !== markers[0]) { + root = root.closest('article, [role="article"]') || root; + } + if (!root) { + root = document.querySelector(${JSON.stringify(ARTICLE_FALLBACK_ROOT)}); + if (root) marker = "fallback"; + } + const main = document.querySelector('[data-testid="primaryColumn"]') || document.body; + const text = ((main && main.innerText) || "").slice(0, 4000); + let out = null; + if (root) { + for (const old of document.querySelectorAll('[data-archilyzer-article]')) { + old.removeAttribute('data-archilyzer-article'); + } + root.setAttribute('data-archilyzer-article', ''); + const r = root.getBoundingClientRect(); + out = { + rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, + marker, + }; + } + return { path: location.pathname, text, root: out }; +})()`; + +// Walk down the page so lazy images load, then back to the top for the shot. +const LOAD_LAZY_SCRIPT = `(async () => { + const root = document.querySelector('[data-archilyzer-article]') || document.body; + for (const img of root.querySelectorAll('img[loading="lazy"]')) img.loading = "eager"; + const wait = (ms) => new Promise((r) => setTimeout(r, ms)); + let steps = 0; + for (let y = 0; y < document.documentElement.scrollHeight && steps < 200; steps++) { + y += Math.max(400, Math.floor(window.innerHeight * 0.8)); + window.scrollTo(0, y); + await wait(250); + } + window.scrollTo(0, 0); + await wait(500); + const pending = Array.from(root.querySelectorAll("img")).filter((i) => !i.complete); + await Promise.all(pending.map((i) => new Promise((r) => { + i.addEventListener("load", r, { once: true }); + i.addEventListener("error", r, { once: true }); + setTimeout(r, 5000); + }))); + return steps; +})()`; + +const ARTICLE_HTML_SCRIPT = `(() => { + const root = document.querySelector('[data-archilyzer-article]'); + return root ? root.outerHTML : ""; +})()`; + +// What an article page means, in a post's terms. A page with a root is +// captured; one without is read for X's markers as a post page is. +export function classifyXArticleSnapshot(s: XArticleSnapshot): XPostVerdict { + return classifyXPostSnapshot( + { + path: s.path, + text: s.text, + article: s.root ? { rect: s.root.rect, sensitive: false } : null, + }, + "article", + ); +} + +// The article a post's rendered card links to: the hrefs read from it. +export function xArticleLinkFromCard(hrefs: ReadonlyArray<string> | undefined): XArticleLink | null { + return hrefs?.length ? findXArticleLink(hrefs) : null; +} + +// The full-size picture of an X media URL (`name=orig`); anything else as is. +export function fullSizeImageUrl(src: string): string { + try { + const u = new URL(src); + if (u.hostname === "pbs.twimg.com" && u.pathname.startsWith("/media/")) { + u.searchParams.set("name", "orig"); + return u.toString(); + } + } catch { + /* not a URL: as is */ + } + return src; +} + +const EXT_BY_TYPE: Record<string, string> = { + "image/jpeg": "jpg", + "image/png": "png", + "image/gif": "gif", + "image/webp": "webp", + "image/avif": "avif", +}; + +// An image's extension: X's `format=` parameter, the path's own, the response's +// type, in that order. +export function imageExtension(url: string, contentType?: string): string { + try { + const u = new URL(url); + const format = u.searchParams.get("format"); + if (format && /^[a-z0-9]{2,5}$/i.test(format)) return format.toLowerCase(); + const m = /\.([a-z0-9]{2,5})$/i.exec(u.pathname); + if (m) return m[1].toLowerCase() === "jpeg" ? "jpg" : m[1].toLowerCase(); + } catch { + /* fall through to the type */ + } + const type = (contentType ?? "").split(";")[0].trim().toLowerCase(); + return EXT_BY_TYPE[type] ?? "img"; +} + +export type XArticleCaptureResult = { + record: ArticleCaptureRecord; + // Set when the run must stop here (a login wall, X refusing pages). + stop?: string; +}; + +export type XArticleCaptureOptions = { + now: () => Date; + onLog?: (line: string) => void; + signal?: AbortSignal; + // The gap between one image fetch and the next. + imageGap?: () => Promise<void>; +}; + +// Open the article, read it, shoot it, fetch its images, write its files into +// `dir`. The record says how it went; nothing is retried here. +export async function captureXArticle( + page: PageLike, + link: XArticleLink, + dir: string, + opts: XArticleCaptureOptions, +): Promise<XArticleCaptureResult> { + const { onLog } = opts; + const base = { articleId: link.articleId, url: link.url }; + const failed = (verdict: XPostVerdict): XArticleCaptureResult => ({ + record: { + ...base, + capturedAt: opts.now().toISOString(), + state: verdict.state, + blocks: 0, + files: [], + ...(verdict.error ? { error: verdict.error } : {}), + }, + ...(verdict.stop ? { stop: verdict.stop } : {}), + }); + + try { + await page.goto(link.url, { waitUntil: "domcontentloaded", timeout: 60_000 }); + } catch (err) { + return failed({ state: "error", error: `Could not load ${link.url}: ${firstLine(err)}` }); + } + await page + .waitForSelector([...ARTICLE_ROOT_MARKERS, ARTICLE_FALLBACK_ROOT].join(", "), { timeout: 20_000 }) + .catch(() => {}); + await page.waitForTimeout(1_500); + let snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; + const verdict = classifyXArticleSnapshot(snap); + if (verdict.state !== "captured") return failed(verdict); + if (snap.root?.marker === "fallback") { + onLog?.(`article ${link.articleId}: no article marker on the page — reading the page's post as the article.`); + } + + await page.evaluate(LOAD_LAZY_SCRIPT).catch(() => {}); + snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; + const html = String((await page.evaluate(ARTICLE_HTML_SCRIPT)) ?? ""); + const rect = snap.root?.rect; + if (!html || !rect || rect.width < 1 || rect.height < 1) { + return failed({ state: "error", error: "The article rendered with nothing to read." }); + } + const content = extractXArticle(html); + const capturedAt = opts.now().toISOString(); + + // The shot: the root, whole, up to the cap. + const trimmed = rect.height > ARTICLE_SHOT_MAX_HEIGHT; + let png: Uint8Array | undefined; + let error: string | undefined; + try { + png = await page.screenshot({ + type: "png", + fullPage: true, + clip: { + x: Math.max(0, Math.floor(rect.x)), + y: Math.max(0, Math.floor(rect.y)), + width: Math.ceil(rect.width), + height: Math.min(Math.ceil(rect.height), ARTICLE_SHOT_MAX_HEIGHT), + }, + }); + } catch (err) { + error = `The article's screenshot failed: ${firstLine(err)}`; + } + + // The images, each once, numbered in reading order. + const images = await fetchArticleImages(page, content.blocks, dir, opts); + if (images.errors.length) { + const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`; + error = error ? `${error}; ${why}` : why; + } + + const blocks: XArticleBlock[] = content.blocks.map((b) => + b.type === "image" && b.src && images.saved.has(b.src) + ? { ...b, file: images.saved.get(b.src)!.name } + : b, + ); + const article = { + version: 1, + ...base, + ...(content.title ? { title: content.title } : {}), + ...(content.author ? { author: content.author } : {}), + ...(content.handle ? { handle: content.handle } : {}), + ...(content.publishedAt ? { publishedAt: content.publishedAt } : {}), + capturedAt, + extraction: content.extraction, + ...(trimmed ? { trimmed: true } : {}), + blocks, + }; + await writeJsonAtomic(path.join(dir, ARTICLE_JSON_FILENAME), article, { mkdir: true }); + await writeFileAtomic( + path.join(dir, ARTICLE_MD_FILENAME), + xArticleMarkdown({ ...content, blocks, url: link.url }), + ); + await writeFileAtomic(path.join(dir, ARTICLE_HTML_FILENAME), html); + if (png) await writeFile(path.join(dir, ARTICLE_SHOT_FILENAME), png); + await removeStaleImages(dir, new Set([...images.saved.values()].map((f) => f.name))); + + const files: CapturedFile[] = [ + await describeCapturedFile(dir, ARTICLE_JSON_FILENAME, link.url), + await describeCapturedFile(dir, ARTICLE_MD_FILENAME, link.url), + await describeCapturedFile(dir, ARTICLE_HTML_FILENAME, link.url), + ...(png ? [await describeCapturedFile(dir, ARTICLE_SHOT_FILENAME, link.url)] : []), + ...images.saved.values(), + ]; + return { + record: { + ...base, + capturedAt, + state: "captured", + ...(content.title ? { title: content.title } : {}), + blocks: blocks.length, + extraction: content.extraction, + files, + ...(trimmed ? { trimmed: true } : {}), + ...(error ? { error } : {}), + }, + }; +} + +// Each distinct image src, fetched through the page's request context into +// `article-img-<n>.<ext>`: the full-size picture first, the src as rendered if +// that is refused. +async function fetchArticleImages( + page: PageLike, + blocks: ReadonlyArray<XArticleBlock>, + dir: string, + opts: XArticleCaptureOptions, +): Promise<{ saved: Map<string, CapturedFile>; errors: string[] }> { + const saved = new Map<string, CapturedFile>(); + const errors: string[] = []; + const srcs = [...new Set(blocks.flatMap((b) => (b.type === "image" && b.src ? [b.src] : [])))]; + if (srcs.length === 0) return { saved, errors }; + if (!page.request) { + errors.push("this browser page cannot fetch (no request context)"); + return { saved, errors }; + } + let n = 0; + for (const src of srcs) { + if (opts.signal?.aborted) { + errors.push("cancelled before the rest of the images"); + break; + } + if (n > 0) await opts.imageGap?.(); + n++; + const tries = [...new Set([fullSizeImageUrl(src), src])]; + let last = ""; + for (const url of tries) { + try { + const res = await page.request.get(url, { timeout: 30_000, failOnStatusCode: false }); + if (!res.ok()) { + last = `HTTP ${res.status()} for ${url}`; + continue; + } + const body = await res.body(); + const name = articleImageFilename(n, imageExtension(url, res.headers()["content-type"])); + await writeFile(path.join(dir, name), body); + saved.set(src, await describeCapturedFile(dir, name, url)); + last = ""; + break; + } catch (err) { + last = `${firstLine(err)} (${url})`; + } + } + if (last) errors.push(last); + } + if (saved.size) opts.onLog?.(`article: ${saved.size} image(s) saved.`); + return { saved, errors }; +} + +// A re-capture with fewer images leaves no older numbered file behind. +async function removeStaleImages(dir: string, keep: ReadonlySet<string>): Promise<void> { + let names: string[]; + try { + names = await readdir(dir); + } catch { + return; + } + for (const name of names) { + if (/^article-img-\d+\./.test(name) && !keep.has(name)) { + await rm(path.join(dir, name), { force: true }); + } + } +} + +function firstLine(err: unknown): string { + return ((err as Error)?.message ?? String(err)).split("\n")[0]; +} diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -41,6 +41,7 @@ import { } from "./xSessionBroker"; import type { XCookieSource } from "./xCookieSource"; import { listCapturedMediaFiles } from "./postCapture"; +import { xArticleLinkFromArchive } from "./xArticle"; import { captureXPosts, xStatusUrl, @@ -328,11 +329,17 @@ export async function captureXPostsByIds( input: PostCaptureInput, ): Promise<PostCaptureResult> { const paths = getPaths(); - if (input.shots ?? true) { + // The profile shoots the posts and opens the articles. A post is known to + // link to an article before any page only by its archived text; one found + // on a card is found by a shot, which already needs the profile. + const articles = + (input.articles ?? true) && + input.ids.some((id) => xArticleLinkFromArchive(input.archived?.get(id)) !== null); + if ((input.shots ?? true) || articles) { const status = await readXSessionStatus(paths); if (!status.hasProfile) { const why = - "No X session profile to shoot posts with: connect an X account on /settings, then run it again."; + "No X session profile to shoot posts or open articles with: connect an X account on /settings, then run it again."; input.onLog?.(`[auth] ${why}`); return { outcomes: [], needsCookies: true, stoppedEarly: why }; } @@ -355,9 +362,14 @@ export async function captureXPostsByIds( }, pauseMs: () => xRequestPauseMs(), pause, + articleImagePauseMs: () => ARTICLE_IMAGE_PAUSE_MS + Math.floor(Math.random() * ARTICLE_IMAGE_PAUSE_MS), }); } +// The gap between an article's images: 1–2 s. They are the page's own +// pictures, already loaded once by the browser, so not the full post gap. +const ARTICLE_IMAGE_PAUSE_MS = 1_000; + // --------------------------------------------------------------------------- // Search: the older-posts backfill // --------------------------------------------------------------------------- diff --git a/common/social/xPostCapture.ts b/common/social/xPostCapture.ts @@ -21,6 +21,10 @@ // - a protected, suspended or vanished account, or a withheld post: // unavailable. // +// A post that is an X Article (long-form) links to it — in its archived text, +// or on its rendered card — and the run then opens the article too, one more +// paced contact (xArticleCapture.ts), unless `articles` is off. +// // NOT VERIFIED AGAINST LIVE X. The page markers below are X's as of this // writing and are tested against recorded snapshots, never x.com; the first // real run is the check. @@ -41,11 +45,14 @@ import { readPostCapture, SHOT_FILENAME, writePostCapture, + type ArticleCaptureRecord, type CapturedFile, type CaptureMediaState, type PostCaptureRecord, type PostCaptureState, } from "./postCapture"; +import { xArticleLinkFromArchive, type XArticleLink } from "./xArticle"; +import { captureXArticle, xArticleLinkFromCard } from "./xArticleCapture"; export function xStatusUrl(id: string): string { return `https://x.com/i/status/${id}`; @@ -60,6 +67,8 @@ export type XPostSnapshot = { rect: { x: number; y: number; width: number; height: number }; // A "Show" / "View" button inside the post: a sensitive-media cover. sensitive: boolean; + // The hrefs inside the post that look like an X Article's: its card. + articleHrefs?: string[]; }; }; @@ -83,9 +92,14 @@ const SNAPSHOT_SCRIPT = (id: string) => `(() => { const sensitive = Array.from(own.querySelectorAll('button, [role="button"]')).some( (b) => /^(show|view)$/i.test((b.innerText || "").trim()), ); + const articleHrefs = Array.from(own.querySelectorAll('a[href*="/article/"]')) + .map((l) => l.getAttribute("href") || "") + .filter(Boolean) + .slice(0, 10); article = { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, sensitive, + articleHrefs, }; } return { path: location.pathname, text, article }; @@ -140,7 +154,12 @@ export type XPostVerdict = { }; // What a snapshot means. Pure, so every marker is testable without a browser. -export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { +// `noun` names what the page should have shown (an article page is read the +// same way, xArticleCapture.ts). +export function classifyXPostSnapshot( + s: Pick<XPostSnapshot, "path" | "text" | "article">, + noun = "post", +): XPostVerdict { if (/^\/(i\/flow\/login|login|i\/flow\/signup)\b/.test(s.path)) { return { state: "login-wall", @@ -153,7 +172,7 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { if (REFUSED_TEXT.test(s.text)) { return { state: "error", - error: "X answered “Something went wrong” instead of the post.", + error: `X answered “Something went wrong” instead of the ${noun}.`, stop: "X is refusing pages right now; stopping rather than asking again.", }; } @@ -162,19 +181,23 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { if (AGE_WALL_TEXT.test(s.text)) { return { state: "error", - error: "X shows this post only to an age-verified session.", + error: `X shows this ${noun} only to an age-verified session.`, }; } if (LOGGED_OUT_TEXT.test(s.text)) { return { state: "login-wall", - stop: "X showed its logged-out page instead of the post.", + stop: `X showed its logged-out page instead of the ${noun}.`, }; } - return { state: "error", error: "The post did not render." }; + return { state: "error", error: `The ${noun} did not render.` }; } -export type XShotResult = XPostVerdict & { shot?: CapturedFile }; +export type XShotResult = XPostVerdict & { + shot?: CapturedFile; + // The X Article the post's card links to, when it does. + articleLink?: XArticleLink; +}; // One post's screenshot: load, read the page, open a sensitive cover, shoot // the post's own article. Writes `shot.png` into `dir` only for a post that @@ -234,7 +257,8 @@ export async function shootXPost( await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, SHOT_FILENAME), png); const shot = await describeCapturedFile(dir, SHOT_FILENAME, url); - return { ...verdict, shot }; + const articleLink = xArticleLinkFromCard(snap.article?.articleHrefs); + return { ...verdict, shot, ...(articleLink ? { articleLink } : {}) }; } export type MediaDownloadResult = @@ -255,6 +279,8 @@ export type XCaptureDeps = { // The gap before each contact with X after the first. pauseMs: () => number; pause: (ms: number, signal: AbortSignal) => Promise<void>; + // The gap between one article image and the next (none when absent). + articleImagePauseMs?: () => number; now?: () => Date; }; @@ -274,6 +300,7 @@ export async function captureXPosts( shots: input.shots ?? true, media: input.media ?? true, force: input.force ?? false, + articles: input.articles ?? true, }; const now = deps.now ?? (() => new Date()); const outcomes: PostCaptureOutcome[] = []; @@ -296,7 +323,15 @@ export async function captureXPosts( const dir = postCaptureDir(input.outDir, id); const existing = await readPostCapture(dir); const work = captureWork(existing, wanted); - if (!work.shot && !work.media) { + // The article the post links to, as far as is known before any page: + // the archived text, or the last capture's record. + let articleLink: XArticleLink | null = wanted.articles + ? (xArticleLinkFromArchive(input.archived?.get(id)) ?? + (existing?.article ? { articleId: existing.article.articleId, url: existing.article.url } : null)) + : null; + const articleOwedNow = + work.article && !!articleLink && (!existing || work.shot || existing.state === "captured"); + if (!work.shot && !work.media && !articleOwedNow) { onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`); continue; } @@ -318,6 +353,7 @@ export async function captureXPosts( shot = res.shot; error = res.error; stop = res.stop; + if (wanted.articles && !articleLink && res.articleLink) articleLink = res.articleLink; } let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped"); @@ -347,6 +383,34 @@ export async function captureXPosts( } } + // The article: after the post and its media, one more paced load in the + // same page. Only for a post that is there, in a run not already + // stopping. + let article: ArticleCaptureRecord | undefined = existing?.article; + let articleFailed = false; + if (work.article && articleLink && (state === undefined || state === "captured") && !stop) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + browser ??= await deps.openPage(); + const got = await captureXArticle(browser.page, articleLink, dir, { + now, + onLog, + signal, + imageGap: deps.articleImagePauseMs + ? () => deps.pause(deps.articleImagePauseMs!(), signal) + : undefined, + }); + article = got.record; + articleFailed = article.state === "error"; + // A run that only read the article learns of the post only that it + // linked to a readable article. + state ??= article.state === "captured" ? "captured" : "error"; + if (got.stop) { + stop = got.stop; + if (article.state === "login-wall") needsCookies = true; + } + } + const finalState: PostCaptureState = state ?? "error"; const record: PostCaptureRecord = { version: 1, @@ -358,6 +422,7 @@ export async function captureXPosts( ...(shot ? { shot } : {}), mediaState, media, + ...(article ? { article } : {}), ...(error ? { error } : {}), }; await writePostCapture(dir, record); @@ -369,13 +434,14 @@ export async function captureXPosts( ...(work.shot && captureAvailability(finalState) ? { availability: captureAvailability(finalState) } : {}), - files: (shot ? 1 : 0) + media.length, + files: (shot ? 1 : 0) + media.length + (article?.files.length ?? 0), ...(error ? { error } : {}), }); onLog?.( `${id}: ${finalState}${sensitive ? " (behind a sensitive-media cover)" : ""}` + (shot ? ", shot" : "") + (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") + + (article && article !== existing?.article ? `, ${describeArticle(article)}` : "") + (error ? ` — ${error}` : ""), ); @@ -384,7 +450,7 @@ export async function captureXPosts( needsCookies: needsCookies || finalState === "login-wall", }); } - errorsInARow = finalState === "error" ? errorsInARow + 1 : 0; + errorsInARow = finalState === "error" || articleFailed ? errorsInARow + 1 : 0; if (errorsInARow >= STOP_AFTER_ERRORS) { return stopped( `${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`, @@ -397,6 +463,19 @@ export async function captureXPosts( } } +function describeArticle(a: ArticleCaptureRecord): string { + if (a.state !== "captured") { + return `article ${a.state}` + (a.error ? ` (${a.error})` : ""); + } + return ( + `article${a.title ? ` “${a.title}”` : ""} (${a.blocks} block(s), ${a.files.length} file(s)` + + (a.extraction === "fallback" ? ", read by the fallback" : "") + + (a.trimmed ? ", shot trimmed" : "") + + ")" + + (a.error ? ` — ${a.error}` : "") + ); +} + function firstLine(err: unknown): string { return ((err as Error)?.message ?? String(err)).split("\n")[0]; }