Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c3f3f4e94f69ae7f10f49269366147b7d270da02
parent 959369a21fb7ee323d6bebb953d42fc9cc665068
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 20:14:30 -0400

Merge x-article-capture (capture-posts also reads the X Article a post links to: article.json/md/png/html and its images beside the post's capture)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/components/PostModal.test.ts | 14++++++++++++++
Mcommon/components/PostModal.tsx | 24+++++++++++++++++++++---
Mcommon/controller/capturePosts.test.ts | 76+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/controller/capturePosts.ts | 34+++++++++++++++++++++++++++++++---
Mcommon/jobs/jobKinds.ts | 9+++++----
Acommon/social/__fixtures__/x-article.html | 1+
Mcommon/social/fetchers.ts | 7+++++++
Mcommon/social/playwrightRuntime.ts | 13+++++++++++++
Mcommon/social/postCapture.test.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Mcommon/social/postCapture.ts | 67+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Acommon/social/xArticle.test.ts | 204+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xArticle.ts | 547+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xArticleCapture.test.ts | 472+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xArticleCapture.ts | 392+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.ts | 16++++++++++++++--
Mcommon/social/xPostCapture.ts | 99+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Meditor/CHANGELOG.md | 1+
Aeditor/app/api/ops/capture-posts/route.test.ts | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/api/ops/capture-posts/route.ts | 15++++++++++-----
Meditor/app/channels/[slug]/socialActions.ts | 10+++++++---
Meditor/app/jobs/jobReplayRegistry.ts | 1+
Mscripts/archilyzer-ops.mjs | 5++++-
Mscripts/archilyzer-ops.test.mjs | 1+
23 files changed, 2098 insertions(+), 46 deletions(-)

diff --git a/common/components/PostModal.test.ts b/common/components/PostModal.test.ts @@ -55,6 +55,20 @@ test("without media or a shot (a deleted post's record): nothing at all", () => assert.equal(render({ ...BASE, state: "deleted" }), ""); }); +test("an article: its title and a link to its markdown; a failed one says how, with no link", () => { + const html = render({ + ...BASE, + shot: { src: "/s.png" }, + article: { state: "captured", title: "A Long <Read>", md: { src: "/capture/demo/1/article.md" } }, + }); + assert.match(html, /data-post-capture-article=""[^>]*>Article: A Long &lt;Read&gt;/); + assert.match(html, /<a href="\/capture\/demo\/1\/article\.md"[^>]*>article\.md<\/a>/); + + const failed = render({ ...BASE, article: { state: "login-wall" } }); + assert.match(failed, /Article: untitled \(login-wall\)/); + assert.doesNotMatch(failed, /article\.md/); +}); + test("media kinds by extension", () => { assert.equal(captureMediaKind("a.JPG"), "image"); assert.equal(captureMediaKind("a.webp"), "image"); diff --git a/common/components/PostModal.tsx b/common/components/PostModal.tsx @@ -46,6 +46,9 @@ export type PostCaptureView = { sensitive?: boolean; shot?: { src: string }; media: { src: string; name: string }[]; + // The X Article the post links to, when the capture opened it: its title, + // how that went, and its markdown when saved. + article?: { state: string; title?: string; md?: { src: string } }; }; export type PostModalProps = { @@ -265,16 +268,31 @@ export function captureMediaKind(name: string): "image" | "video" | "other" { return "other"; } -// The captured screenshot and media of one post. Nothing when the capture -// holds neither (a deleted post's record, say). +// The captured screenshot and media of one post, and a line for its article. +// Nothing when the capture holds none of them (a deleted post's record, say). export function PostCapturePanel({ capture }: { capture: PostCaptureView }) { - if (!capture.shot && capture.media.length === 0) return null; + const { article } = capture; + if (!capture.shot && capture.media.length === 0 && !article) return null; return ( <section data-post-capture="" className="mt-3 flex flex-col gap-2"> <h3 className="text-xs uppercase tracking-wide text-muted-foreground"> Captured {formatWhen(capture.capturedAt)} {capture.sensitive ? " · behind a sensitive-media cover" : ""} </h3> + {article && ( + <p data-post-capture-article="" className="text-xs text-muted-foreground"> + Article: {article.title ?? "untitled"} + {article.state !== "captured" ? ` (${article.state})` : ""} + {article.md && ( + <> + {" · "} + <a href={article.md.src} className="text-primary underline"> + article.md + </a> + </> + )} + </p> + )} {capture.shot && ( <img src={capture.shot.src} diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts @@ -14,7 +14,7 @@ process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); const { getPaths } = await import("../lib/paths"); const { registerSocialFetcher } = await import("../social/fetchers"); -const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server"); +const { readPostAvailability, writePostAvailability, writePosts } = await import("../lib/posts-server"); const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts"); type Input = import("../social/fetchers").PostCaptureInput; type Result = import("../social/fetchers").PostCaptureResult; @@ -36,6 +36,19 @@ registerSocialFetcher({ }, }); registerSocialFetcher({ + id: "test-capture-x", + label: "Test capture (X)", + platform: "twitter", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), + captureByIds: async (input) => { + calls.push(input); + return answer; + }, +}); +registerSocialFetcher({ id: "test-no-capture", label: "No-capture fetcher", platform: "bluesky", @@ -143,3 +156,64 @@ test("a run stopped by the source fails the job; one cancelled by the operator d const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal }); assert.equal(cancelled.ok, true); }); + +test("the articles: on by default, refused alone only when not asked for by name, archived text passed (X)", async () => { + const root = path.join(channelsDir, "demo-x"); + await mkdir(root, { recursive: true }); + await writeFile( + path.join(root, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + platform: "twitter", + postFetcher: "test-capture-x", + socialHandle: "example_user", + name: "Example (X)", + url: "https://x.com/example_user", + }), + ); + const post = (id: string, text: string, links: string[] = []) => ({ + id, + slug: `demo-x/${id}`, + channelSlug: "demo-x", + author: "example_user", + createdAt: "2026-01-02T03:04:05.000Z", + uploadDate: "20260102", + text, + url: `https://x.com/example_user/status/${id}`, + platform: "twitter" as const, + isReply: false, + isRepost: false, + links, + }); + await writePosts(root, [ + post("111", "https://x.com/i/article/999"), + post("222", "a plain post", ["https://example.com/page"]), + post("333", "not asked for"), + ]); + answer = { outcomes: [] }; + + calls = []; + assert.equal((await run("demo-x", ["111", "222"])).ok, true); + assert.equal(calls[0].articles, undefined); + const archived = calls[0].archived ?? new Map(); + assert.deepEqual([...archived.keys()].sort(), ["111", "222"]); + assert.deepEqual(archived.get("111"), { text: "https://x.com/i/article/999", links: [] }); + assert.deepEqual(archived.get("222"), { text: "a plain post", links: ["https://example.com/page"] }); + + calls = []; + assert.equal((await run("demo-x", ["111"], { articles: false })).ok, true); + assert.equal(calls[0].articles, false); + assert.equal(calls[0].archived, undefined); + + // Both halves off: refused, unless the articles are asked for by name. + calls = []; + assert.equal((await run("demo-x", ["111"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE); + assert.equal( + (await run("demo-x", ["111"], { shots: false, media: false, articles: false })).error, + NOTHING_TO_CAPTURE, + ); + assert.equal(calls.length, 0); + assert.equal((await run("demo-x", ["111"], { shots: false, media: false, articles: true })).ok, true); + assert.equal(calls[0].articles, true); +}); diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts @@ -10,7 +10,9 @@ // nothing about the post). // // Only ids already in the channel's posts archive are captured: a capture is -// of the archive, and an id from somewhere else is refused by name. +// of the archive, and an id from somewhere else is refused by name. Each id's +// archived text and links go to the fetcher with it, so a post that is an X +// Article's link is known as one before its page is opened. import path from "node:path"; import { readChannelConfig } from "./channels"; @@ -24,6 +26,7 @@ import { } from "../lib/cookiePolicy"; import { mergePostAvailability, + readAllPosts, readPostAvailability, readSeenPostIds, writePostAvailability, @@ -54,6 +57,8 @@ export type CapturePostsOptions = { media?: boolean; // Capture again what is already captured. force?: boolean; + // Also open and save the X Article a post links to. Defaults to true. + articles?: boolean; onLog?: (line: string) => void; signal?: AbortSignal; // The job's drain: stop between posts. @@ -79,7 +84,17 @@ export function capturePostsProblem( } export const NOTHING_TO_CAPTURE = - "Nothing to capture: both the screenshot and the media are turned off."; + 'Nothing to capture: both the screenshot and the media are turned off (to read only the articles, ask for "articles": true).'; + +// Both halves off is nothing to do — unless the articles were asked for by +// name: their default alone does not make a run. +export function nothingToCapture(o: { + shots?: boolean; + media?: boolean; + articles?: boolean; +}): boolean { + return o.shots === false && o.media === false && o.articles !== true; +} // The ids not in the channel's archive, or null when every one is. export function strayCaptureIds( @@ -103,7 +118,7 @@ export async function capturePosts( const channelRoot = path.join(paths.channelsDir, slug); const ids = [...new Set(opts.ids)]; - if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE); + if (nothingToCapture(opts)) return fail(NOTHING_TO_CAPTURE); if (ids.length === 0) return fail("No post ids to capture."); const config = await readChannelConfig(paths, slug); @@ -121,6 +136,16 @@ export async function capturePosts( const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot)); if (stray) return fail(strayIdsRefusal(slug, stray)); + // The archived text of each id, for the article links in it (X only). + let archived: Map<string, { text: string; links: string[] }> | undefined; + if (opts.articles !== false && fetcher.platform === "twitter") { + const wanted = new Set(ids); + archived = new Map(); + for (const p of await readAllPosts(channelRoot)) { + if (wanted.has(p.id)) archived.set(p.id, { text: p.text, links: p.links ?? [] }); + } + } + const policy = resolveCookiePolicy(settings, config); const xLogin = fetcher.platform === "twitter" @@ -132,6 +157,7 @@ export async function capturePosts( [ opts.shots === false ? "" : " screenshot", opts.media === false ? "" : " media", + opts.articles === false || fetcher.platform !== "twitter" ? "" : " articles", ].join("") + (opts.force ? ", again where already captured" : "") + ".", @@ -147,6 +173,8 @@ export async function capturePosts( shots: opts.shots, media: opts.media, force: opts.force, + articles: opts.articles, + archived, cookies: alwaysCookies(policy), cookieSource: xLogin?.source, browserCookies: xLogin?.browserSpec, diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -523,10 +523,11 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: true, queueKeyStrategy: "platform", }, - // A screenshot and the attached media of specific archived posts, into the - // channel's `posts-media/` (controller/capturePosts.ts). On the PLATFORM - // queue, like fetch-posts: each post is a page load and a download against - // the same source, so it serialises with the fetch and shares its backoff. + // A screenshot and the attached media of specific archived posts (and the X + // Article a post links to), into the channel's `posts-media/` + // (controller/capturePosts.ts). On the PLATFORM queue, like fetch-posts: + // each post is a page load and a download against the same source, so it + // serialises with the fetch and shares its backoff. // Drainable (it stops between posts) and replayable (the ids are in the // spec; posts already captured are skipped on a re-run). "capture-posts": { diff --git a/common/social/__fixtures__/x-article.html b/common/social/__fixtures__/x-article.html @@ -0,0 +1 @@ +<div data-testid="twitterArticleReadView" class="css-175oi2r" data-archilyzer-article=""><div class="css-175oi2r"><div data-testid="tweetPhoto" class="css-175oi2r"><img alt="Image" draggable="true" src="https://pbs.twimg.com/media/CoverAbc?format=jpg&amp;name=small" class="css-9pa8cd"></div><div data-testid="twitter-article-title" dir="auto" class="css-146c3p1"><span class="css-1jxf684">A Worked Example &amp; Its Notes</span></div><div class="css-175oi2r"><div data-testid="UserAvatar-Container-demo_author" class="css-175oi2r"><img alt="" src="https://pbs.twimg.com/profile_images/1/avatar_normal.jpg"></div><div data-testid="User-Name" class="css-175oi2r"><div class="css-175oi2r"><a href="/demo_author" role="link"><div dir="ltr"><span>Demo Author</span></div></a></div><div class="css-175oi2r"><a href="/demo_author" role="link" tabindex="-1"><div dir="ltr"><span>@demo_author</span></div></a><div aria-hidden="true" dir="ltr"><span>·</span></div><a href="/demo_author/article/1234567890" role="link"><time datetime="2026-01-02T03:04:05.000Z">Jan 2</time></a></div></div><div class="css-175oi2r"><button role="button" type="button"><span>Follow</span></button></div></div></div><div data-testid="twitterArticleRichTextView" class="css-175oi2r"><div data-testid="longformRichTextComponent" class="css-175oi2r"><div class="DraftEditor-root"><div class="DraftEditor-editorContainer"><div aria-describedby="placeholder" class="public-DraftEditor-content" contenteditable="false" role="textbox" spellcheck="false"><div data-contents="true"><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="a-0-0"><div data-offset-key="a-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="a-0-0"><span data-text="true">The first paragraph, with a </span></span><a href="https://example.com/source" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="a-1-0"><span data-text="true">linked source</span></span></a><span data-offset-key="a-2-0"><span data-text="true"> and an&nbsp;emoji </span></span><img alt="🙂" draggable="false" src="https://abs-0.twimg.com/emoji/v2/svg/1f642.svg" class="r-4qtqp9"><span data-offset-key="a-3-0"><span data-text="true">.</span></span></div></div><h2 class="longform-header-two" data-block="true" data-editor="ed1" data-offset-key="b-0-0"><div data-offset-key="b-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="b-0-0"><span data-text="true">A Section Heading</span></span></div></h2><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="c-0-0"><div data-offset-key="c-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="c-0-0"><span data-text="true">A second paragraph that runs</span></span><br data-text="true"><span data-offset-key="c-0-1"><span data-text="true">onto a second line.</span></span></div></div><blockquote class="longform-blockquote" data-block="true" data-editor="ed1" data-offset-key="d-0-0"><div data-offset-key="d-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="d-0-0"><span data-text="true">A quoted passage, set apart.</span></span></div></blockquote><ul class="public-DraftStyleDefault-ul" data-offset-key="e-0-0"><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-reset public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="e-0-0"><div data-offset-key="e-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="e-0-0"><span data-text="true">The first point</span></span></div></li><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="f-0-0"><div data-offset-key="f-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="f-0-0"><span data-text="true"># The second point</span></span></div></li></ul><section data-block="true" data-editor="ed1" data-offset-key="g-0-0"><div class="longform-media" contenteditable="false"><a href="/demo_author/article/1234567890/media/555" role="link"><div data-testid="tweetPhoto"><img alt="A chart of the numbers" draggable="true" src="https://pbs.twimg.com/media/BodyImg1?format=png&amp;name=small"></div></a></div></section><section data-block="true" data-editor="ed1" data-offset-key="h-0-0"><div contenteditable="false"><div data-testid="simpleTweet" class="css-175oi2r"><article aria-labelledby="id1" role="article" tabindex="-1" data-testid="tweet"><div data-testid="User-Name"><a href="/other_user"><span>Other User</span></a><a href="/other_user"><span>@other_user</span></a><a href="/other_user/status/4444444444" role="link"><time datetime="2025-12-01T00:00:00.000Z">Dec 1</time></a></div><div data-testid="tweetText" dir="auto" lang="en"><span>An embedded post's own words.</span></div><div role="group" aria-label="12 replies"><button data-testid="reply"><span>12</span></button></div><a href="/other_user/status/4444444444/analytics">Views</a></article></div></div></section><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="i-0-0"><div data-offset-key="i-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><a href="https://example.com/further-reading" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="i-0-0"><span data-text="true">Further reading</span></span></a></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="j-0-0"><div data-offset-key="j-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="j-0-0"><br data-text="true"></span></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="k-0-0"><div data-offset-key="k-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="k-0-0"><span data-text="true">The last word: 3 &lt; 4.</span></span></div></div></div></div></div></div></div></div><div role="group" aria-label="40 replies, 7 reposts, 300 likes" class="css-175oi2r"><button data-testid="reply" type="button"><span>40</span></button><button data-testid="like" type="button"><span>300</span></button></div><div class="css-175oi2r"><span>A trailing note outside the body.</span></div></div> diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts @@ -182,6 +182,13 @@ export type PostCaptureInput = Pick< media?: boolean; // Capture again what is already captured. force?: boolean; + // Also capture the long-form article a post links to (X Articles), where it + // links to one. Defaults to true. + articles?: boolean; + // What the posts archive holds for each id — its text and expanded links — + // for the links a capture follows (an X Article's). Ids absent are read + // from the page alone. + archived?: ReadonlyMap<string, { text: string; links?: ReadonlyArray<string> }>; // A soft stop: no new post is started once it fires, and the one in hand // finishes (`signal` cancels outright). drain?: AbortSignal; diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts @@ -48,6 +48,19 @@ export type PageLike = { fullPage?: boolean; clip?: { x: number; y: number; width: number; height: number }; }) => Promise<Uint8Array>; + // The page's own request context (the profile's cookies): an X Article's + // inline images are fetched through it (xArticleCapture.ts). + request?: { + get: ( + url: string, + opts?: { timeout?: number; failOnStatusCode?: boolean }, + ) => Promise<{ + ok: () => boolean; + status: () => number; + headers: () => Record<string, string>; + body: () => Promise<Uint8Array>; + }>; + }; }; export type BrowserContextLike = { diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts @@ -11,10 +11,12 @@ import os from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { + articleImageFilename, CAPTURE_FILENAME, captureAvailability, captureWork, describeCapturedFile, + isArticleFile, listCapturedMediaFiles, POSTS_MEDIA_DIRNAME, postCaptureDir, @@ -22,6 +24,7 @@ import { readPostCapture, SHOT_FILENAME, writePostCapture, + type ArticleCaptureRecord, type PostCaptureRecord, } from "./postCapture"; @@ -75,6 +78,12 @@ test("capture.json: each file's sha256 and byte size, the URLs, round-tripped", assert.deepEqual(await readPostCapture(dir), rec); // The record and the shot are not media. assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]); + + // Nor is the article half, whatever a later gallery-dl run finds beside it. + for (const name of ["article.json", "article.md", "article.png", "article.html", "article-img-1.jpg"]) { + await writeFile(path.join(dir, name), "x"); + } + assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]); }); test("capture.json: absent, unparseable or of another shape reads as no capture", async () => { @@ -97,27 +106,65 @@ test("availability: only what the page said about the post — never a verdict f test("what a capture owes: nothing for a settled post, only the missing half otherwise, everything when forced", () => { const all = { shots: true, media: true, force: false }; - assert.deepEqual(captureWork(null, all), { shot: true, media: true }); - assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false }); + assert.deepEqual(captureWork(null, all), { shot: true, media: true, article: false }); + assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false, article: false }); const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" }); - assert.deepEqual(captureWork(done, all), { shot: false, media: false }); - assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false }); + assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: false }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false, article: false }); // The media failed (or was never asked for): only the media is owed. - assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true }); - assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true, article: false }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true, article: false }); // A shot that failed is owed again. - assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false }); + assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false, article: false }); // Deleted is settled: asking again is a request for the same answer. assert.deepEqual(captureWork(record({ state: "deleted", mediaState: "skipped" }), all), { shot: false, media: false, + article: false, }); // force re-takes whatever was asked for. - assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true }); + assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true, article: false }); assert.deepEqual(captureWork(record({ state: "deleted" }), { shots: false, media: true, force: true }), { shot: false, media: true, + article: false, + }); +}); + +test("what a capture owes the article half: only when asked, settled once captured, deleted or walled", () => { + const all = { shots: true, media: true, force: false, articles: true }; + const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" }); + const article = (state: ArticleCaptureRecord["state"]): ArticleCaptureRecord => ({ + articleId: "9", + url: "https://x.com/i/article/9", + capturedAt: "2026-01-02T03:04:05.000Z", + state, + blocks: 0, + files: [], }); + assert.equal(captureWork(null, all).article, true); + assert.equal(captureWork(null, { ...all, articles: false }).article, false); + // Not asked for by name: the older callers' shape owes no article. + assert.equal(captureWork(null, { shots: true, media: true, force: false }).article, false); + // A post shot before articles were captured owes only its article. + assert.deepEqual(captureWork(done, all), { shot: false, media: false, article: true }); + for (const state of ["captured", "deleted", "unavailable"] as const) { + assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, false, state); + } + for (const state of ["error", "login-wall"] as const) { + assert.equal(captureWork(record({ ...done, article: article(state) }), all).article, true, state); + } + // force re-reads a captured article; a deleted post owes nothing. + assert.equal(captureWork(record({ ...done, article: article("captured") }), { ...all, force: true }).article, true); + assert.equal(captureWork(record({ state: "deleted" }), all).article, false); +}); + +test("article files are named by the layout, images numbered from 1", () => { + assert.equal(articleImageFilename(1, "jpg"), "article-img-1.jpg"); + assert.ok(isArticleFile("article.md")); + assert.ok(isArticleFile("article-img-12.png")); + assert.ok(!isArticleFile("123_1.jpg")); + assert.ok(!isArticleFile("articles.txt")); }); // THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON diff --git a/common/social/postCapture.ts b/common/social/postCapture.ts @@ -4,6 +4,10 @@ // channels/<slug>/posts-media/<post id>/shot.png — the post, as rendered // channels/<slug>/posts-media/<post id>/<media…> — its attached media // channels/<slug>/posts-media/<post id>/capture.json — this module's record +// channels/<slug>/posts-media/<post id>/article.* — the X Article the post +// links to, when it is one: article.json (its blocks), article.md, +// article.png, article.html (the root as X served it) and +// article-img-<n>.<ext> (its inline images) — xArticleCapture.ts // // A directory per post, unlike the posts themselves (month-sharded JSONL): a // capture is a handful of files, and only for the posts someone asked for. It @@ -26,6 +30,20 @@ import type { PostAvailability } from "../lib/posts"; export const POSTS_MEDIA_DIRNAME = "posts-media"; export const CAPTURE_FILENAME = "capture.json"; export const SHOT_FILENAME = "shot.png"; +export const ARTICLE_JSON_FILENAME = "article.json"; +export const ARTICLE_MD_FILENAME = "article.md"; +export const ARTICLE_SHOT_FILENAME = "article.png"; +export const ARTICLE_HTML_FILENAME = "article.html"; + +// The n-th (1-based) inline image of an article. +export function articleImageFilename(n: number, ext: string): string { + return `article-img-${n}.${ext}`; +} + +// A file of the article half, never a medium of the post. +export function isArticleFile(name: string): boolean { + return /^article(\.|-img-)/.test(name); +} export function postsMediaDir(channelRoot: string): string { return path.join(channelRoot, POSTS_MEDIA_DIRNAME); @@ -88,6 +106,28 @@ export type CapturedFile = { // asked for or not attempted ("skipped"), or failed ("error"). export type CaptureMediaState = "ok" | "none" | "skipped" | "error"; +// The article half: an X Article the post links to, opened and read +// (xArticleCapture.ts). `state` is the article page's, in the post's terms: a +// deleted or walled article is settled, an error is owed again, a login wall +// stopped the run. +export type ArticleCaptureRecord = { + articleId: string; + url: string; + capturedAt: string; + state: PostCaptureState; + title?: string; + // The blocks read, for a captured article (article.json holds them). + blocks: number; + // How the body was read: by X's markers, or the root's text split at block + // elements because the markers were missing. + extraction?: "structured" | "fallback"; + // article.json, article.md, article.png, article.html, each image. + files: CapturedFile[]; + // article.png stops at a height cap; the article ran longer. + trimmed?: boolean; + error?: string; +}; + export type PostCaptureRecord = { version: 1; id: string; @@ -102,6 +142,8 @@ export type PostCaptureRecord = { shot?: CapturedFile; mediaState: CaptureMediaState; media: CapturedFile[]; + // The X Article the post links to, when one was captured or tried. + article?: ArticleCaptureRecord; error?: string; }; @@ -128,8 +170,8 @@ export async function describeCapturedFile( return { name, ...digest, ...(url ? { url } : {}) }; } -// The media files in a post's directory: everything but the shot, the record -// and a download's leftovers. +// The media files in a post's directory: everything but the shot, the record, +// the article half and a download's leftovers. export async function listCapturedMediaFiles(dir: string): Promise<string[]> { let names: string[]; try { @@ -142,6 +184,7 @@ export async function listCapturedMediaFiles(dir: string): Promise<string[]> { (n) => n !== SHOT_FILENAME && n !== CAPTURE_FILENAME && + !isArticleFile(n) && !n.endsWith(".part") && !n.startsWith("."), ) @@ -171,19 +214,31 @@ export async function writePostCapture( // A deleted post is settled: the platform said so, and asking again costs a // request for the same answer (`force` asks anyway). A shot on disk is kept; a // media download that did not finish ("error", or never attempted) is owed. +// The article half is owed only if the post links to one (the caller knows); +// one captured, deleted or walled is settled, one that failed is owed. export function captureWork( existing: PostCaptureRecord | null, - wanted: { shots: boolean; media: boolean; force: boolean }, -): { shot: boolean; media: boolean } { + wanted: { shots: boolean; media: boolean; force: boolean; articles?: boolean }, +): { shot: boolean; media: boolean; article: boolean } { + const articles = wanted.articles ?? false; if (wanted.force || !existing) { - return { shot: wanted.shots, media: wanted.media }; + return { shot: wanted.shots, media: wanted.media, article: articles }; } - if (existing.state === "deleted") return { shot: false, media: false }; + if (existing.state === "deleted") return { shot: false, media: false, article: false }; return { shot: wanted.shots && !existing.shot, media: wanted.media && existing.mediaState !== "ok" && existing.mediaState !== "none", + article: articles && !articleSettled(existing.article), }; } + +export function articleSettled(article: ArticleCaptureRecord | undefined): boolean { + return ( + article?.state === "captured" || + article?.state === "deleted" || + article?.state === "unavailable" + ); +} diff --git a/common/social/xArticle.test.ts b/common/social/xArticle.test.ts @@ -0,0 +1,204 @@ +// X Articles, the pure half: the article link in a post's text or on its card, +// the article root's HTML read into blocks (against a written fixture, never +// x.com), and the markdown written from them. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticle.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + decodeEntities, + extractXArticle, + findXArticleLink, + parseHtml, + textOf, + xArticleLinkFromArchive, + xArticleMarkdown, +} from "./xArticle"; +import { xArticleLinkFromCard } from "./xArticleCapture"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8"); + +// --- the link ---------------------------------------------------------------- + +test("link: an article post's archived text, its expanded links, or nothing", () => { + assert.deepEqual(xArticleLinkFromArchive({ text: "https://x.com/i/article/1987654321" }), { + articleId: "1987654321", + url: "https://x.com/i/article/1987654321", + }); + assert.deepEqual( + xArticleLinkFromArchive({ text: "Read it: https://t.co/abc", links: ["https://twitter.com/i/article/42"] }), + { articleId: "42", url: "https://x.com/i/article/42" }, + ); + assert.deepEqual(xArticleLinkFromArchive({ text: "x.com/demo_author/article/77 is up" }), { + articleId: "77", + url: "https://x.com/demo_author/article/77", + }); + for (const text of [ + "a plain post", + "https://x.com/demo_author/status/123", + "https://example.com/i/article/5", + "https://x.com/i/articles", + ]) { + assert.equal(xArticleLinkFromArchive({ text }), null, text); + } + assert.equal(xArticleLinkFromArchive(undefined), null); +}); + +test("link: a card's hrefs, relative as the page has them", () => { + assert.deepEqual(xArticleLinkFromCard(["/demo_author", "/demo_author/article/1234567890"]), { + articleId: "1234567890", + url: "https://x.com/demo_author/article/1234567890", + }); + assert.deepEqual(xArticleLinkFromCard(["/i/article/99/media/1"]), { + articleId: "99", + url: "https://x.com/i/article/99", + }); + assert.equal(xArticleLinkFromCard(["/demo_author/status/1/photo/1"]), null); + assert.equal(xArticleLinkFromCard([]), null); + assert.equal(xArticleLinkFromCard(undefined), null); + assert.equal(findXArticleLink([null, undefined, ""]), null); +}); + +// --- the HTML reader --------------------------------------------------------- + +test("html: elements, attributes, void and self-closing tags, entities, comments, raw text", () => { + const root = parseHtml( + '<!-- c --><div class="a" data-x=\'q\' hidden><img src="s?a=1&amp;b=2"><br/>a &lt;b&gt; &#39;c&#x27; &nbsp;d' + + "<script>if (a < b) {}</script><span>e</span></p></div>tail", + ); + const div = root.children[0] as { tag: string; attrs: Record<string, string>; children: unknown[] }; + assert.equal(div.tag, "div"); + assert.deepEqual(div.attrs, { class: "a", "data-x": "q", hidden: "" }); + assert.equal((div.children[0] as { attrs: { src: string } }).attrs.src, "s?a=1&b=2"); + // A stray end tag is ignored; the text after the div is the root's. + assert.equal(root.children[1], "tail"); + assert.equal(textOf(div as never), "a <b> 'c' de"); + assert.equal(decodeEntities("&amp;&unknown;&#128578;"), "&&unknown;🙂"); +}); + +test("html: text reads as innerText would — line breaks at <br> and blocks, script and svg silent", () => { + const root = parseHtml("<div><p>one <b>two</b></p><p>three<br>four</p><svg><text>no</text></svg><script>x</script></div>"); + assert.equal(textOf(root), "one two\nthree\nfour"); +}); + +// --- the extractor ------------------------------------------------------------- + +test("extract: title, byline and the body's blocks in reading order, from the written fixture", () => { + const a = extractXArticle(FIXTURE); + assert.equal(a.extraction, "structured"); + assert.equal(a.title, "A Worked Example & Its Notes"); + assert.equal(a.author, "Demo Author"); + assert.equal(a.handle, "@demo_author"); + assert.equal(a.publishedAt, "2026-01-02T03:04:05.000Z"); + assert.deepEqual(a.blocks, [ + // The cover, above the body. + { type: "image", src: "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" }, + { type: "paragraph", text: "The first paragraph, with a linked source and an emoji 🙂." }, + { type: "link", text: "linked source", href: "https://example.com/source" }, + { type: "heading", text: "A Section Heading" }, + { type: "paragraph", text: "A second paragraph that runs\nonto a second line." }, + { type: "quote", text: "A quoted passage, set apart." }, + { type: "list-item", text: "The first point" }, + { type: "list-item", text: "# The second point" }, + { + type: "image", + src: "https://pbs.twimg.com/media/BodyImg1?format=png&name=small", + text: "A chart of the numbers", + }, + { + type: "embedded-post", + text: "An embedded post's own words.", + href: "https://x.com/other_user/status/4444444444", + }, + // A paragraph that is only a link is the link. + { type: "link", text: "Further reading", href: "https://example.com/further-reading" }, + { type: "paragraph", text: "The last word: 3 < 4." }, + ]); +}); + +test("extract: no body marker — the root's text split at block elements, and it says so", () => { + const a = extractXArticle( + '<article><h1>Plain Title</h1><div><p>One</p><p>Two <a href="/demo_author">a handle</a></p>' + + '<figure><img src="https://pbs.twimg.com/media/Fig?format=webp"></figure><ul><li>Item</li></ul>' + + '<blockquote>Said</blockquote></div><div role="group">40 likes</div></article>', + ); + assert.equal(a.extraction, "fallback"); + assert.equal(a.title, "Plain Title"); + assert.deepEqual(a.blocks, [ + { type: "paragraph", text: "One" }, + { type: "paragraph", text: "Two a handle" }, + { type: "link", text: "a handle", href: "https://x.com/demo_author" }, + { type: "image", src: "https://pbs.twimg.com/media/Fig?format=webp" }, + { type: "list-item", text: "Item" }, + { type: "quote", text: "Said" }, + ]); +}); + +test("extract: an empty root is an untitled article with no blocks", () => { + assert.deepEqual(extractXArticle(""), { extraction: "fallback", blocks: [] }); +}); + +// --- markdown ---------------------------------------------------------------- + +test("markdown: title, byline, the URL, then each block — saved images linked locally, markdown in text escaped", () => { + const a = extractXArticle(FIXTURE); + const blocks = a.blocks.map((b) => + b.src?.includes("BodyImg1") ? { ...b, file: "article-img-2.png" } : b, + ); + const md = xArticleMarkdown({ ...a, blocks, url: "https://x.com/i/article/1234567890" }); + assert.equal( + md, + [ + "# A Worked Example & Its Notes", + "", + "By Demo Author @demo_author · 2026-01-02T03:04:05.000Z", + "", + "<https://x.com/i/article/1234567890>", + "", + "![](<https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small>)", + "", + "The first paragraph, with a linked source and an emoji 🙂.", + "", + "[linked source](<https://example.com/source>)", + "", + "## A Section Heading", + "", + "A second paragraph that runs", + "onto a second line.", + "", + "> A quoted passage, set apart.", + "", + "- The first point", + "- \\# The second point", + "", + "![A chart of the numbers](<article-img-2.png>)", + "", + "> Embedded post: <https://x.com/other_user/status/4444444444>", + ">", + "> An embedded post's own words.", + "", + "[Further reading](<https://example.com/further-reading>)", + "", + "The last word: 3 < 4.", + "", + ].join("\n"), + ); +}); + +test("markdown: an untitled article, no byline, a paragraph that would read as syntax", () => { + const md = xArticleMarkdown({ + extraction: "fallback", + url: "https://x.com/i/article/1", + blocks: [{ type: "paragraph", text: "1. not a list\n> not a quote" }, { type: "link", href: "https://example.com/a b" }], + }); + assert.equal( + md, + "# Untitled article\n\n<https://x.com/i/article/1>\n\n1\\. not a list\n\\> not a quote\n\n" + + "[https://example.com/a b](<https://example.com/a%20b>)\n", + ); +}); diff --git a/common/social/xArticle.ts b/common/social/xArticle.ts @@ -0,0 +1,547 @@ +// X Articles (long-form posts), the pure half: finding a post's article link, +// reading the article's HTML into blocks, and writing those blocks as +// markdown. No browser, no network, no fs — xArticleCapture.ts loads the page +// and hands this module the article root's outerHTML. +// +// An article post archives as text that is only its link +// (`https://x.com/i/article/<id>`): gallery-dl cannot read an article's body, +// and the post's card shows a title and a preview. The article itself is a +// page of its own, which the capture opens. +// +// WHY HTML AND NOT A WALK IN THE PAGE. The browser hands back the root's +// outerHTML once, and everything after that is parsed and read here, so the +// whole extractor runs in tests against a saved HTML fixture, and the HTML is +// kept beside the capture (`article.html`) so a better reading later costs no +// second visit to X. +// +// NOT VERIFIED AGAINST LIVE X. The markers below (data-testid names, the +// Draft.js block classes) are X's as of this writing; every one is optional, +// and an article whose body marker is missing is read by the fallback — the +// root's text split at block elements — and says so (`extraction`). + +// --- the link ------------------------------------------------------------------ + +export type XArticleLink = { + // The id X's article URL carries. + articleId: string; + url: string; +}; + +// `x.com/i/article/<id>`, or `x.com/<handle>/article/<id>` (the form a card +// links to), absolute or as a path. +const ARTICLE_URL_RE = + /(?:https?:\/\/)?(?:www\.|mobile\.)?(?:x|twitter)\.com\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/i; +const ARTICLE_PATH_RE = /^\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/; + +function linkFrom(owner: string, articleId: string): XArticleLink { + return { articleId, url: `https://x.com/${owner}/article/${articleId}` }; +} + +// The first article link in these strings (a post's text, its expanded links, +// a card's hrefs), or null. +export function findXArticleLink( + candidates: ReadonlyArray<string | null | undefined>, +): XArticleLink | null { + for (const c of candidates) { + if (!c) continue; + const m = ARTICLE_URL_RE.exec(c) ?? ARTICLE_PATH_RE.exec(c.trim()); + if (m) return linkFrom(m[1].toLowerCase() === "i" ? "i" : m[1], m[2]); + } + return null; +} + +// What the posts archive holds for a post, as far as its links go. +export type ArchivedPostText = { text: string; links?: ReadonlyArray<string> }; + +export function xArticleLinkFromArchive( + archived: ArchivedPostText | undefined, +): XArticleLink | null { + if (!archived) return null; + return findXArticleLink([archived.text, ...(archived.links ?? [])]); +} + +// --- a small HTML reader ------------------------------------------------------- +// +// For HTML a browser serialised (outerHTML): attributes are double-quoted, void +// elements are unclosed, text escapes only & < > and nbsp. It tolerates more +// (single quotes, bare values, stray end tags), but it is not a general parser. + +export type HtmlElement = { + tag: string; + attrs: Record<string, string>; + children: HtmlNode[]; +}; +export type HtmlNode = HtmlElement | string; + +const VOID_TAGS = new Set([ + "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", + "param", "source", "track", "wbr", +]); +const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]); + +const NAMED_ENTITIES: Record<string, string> = { + amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0", + hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“", + rdquo: "”", copy: "©", reg: "®", trade: "™", +}; + +export function decodeEntities(s: string): string { + return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => { + if (name[0] === "#") { + const code = + name[1] === "x" || name[1] === "X" + ? parseInt(name.slice(2), 16) + : parseInt(name.slice(1), 10); + return Number.isFinite(code) && code > 0 && code <= 0x10ffff + ? String.fromCodePoint(code) + : whole; + } + return NAMED_ENTITIES[name.toLowerCase()] ?? whole; + }); +} + +const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y; +const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y; + +// The document's top-level nodes, under a synthetic root element. +export function parseHtml(html: string): HtmlElement { + const root: HtmlElement = { tag: "#root", attrs: {}, children: [] }; + const stack: HtmlElement[] = [root]; + const top = () => stack[stack.length - 1]; + let i = 0; + while (i < html.length) { + const lt = html.indexOf("<", i); + if (lt < 0) { + top().children.push(decodeEntities(html.slice(i))); + break; + } + if (lt > i) top().children.push(decodeEntities(html.slice(i, lt))); + i = lt; + if (html.startsWith("<!--", i)) { + const end = html.indexOf("-->", i + 4); + i = end < 0 ? html.length : end + 3; + continue; + } + if (html[i + 1] === "!" || html[i + 1] === "?") { + const end = html.indexOf(">", i); + i = end < 0 ? html.length : end + 1; + continue; + } + if (html[i + 1] === "/") { + const end = html.indexOf(">", i); + const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase(); + i = end < 0 ? html.length : end + 1; + // Close up to the matching open element; a stray end tag is ignored. + for (let k = stack.length - 1; k > 0; k--) { + if (stack[k].tag === name) { + stack.length = k; + break; + } + } + continue; + } + TAG_NAME_RE.lastIndex = i + 1; + const nameMatch = TAG_NAME_RE.exec(html); + if (!nameMatch) { + // A "<" that opens no tag is text. + top().children.push("<"); + i++; + continue; + } + const tag = nameMatch[0].toLowerCase(); + let j = TAG_NAME_RE.lastIndex; + const attrs: Record<string, string> = {}; + for (;;) { + ATTR_RE.lastIndex = j; + const a = ATTR_RE.exec(html); + if (!a || a[0].length === 0) break; + attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? ""); + j = ATTR_RE.lastIndex; + } + const close = html.indexOf(">", j); + const selfClosing = close > 0 && html[close - 1] === "/"; + i = close < 0 ? html.length : close + 1; + const el: HtmlElement = { tag, attrs, children: [] }; + top().children.push(el); + if (RAW_TEXT_TAGS.has(tag)) { + const end = html.toLowerCase().indexOf(`</${tag}`, i); + const stop = end < 0 ? html.length : end; + if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop))); + const gt = end < 0 ? -1 : html.indexOf(">", end); + i = gt < 0 ? html.length : gt + 1; + continue; + } + if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el); + } + return root; +} + +// --- reading the article ------------------------------------------------------ + +export type XArticleBlockType = + | "heading" + | "paragraph" + | "quote" + | "list-item" + | "image" + | "embedded-post" + | "link"; + +export type XArticleBlock = { + type: XArticleBlockType; + text?: string; + href?: string; + src?: string; + // An image saved beside the capture: its file name there. + file?: string; +}; + +export type XArticleContent = { + title?: string; + author?: string; + handle?: string; + publishedAt?: string; + // "structured": the body marker was found and read block by block; + // "fallback": it was not, and the root's text was split at block elements. + extraction: "structured" | "fallback"; + blocks: XArticleBlock[]; +}; + +const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string"; +const testid = (el: HtmlElement) => el.attrs["data-testid"] ?? ""; +const classes = (el: HtmlElement) => el.attrs["class"] ?? ""; + +function findFirst( + el: HtmlElement, + pred: (e: HtmlElement) => boolean, + skip?: (e: HtmlElement) => boolean, +): HtmlElement | null { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) return c; + if (skip?.(c)) continue; + const hit = findFirst(c, pred, skip); + if (hit) return hit; + } + return null; +} + +function findAll(el: HtmlElement, pred: (e: HtmlElement) => boolean, out: HtmlElement[] = []): HtmlElement[] { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) out.push(c); + findAll(c, pred, out); + } + return out; +} + +const SILENT_TAGS = new Set([ + "script", "style", "svg", "noscript", "template", "button", "input", "select", + "video", "audio", "iframe", "canvas", +]); + +// An element's text as innerText would give it, near enough: `<br>` and block +// boundaries are line breaks (never a blank line), runs of other whitespace are +// one space, an emoji drawn as an image is its alt. +export function textOf(node: HtmlNode): string { + const parts: string[] = []; + const walk = (n: HtmlNode) => { + if (!isEl(n)) { + parts.push(n.replace(/[\s\u00a0]+/g, " ")); + return; + } + if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true") return; + if (n.tag === "br") { + parts.push("\n"); + return; + } + if (n.tag === "img") { + parts.push(emojiText(n)); + return; + } + const block = BLOCK_TAGS.has(n.tag); + if (block) parts.push("\n"); + for (const c of n.children) walk(c); + if (block) parts.push("\n"); + }; + walk(node); + return parts + .join("") + .split("\n") + .map((l) => l.replace(/ +/g, " ").trim()) + .filter(Boolean) + .join("\n"); +} + +const BLOCK_TAGS = new Set([ + "address", "article", "aside", "blockquote", "dd", "details", "div", "dl", + "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", + "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section", + "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", +]); + +// The article's body: X's long-form rich-text view, else the Draft.js content +// it is built on. +const BODY_TESTIDS = new Set(["longformRichTextComponent", "twitterArticleRichTextView"]); +const isBody = (el: HtmlElement) => + BODY_TESTIDS.has(testid(el)) || el.attrs["data-contents"] === "true"; + +const isTitle = (el: HtmlElement) => testid(el) === "twitter-article-title"; +const isByline = (el: HtmlElement) => testid(el) === "User-Name"; +// The chrome around an article that is not its text. +const isChrome = (el: HtmlElement) => + /^(UserAvatar|UserAvatar-Container|reply|retweet|like|bookmark|caret|app-text-transition-container)/.test( + testid(el), + ) || el.attrs["role"] === "group"; + +const EMBED_TESTIDS = new Set(["tweet", "simpleTweet", "quoteTweet"]); +const isEmbeddedPost = (el: HtmlElement) => + EMBED_TESTIDS.has(testid(el)) || + (el.tag === "blockquote" && /\btwitter-tweet\b/.test(classes(el))); + +const STATUS_RE = + /^(?:https?:\/\/(?:www\.|mobile\.)?(?:x|twitter)\.com)?\/([A-Za-z0-9_]{1,15})\/status\/(\d{1,25})/i; + +// An embedded post's status URL: the link around its timestamp, else its first +// status link. +function statusUrlIn(el: HtmlElement): string | undefined { + const links = findAll(el, (e) => e.tag === "a" && STATUS_RE.test(e.attrs.href ?? "")); + const best = links.find((a) => findFirst(a, (e) => e.tag === "time")) ?? links[0]; + const m = best ? STATUS_RE.exec(best.attrs.href) : null; + return m ? `https://x.com/${m[1]}/status/${m[2]}` : undefined; +} + +// An image that is the article's, not an emoji, an avatar or an icon. +export function isContentImageSrc(src: string | undefined): src is string { + if (!src || src.startsWith("data:")) return false; + if (/\/emoji\/|\/hashflags\/|profile_images|profile_banners|\.svg(\?|$)/i.test(src)) return false; + return /^https?:\/\//i.test(src); +} + +// X draws an emoji as an image whose alt is the emoji: its text. +function emojiText(img: HtmlElement): string { + return /\/emoji\//.test(img.attrs.src ?? "") ? (img.attrs.alt ?? "") : ""; +} + +// An href as an absolute URL, or undefined for one that goes nowhere. +function absoluteHref(href: string | undefined): string | undefined { + if (!href) return undefined; + const h = href.trim(); + if (h === "" || h.startsWith("#") || /^(javascript|mailto|data):/i.test(h)) { + return /^mailto:/i.test(h) ? h : undefined; + } + if (/^https?:\/\//i.test(h)) return h; + if (h.startsWith("//")) return `https:${h}`; + if (h.startsWith("/")) return `https://x.com${h}`; + return undefined; +} + +function blockTypeOf(el: HtmlElement): "heading" | "quote" | "list-item" | null { + const c = classes(el); + if (/^h[1-6]$/.test(el.tag) || /\blongform-header-/.test(c)) return "heading"; + if (el.tag === "blockquote" || /\blongform-blockquote\b/.test(c)) return "quote"; + if (el.tag === "li" || /\blongform-(un)?ordered-list-item\b/.test(c)) return "list-item"; + return null; +} + +// The blocks of `body` in reading order. Text outside any typed block gathers +// into a paragraph that ends at the next block boundary; a link inside text is +// kept as a link block after it (a paragraph that is only a link is just the +// link). +function readBlocks(body: HtmlElement, skip: (el: HtmlElement) => boolean): XArticleBlock[] { + const blocks: XArticleBlock[] = []; + let inline: string[] = []; + let links: XArticleBlock[] = []; + + const pushLinks = (from: HtmlElement) => { + for (const a of findAll(from, (e) => e.tag === "a")) { + if (findFirst(a, (e) => e.tag === "img")) continue; + const href = absoluteHref(a.attrs.href); + const text = textOf(a); + if (href) blocks.push({ type: "link", ...(text ? { text } : {}), href }); + } + }; + const flush = () => { + const text = inline + .join("") + .split("\n") + .map((l) => l.replace(/[ \t\u00a0]+/g, " ").trim()) + .filter(Boolean) + .join("\n"); + inline = []; + const onlyLink = links.length === 1 && links[0].text === text; + if (text && !onlyLink) blocks.push({ type: "paragraph", text }); + blocks.push(...links); + links = []; + }; + + const walk = (n: HtmlNode) => { + if (!isEl(n)) { + inline.push(n.replace(/[\s\u00a0]+/g, " ")); + return; + } + if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true" || skip(n)) return; + if (n.tag === "br") { + inline.push("\n"); + return; + } + if (isEmbeddedPost(n)) { + flush(); + const textEl = findFirst(n, (e) => testid(e) === "tweetText"); + const text = textEl ? textOf(textEl) : ""; + const href = statusUrlIn(n); + blocks.push({ type: "embedded-post", ...(text ? { text } : {}), ...(href ? { href } : {}) }); + return; + } + if (n.tag === "img") { + if (!isContentImageSrc(n.attrs.src)) { + inline.push(emojiText(n)); + return; + } + flush(); + const alt = (n.attrs.alt ?? "").trim(); + blocks.push({ type: "image", src: n.attrs.src, ...(alt && alt.toLowerCase() !== "image" ? { text: alt } : {}) }); + return; + } + const typed = blockTypeOf(n); + if (typed) { + flush(); + const text = textOf(n); + if (text) blocks.push({ type: typed, text }); + pushLinks(n); + // An image inside a typed block (rare) is still the article's. + for (const img of findAll(n, (e) => e.tag === "img" && isContentImageSrc(e.attrs.src))) { + blocks.push({ type: "image", src: img.attrs.src }); + } + return; + } + if (n.tag === "a" && !findFirst(n, (e) => e.tag === "img")) { + const href = absoluteHref(n.attrs.href); + const text = textOf(n); + inline.push(text); + if (href) links.push({ type: "link", ...(text ? { text } : {}), href }); + return; + } + const block = BLOCK_TAGS.has(n.tag); + if (block) flush(); + for (const c of n.children) walk(c); + if (block) flush(); + }; + walk(body); + flush(); + return blocks; +} + +// The article root's HTML, read: title, byline, and the body's blocks. +export function extractXArticle(html: string): XArticleContent { + const root = parseHtml(html); + const titleEl = findFirst(root, isTitle); + const fallbackTitle = titleEl ? null : findFirst(root, (e) => e.tag === "h1"); + const title = textOf(titleEl ?? fallbackTitle ?? { tag: "#none", attrs: {}, children: [] }) || undefined; + + const bylineEl = findFirst(root, isByline, isEmbeddedPost); + let author: string | undefined; + let handle: string | undefined; + if (bylineEl) { + const lines = textOf(bylineEl).split(/\n|·/).map((s) => s.trim()).filter(Boolean); + handle = lines.find((l) => /^@[A-Za-z0-9_]{1,15}$/.test(l)); + author = lines.find((l) => !l.startsWith("@") && l !== handle); + } + const time = findFirst(root, (e) => e.tag === "time" && !!e.attrs.datetime, isEmbeddedPost); + const publishedAt = time?.attrs.datetime; + + const body = findFirst(root, isBody); + const skipHeader = (el: HtmlElement) => + el === titleEl || el === fallbackTitle || isByline(el) || isChrome(el); + let blocks: XArticleBlock[]; + if (body) { + // A cover image sits above the body: the images before it are the + // article's; nothing after the body (the engagement bar, replies) is. + const cover: XArticleBlock[] = []; + const before = (el: HtmlElement): boolean => { + for (const c of el.children) { + if (!isEl(c)) continue; + if (c === body) return true; + if (isEmbeddedPost(c) || skipHeader(c)) continue; + if (c.tag === "img" && isContentImageSrc(c.attrs.src)) { + cover.push({ type: "image", src: c.attrs.src }); + continue; + } + if (before(c)) return true; + } + return false; + }; + before(root); + blocks = [...cover, ...readBlocks(body, isChrome)]; + } else { + blocks = readBlocks(root, skipHeader); + } + return { + ...(title ? { title } : {}), + ...(author ? { author } : {}), + ...(handle ? { handle } : {}), + ...(publishedAt ? { publishedAt } : {}), + extraction: body ? "structured" : "fallback", + blocks, + }; +} + +// --- markdown ---------------------------------------------------------------- + +// A line that would read as markdown syntax, escaped. +function mdText(s: string): string { + return s + .split("\n") + .map((l) => + l.replace(/^(\s*)([#>+\-*])(\s)/, "$1\\$2$3").replace(/^(\s*)(\d+)([.)])(\s)/, "$1$2\\$3$4"), + ) + .join("\n"); +} +const mdLabel = (s: string) => s.replace(/([\[\]\\])/g, "\\$1").replace(/\n+/g, " "); +const mdDest = (u: string) => `<${u.replace(/[<>\s]/g, encodeURIComponent)}>`; + +// An image block links to its saved file when it has one, else to its src. +export type XArticleMarkdownInput = XArticleContent & { url: string }; + +export function xArticleMarkdown(a: XArticleMarkdownInput): string { + const oneLine = (t: string | undefined) => (t ?? "").replace(/\s*\n\s*/g, " ").trim(); + let md = `# ${oneLine(a.title) || "Untitled article"}`; + const add = (chunk: string, tight = false) => { + md += (tight ? "\n" : "\n\n") + chunk; + }; + const by = [a.author, a.handle].filter(Boolean).join(" "); + const meta = [by ? `By ${by}` : "", a.publishedAt ?? ""].filter(Boolean).join(" · "); + if (meta) add(mdText(meta)); + add(mdDest(a.url)); + let prev: XArticleBlockType | undefined; + for (const b of a.blocks) { + switch (b.type) { + case "heading": + add(`## ${oneLine(b.text)}`); + break; + case "quote": + add((b.text ?? "").split("\n").map((l) => `> ${l}`.trimEnd()).join("\n")); + break; + case "list-item": + // List items run together. + add(`- ${mdText(b.text ?? "").replace(/\n/g, "\n ")}`, prev === "list-item"); + break; + case "image": + add(`![${mdLabel(b.text ?? "")}](${mdDest(b.file ?? b.src ?? "")})`); + break; + case "embedded-post": + add( + `> Embedded post: ${b.href ? mdDest(b.href) : "(no link)"}` + + (b.text ? "\n>\n" + b.text.split("\n").map((l) => `> ${l}`.trimEnd()).join("\n") : ""), + ); + break; + case "link": + add(`[${mdLabel(b.text || b.href || "")}](${mdDest(b.href ?? "")})`); + break; + default: + add(mdText(b.text ?? "")); + } + prev = b.type; + } + return md + "\n"; +} diff --git a/common/social/xArticleCapture.test.ts b/common/social/xArticleCapture.test.ts @@ -0,0 +1,472 @@ +// The X Article half of a post capture, against fakes only: a page that +// answers the capture's evaluate calls from recorded snapshots and the written +// HTML fixture, and a request context that serves images from a table. No +// browser, no network. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticleCapture.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { mkdtemp, readdir, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import type { PageLike } from "./playwrightRuntime"; +import { + ARTICLE_SHOT_MAX_HEIGHT, + captureXArticle, + classifyXArticleSnapshot, + fullSizeImageUrl, + imageExtension, + type XArticleSnapshot, +} from "./xArticleCapture"; +import { captureXPosts, type XCaptureDeps, type XPostSnapshot } from "./xPostCapture"; +import { readPostCapture, writePostCapture } from "./postCapture"; +import type { PostCaptureInput } from "./fetchers"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8"); +const sha = (b: string | Uint8Array) => createHash("sha256").update(b).digest("hex"); +const PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 7, 7]); +const NOW = () => new Date("2026-02-03T04:05:06.000Z"); +const LINK = { articleId: "999", url: "https://x.com/i/article/999" }; + +const COVER = "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small"; +const BODY_IMG = "https://pbs.twimg.com/media/BodyImg1?format=png&name=small"; +const orig = (u: string) => fullSizeImageUrl(u); + +const articleSnap = (height = 2400): XArticleSnapshot => ({ + path: "/i/article/999", + text: "A Worked Example", + root: { rect: { x: 0, y: 53.5, width: 600.2, height }, marker: '[data-testid="twitterArticleReadView"]' }, +}); +const noArticle = (text: string, p = "/i/article/999"): XArticleSnapshot => ({ path: p, text, root: null }); + +type Served = Record<string, { status: number; body?: string; type?: string }>; + +type FakeArticlePage = PageLike & { calls: string[]; shots: unknown[]; fetched: string[] }; + +// A page for articles: the snapshot script answers `snaps` in turn (the last +// repeats), the HTML script the fixture, the lazy-load walk nothing; images +// come from `served` (every full-size picture by default). +function articlePage( + snaps: XArticleSnapshot[], + opts: { html?: string; served?: Served; gotoFails?: boolean; noRequest?: boolean } = {}, +): FakeArticlePage { + let i = 0; + const calls: string[] = []; + const shots: unknown[] = []; + const fetched: string[] = []; + const served: Served = opts.served ?? { + [orig(COVER)]: { status: 200, body: "cover bytes", type: "image/jpeg" }, + [orig(BODY_IMG)]: { status: 200, body: "chart bytes", type: "image/png" }, + }; + const p: FakeArticlePage = { + calls, + shots, + fetched, + goto: async (url: string) => { + calls.push(`goto ${url}`); + if (opts.gotoFails) throw new Error("net::ERR_TIMED_OUT\nat goto"); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: async (script: string) => { + if (script.includes("outerHTML")) return opts.html ?? FIXTURE; + if (script.includes("scrollTo")) { + calls.push("load-lazy"); + return 3; + } + calls.push("snapshot"); + return snaps[Math.min(i++, snaps.length - 1)]; + }, + screenshot: async (o?: unknown) => { + shots.push(o); + return PNG; + }, + }; + if (!opts.noRequest) { + p.request = { + get: async (url: string) => { + fetched.push(url); + const s = served[url] ?? { status: 404 }; + return { + ok: () => s.status >= 200 && s.status < 300, + status: () => s.status, + headers: (): Record<string, string> => (s.type ? { "content-type": s.type } : {}), + body: async () => new TextEncoder().encode(s.body ?? ""), + }; + }, + }; + } + return p; +} + +const tmpDir = async () => path.join(await mkdtemp(path.join(os.tmpdir(), "xarticle-")), "1"); + +// --- classification and helpers ------------------------------------------------- + +test("classify: an article page with a root is captured; walls, deletions and refusals as a post's", () => { + assert.deepEqual(classifyXArticleSnapshot(articleSnap()), { state: "captured" }); + const login = classifyXArticleSnapshot(noArticle("", "/i/flow/login")); + assert.equal(login.state, "login-wall"); + assert.ok(login.stop); + assert.equal(classifyXArticleSnapshot(noArticle("Log in Sign up")).state, "login-wall"); + assert.deepEqual(classifyXArticleSnapshot(noArticle("Hmm...this page doesn’t exist.")), { state: "deleted" }); + assert.deepEqual(classifyXArticleSnapshot(noArticle("This account is suspended")), { state: "unavailable" }); + const refused = classifyXArticleSnapshot(noArticle("Something went wrong. Try reloading.")); + assert.equal(refused.state, "error"); + assert.ok(refused.stop); + assert.match(refused.error ?? "", /instead of the article/); + assert.deepEqual(classifyXArticleSnapshot(noArticle("")), { state: "error", error: "The article did not render." }); +}); + +test("images: the full-size picture of an X media URL, and an extension from the format, path or type", () => { + assert.equal(fullSizeImageUrl(COVER), "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=orig"); + assert.equal(fullSizeImageUrl("https://example.com/a.png?name=small"), "https://example.com/a.png?name=small"); + assert.equal(fullSizeImageUrl("not a url"), "not a url"); + assert.equal(imageExtension(COVER), "jpg"); + assert.equal(imageExtension("https://example.com/p/pic.JPEG"), "jpg"); + assert.equal(imageExtension("https://example.com/p/pic", "image/webp; q=1"), "webp"); + assert.equal(imageExtension("https://example.com/p/pic"), "img"); +}); + +// --- one article ------------------------------------------------------------------ + +test("article: read, shot whole, images fetched full-size through the page; every file recorded", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap(), articleSnap(2500.4)]); + const gaps: number[] = []; + const res = await captureXArticle(p, LINK, dir, { now: NOW, imageGap: async () => void gaps.push(1) }); + assert.equal(res.stop, undefined); + assert.equal(p.calls[0], "goto https://x.com/i/article/999"); + assert.ok(p.calls.includes("load-lazy")); + // The box re-read after the lazy images loaded. + assert.deepEqual(p.shots, [ + { type: "png", fullPage: true, clip: { x: 0, y: 53, width: 601, height: 2501 } }, + ]); + assert.deepEqual(p.fetched, [orig(COVER), orig(BODY_IMG)]); + assert.deepEqual(gaps, [1], "a gap between images, none before the first"); + + const r = res.record; + assert.equal(r.state, "captured"); + assert.equal(r.title, "A Worked Example & Its Notes"); + assert.equal(r.blocks, 12); + assert.equal(r.extraction, "structured"); + assert.equal(r.trimmed, undefined); + assert.equal(r.error, undefined); + assert.deepEqual( + r.files.map((f) => [f.name, f.url]), + [ + ["article.json", LINK.url], + ["article.md", LINK.url], + ["article.html", LINK.url], + ["article.png", LINK.url], + ["article-img-1.jpg", orig(COVER)], + ["article-img-2.png", orig(BODY_IMG)], + ], + ); + const img = r.files.find((f) => f.name === "article-img-2.png")!; + assert.deepEqual([img.bytes, img.sha256], ["chart bytes".length, sha("chart bytes")]); + assert.deepEqual(new Uint8Array(await readFile(path.join(dir, "article.png"))), PNG); + assert.equal(await readFile(path.join(dir, "article.html"), "utf8"), FIXTURE); + + const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")); + assert.equal(json.articleId, "999"); + assert.equal(json.url, LINK.url); + assert.equal(json.author, "Demo Author"); + assert.equal(json.handle, "@demo_author"); + assert.equal(json.publishedAt, "2026-01-02T03:04:05.000Z"); + assert.equal(json.capturedAt, "2026-02-03T04:05:06.000Z"); + assert.equal(json.blocks.length, 12); + assert.deepEqual(json.blocks[8], { + type: "image", + src: BODY_IMG, + text: "A chart of the numbers", + file: "article-img-2.png", + }); + assert.deepEqual(json.blocks[9], { + type: "embedded-post", + text: "An embedded post's own words.", + href: "https://x.com/other_user/status/4444444444", + }); + const md = await readFile(path.join(dir, "article.md"), "utf8"); + assert.match(md, /^# A Worked Example & Its Notes\n/); + assert.match(md, /!\[A chart of the numbers\]\(<article-img-2\.png>\)/); + assert.match(md, /> Embedded post: <https:\/\/x\.com\/other_user\/status\/4444444444>/); +}); + +test("article: a tall one is shot to the cap and recorded trimmed", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap(50_000)]); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.equal(res.record.trimmed, true); + assert.equal((p.shots[0] as { clip: { height: number } }).clip.height, ARTICLE_SHOT_MAX_HEIGHT); + assert.equal(JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")).trimmed, true); +}); + +test("article: a full-size picture refused falls back to the src as rendered; one not served at all is noted", async () => { + const dir = await tmpDir(); + const p = articlePage([articleSnap()], { + served: { [COVER]: { status: 200, body: "small cover", type: "image/jpeg" } }, + }); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.deepEqual(p.fetched, [orig(COVER), COVER, orig(BODY_IMG), BODY_IMG]); + assert.equal(res.record.state, "captured"); + assert.deepEqual( + res.record.files.filter((f) => f.name.startsWith("article-img-")).map((f) => [f.name, f.url]), + [["article-img-1.jpg", COVER]], + ); + assert.match(res.record.error ?? "", /1 image\(s\) could not be fetched: HTTP 404/); + const json = JSON.parse(await readFile(path.join(dir, "article.json"), "utf8")); + assert.equal(json.blocks[8].file, undefined, "an image not saved links to its src"); +}); + +test("article: a page with no request context still saves the text, and says the images are missing", async () => { + const res = await captureXArticle(articlePage([articleSnap()], { noRequest: true }), LINK, await tmpDir(), { + now: NOW, + }); + assert.equal(res.record.state, "captured"); + assert.match(res.record.error ?? "", /no request context/); + assert.equal(res.record.files.length, 4); +}); + +test("article: a re-capture with fewer images leaves no older numbered file behind", async () => { + const dir = await tmpDir(); + await captureXArticle(articlePage([articleSnap()]), LINK, dir, { now: NOW }); + await writeFile(path.join(dir, "article-img-3.jpg"), "stale"); + await captureXArticle(articlePage([articleSnap()], { html: "<div data-testid=\"twitterArticleRichTextView\"><p>Just text</p></div>" }), LINK, dir, { now: NOW }); + const names = (await readdir(dir)).sort(); + assert.deepEqual(names, ["article.html", "article.json", "article.md", "article.png"]); +}); + +test("article: a login wall stops the run; deleted is recorded; a failed load is an error; no file is written", async () => { + for (const [p, state, stops] of [ + [articlePage([noArticle("", "/i/flow/login")]), "login-wall", true], + [articlePage([noArticle("Something went wrong. Try reloading.")]), "error", true], + [articlePage([noArticle("Hmm...this page doesn’t exist.")]), "deleted", false], + [articlePage([articleSnap()], { gotoFails: true }), "error", false], + ] as const) { + const dir = await tmpDir(); + const res = await captureXArticle(p, LINK, dir, { now: NOW }); + assert.equal(res.record.state, state); + assert.equal(Boolean(res.stop), stops, state); + assert.deepEqual(res.record.files, []); + assert.equal(res.record.blocks, 0); + assert.equal(p.shots.length, 0); + assert.deepEqual(await readdir(dir).catch(() => []), []); + } +}); + +// --- in the run ---------------------------------------------------------------- + +const post = (articleHrefs?: string[]): XPostSnapshot => ({ + path: "/someone/status/1", + text: "the post", + article: { rect: { x: 10, y: 120, width: 598, height: 300 }, sensitive: false, ...(articleHrefs ? { articleHrefs } : {}) }, +}); + +// Post pages by id, article pages by article id, one routed page over both. +function runHarness( + postsById: Record<string, XPostSnapshot>, + articlesById: Record<string, XArticleSnapshot[]>, +): { deps: XCaptureDeps; pauses: number[]; visits: string[]; downloads: string[] } { + const visits: string[] = []; + const pauses: number[] = []; + const downloads: string[] = []; + let current: PageLike = articlePage([]); + const articlePages = Object.fromEntries( + Object.entries(articlesById).map(([id, snaps]) => [id, articlePage(snaps)]), + ); + const routed: PageLike = { + goto: async (url: string, o?: unknown) => { + visits.push(url); + const id = url.split("/").pop()!; + if (url.includes("/article/")) { + current = articlePages[id]; + } else { + const snap = postsById[id]; + current = { + ...articlePage([]), + evaluate: async (s: string) => (s.includes("imgs.length") ? 0 : snap), + screenshot: async () => PNG, + }; + } + return current.goto(url, o); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: (s: string) => current.evaluate(s), + screenshot: (o) => current.screenshot(o), + request: { get: (url, o) => current.request!.get(url, o) }, + }; + return { + visits, + pauses, + downloads, + deps: { + openPage: async () => ({ page: routed, close: async () => {} }), + downloadMedia: async ({ id }) => { + downloads.push(id); + return { ok: true, files: [] }; + }, + pauseMs: () => 5_000, + pause: async (ms) => void pauses.push(ms), + articleImagePauseMs: () => 1_000, + now: NOW, + }, + }; +} + +async function input(ids: string[], over: Partial<PostCaptureInput> = {}): Promise<PostCaptureInput> { + return { + ids, + handle: "example_user", + outDir: await mkdtemp(path.join(os.tmpdir(), "posts-media-")), + signal: new AbortController().signal, + ...over, + }; +} + +const ARCHIVED = new Map([["11", { text: "https://x.com/i/article/999" }]]); + +test("run: a post whose archived text is an article link — shot, media, then the article, each paced", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + const inp = await input(["11", "22"], { archived: ARCHIVED }); + const res = await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, [ + "https://x.com/i/status/11", + "https://x.com/i/article/999", + "https://x.com/i/status/22", + ]); + // Five contacts (two shots, two downloads, one article): four post gaps, + // and one short gap between the article's two images. + assert.deepEqual(h.pauses, [5_000, 5_000, 1_000, 5_000, 5_000]); + assert.deepEqual( + res.outcomes.map((o) => [o.id, o.state, o.files]), + [ + ["11", "captured", 7], + ["22", "captured", 1], + ], + ); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.article?.state, "captured"); + assert.equal(rec.article?.articleId, "999"); + assert.equal(rec.article?.title, "A Worked Example & Its Notes"); + assert.equal(rec.article?.blocks, 12); + assert.equal(rec.article?.files.length, 6); + assert.equal((await readPostCapture(path.join(inp.outDir, "22")))!.article, undefined); + + // Captured is settled: a second run opens nothing. + const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []); + assert.deepEqual(again.visits, []); + + // force reads it again. + const forced = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + await captureXPosts({ ...inp, ids: ["11"], force: true }, forced.deps); + assert.deepEqual(forced.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]); +}); + +test("run: an article found only on the post's card is opened too; articles: false opens none", async () => { + const h = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] }); + const inp = await input(["11"]); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/demo_author/article/999"]); + assert.equal((await readPostCapture(path.join(inp.outDir, "11")))!.article?.state, "captured"); + + const off = runHarness({ "11": post(["/demo_author/article/999"]) }, { "999": [articleSnap()] }); + const inp2 = await input(["11"], { archived: ARCHIVED, articles: false }); + await captureXPosts(inp2, off.deps); + assert.deepEqual(off.visits, ["https://x.com/i/status/11"]); + assert.equal((await readPostCapture(path.join(inp2.outDir, "11")))!.article, undefined); +}); + +test("run: a post captured before articles owes only its article — no second shot or download", async () => { + const first = runHarness({ "11": post() }, { "999": [articleSnap()] }); + const inp = await input(["11"], { articles: false }); + await captureXPosts(inp, first.deps); + + const h = runHarness({ "11": post() }, { "999": [articleSnap()] }); + const res = await captureXPosts({ ...inp, articles: true, archived: ARCHIVED }, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/article/999"]); + assert.deepEqual(h.downloads, []); + assert.equal(res.outcomes[0].availability, undefined, "an article-only pass says nothing about the post's liveness"); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.state, "captured"); + assert.ok(rec.shot, "the shot is kept"); + assert.equal(rec.article?.state, "captured"); +}); + +test("run: an article behind a login wall stops the run with needsCookies; the rest untouched", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("", "/i/flow/login")] }); + const inp = await input(["11", "22"], { archived: ARCHIVED, media: false }); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.needsCookies, true); + assert.match(res.stoppedEarly ?? "", /Stopped at 11/); + assert.deepEqual(h.visits, ["https://x.com/i/status/11", "https://x.com/i/article/999"]); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.state, "captured", "the post itself was shot"); + assert.equal(rec.article?.state, "login-wall"); + assert.equal(await readPostCapture(path.join(inp.outDir, "22")), null); + + // A login wall is owed again on the next run. + const next = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + await captureXPosts({ ...inp, ids: ["11"] }, next.deps); + assert.deepEqual(next.visits, ["https://x.com/i/article/999"]); +}); + +test("run: a deleted article is recorded and settled; the run goes on", async () => { + const h = runHarness({ "11": post(), "22": post() }, { "999": [noArticle("Hmm...this page doesn’t exist.")] }); + const inp = await input(["11", "22"], { archived: ARCHIVED, media: false }); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.stoppedEarly, undefined); + assert.equal(res.outcomes.length, 2); + const rec = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(rec.article?.state, "deleted"); + const again = runHarness({ "11": post(), "22": post() }, { "999": [articleSnap()] }); + assert.deepEqual((await captureXPosts(inp, again.deps)).outcomes, []); + assert.deepEqual(again.visits, []); +}); + +test("run: a deleted post opens no article", async () => { + const h = runHarness( + { "11": { path: "/i/status/11", text: "This post was deleted by the post author.", article: null } }, + { "999": [articleSnap()] }, + ); + const inp = await input(["11"], { archived: ARCHIVED }); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.visits, ["https://x.com/i/status/11"]); +}); + +test("run: an older record's article is not lost by a run that does not read it", async () => { + const inp = await input(["11"], { archived: ARCHIVED }); + const dir = path.join(inp.outDir, "11"); + const article = { + articleId: "999", + url: LINK.url, + capturedAt: "2026-01-01T00:00:00.000Z", + state: "captured" as const, + blocks: 3, + files: [], + }; + await writePostCapture(dir, { + version: 1, + id: "11", + url: "https://x.com/i/status/11", + capturedAt: "2026-01-01T00:00:00.000Z", + state: "captured", + shot: { name: "shot.png", bytes: 1, sha256: "x" }, + mediaState: "error", + media: [], + article, + }); + const h = runHarness({ "11": post() }, {}); + await captureXPosts(inp, h.deps); + assert.deepEqual(h.downloads, ["11"], "only the owed media"); + assert.deepEqual(h.visits, []); + assert.deepEqual((await readPostCapture(dir))!.article, article); +}); diff --git a/common/social/xArticleCapture.ts b/common/social/xArticleCapture.ts @@ -0,0 +1,392 @@ +// Capturing an X Article (a long-form post) beside the post that links to it: +// the article page opened in the same logged-in profile and page as the post's +// shot, read into blocks (xArticle.ts), and saved as +// article.json — the blocks, title, byline, in reading order +// article.md — the same as readable markdown +// article.png — the article root, shot whole up to a height cap +// article.html — the root as X served it, so a better reading later costs no +// second visit +// article-img-<n>.<ext> — its inline images, fetched through the page's own +// request context (the profile's cookies), not a new tool +// +// PACED AS ONE CONTACT. The article load is a contact with X like the post's +// shot: the run waits its 4–10 s gap before it (xPostCapture.ts). The images +// are the page's own, already loaded once by the browser; they are fetched +// again a short gap apart, not at the post gap. +// +// The page is read as a post page is (classifyXPostSnapshot): a login wall +// stops the run (needsCookies), "Something went wrong" stops it, a deleted or +// unavailable article is recorded and settled, anything else is an error a +// later run tries again. Nothing is retried in the run that met it. +// +// NOT VERIFIED AGAINST LIVE X. The root markers are X's as of this writing, +// tested against recorded snapshots and a written HTML fixture, never x.com. + +import { mkdir, readdir, rm, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { writeFileAtomic, writeJsonAtomic } from "../lib/jsonFile-server"; +import type { PageLike } from "./playwrightRuntime"; +import { + ARTICLE_HTML_FILENAME, + ARTICLE_JSON_FILENAME, + ARTICLE_MD_FILENAME, + ARTICLE_SHOT_FILENAME, + articleImageFilename, + describeCapturedFile, + type ArticleCaptureRecord, + type CapturedFile, +} from "./postCapture"; +import { + extractXArticle, + findXArticleLink, + xArticleMarkdown, + type XArticleBlock, + type XArticleLink, +} from "./xArticle"; +import { classifyXPostSnapshot, type XPostVerdict } from "./xPostCapture"; + +// The article root, most specific first. The read view holds the title, the +// byline and the body; the rich-text view only the body, so it is widened to +// the article around it. +export const ARTICLE_ROOT_MARKERS = [ + '[data-testid="twitterArticleReadView"]', + '[data-testid="twitterArticleRichTextView"]', + '[data-testid="longformRichTextComponent"]', +]; +const ARTICLE_FALLBACK_ROOT = '[data-testid="primaryColumn"] article, [data-testid="primaryColumn"] [role="article"]'; + +// article.png stops here. Chromium's full-page capture is a single texture, +// and past its 16384 px limit a taller shot repeats or blanks, so the cap sits +// under it; a longer article is recorded `trimmed` (article.md has it all). +export const ARTICLE_SHOT_MAX_HEIGHT = 16_000; + +export type XArticleSnapshot = { + path: string; + text: string; + root: null | { + rect: { x: number; y: number; width: number; height: number }; + // Which marker found the root ("fallback": none did). + marker: string; + }; +}; + +const ARTICLE_SNAPSHOT_SCRIPT = `(() => { + const markers = ${JSON.stringify(ARTICLE_ROOT_MARKERS)}; + let root = null; + let marker = null; + for (const m of markers) { + const el = document.querySelector(m); + if (el) { root = el; marker = m; break; } + } + if (root && marker !== markers[0]) { + root = root.closest('article, [role="article"]') || root; + } + if (!root) { + root = document.querySelector(${JSON.stringify(ARTICLE_FALLBACK_ROOT)}); + if (root) marker = "fallback"; + } + const main = document.querySelector('[data-testid="primaryColumn"]') || document.body; + const text = ((main && main.innerText) || "").slice(0, 4000); + let out = null; + if (root) { + for (const old of document.querySelectorAll('[data-archilyzer-article]')) { + old.removeAttribute('data-archilyzer-article'); + } + root.setAttribute('data-archilyzer-article', ''); + const r = root.getBoundingClientRect(); + out = { + rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, + marker, + }; + } + return { path: location.pathname, text, root: out }; +})()`; + +// Walk down the page so lazy images load, then back to the top for the shot. +const LOAD_LAZY_SCRIPT = `(async () => { + const root = document.querySelector('[data-archilyzer-article]') || document.body; + for (const img of root.querySelectorAll('img[loading="lazy"]')) img.loading = "eager"; + const wait = (ms) => new Promise((r) => setTimeout(r, ms)); + let steps = 0; + for (let y = 0; y < document.documentElement.scrollHeight && steps < 200; steps++) { + y += Math.max(400, Math.floor(window.innerHeight * 0.8)); + window.scrollTo(0, y); + await wait(250); + } + window.scrollTo(0, 0); + await wait(500); + const pending = Array.from(root.querySelectorAll("img")).filter((i) => !i.complete); + await Promise.all(pending.map((i) => new Promise((r) => { + i.addEventListener("load", r, { once: true }); + i.addEventListener("error", r, { once: true }); + setTimeout(r, 5000); + }))); + return steps; +})()`; + +const ARTICLE_HTML_SCRIPT = `(() => { + const root = document.querySelector('[data-archilyzer-article]'); + return root ? root.outerHTML : ""; +})()`; + +// What an article page means, in a post's terms. A page with a root is +// captured; one without is read for X's markers as a post page is. +export function classifyXArticleSnapshot(s: XArticleSnapshot): XPostVerdict { + return classifyXPostSnapshot( + { + path: s.path, + text: s.text, + article: s.root ? { rect: s.root.rect, sensitive: false } : null, + }, + "article", + ); +} + +// The article a post's rendered card links to: the hrefs read from it. +export function xArticleLinkFromCard(hrefs: ReadonlyArray<string> | undefined): XArticleLink | null { + return hrefs?.length ? findXArticleLink(hrefs) : null; +} + +// The full-size picture of an X media URL (`name=orig`); anything else as is. +export function fullSizeImageUrl(src: string): string { + try { + const u = new URL(src); + if (u.hostname === "pbs.twimg.com" && u.pathname.startsWith("/media/")) { + u.searchParams.set("name", "orig"); + return u.toString(); + } + } catch { + /* not a URL: as is */ + } + return src; +} + +const EXT_BY_TYPE: Record<string, string> = { + "image/jpeg": "jpg", + "image/png": "png", + "image/gif": "gif", + "image/webp": "webp", + "image/avif": "avif", +}; + +// An image's extension: X's `format=` parameter, the path's own, the response's +// type, in that order. +export function imageExtension(url: string, contentType?: string): string { + try { + const u = new URL(url); + const format = u.searchParams.get("format"); + if (format && /^[a-z0-9]{2,5}$/i.test(format)) return format.toLowerCase(); + const m = /\.([a-z0-9]{2,5})$/i.exec(u.pathname); + if (m) return m[1].toLowerCase() === "jpeg" ? "jpg" : m[1].toLowerCase(); + } catch { + /* fall through to the type */ + } + const type = (contentType ?? "").split(";")[0].trim().toLowerCase(); + return EXT_BY_TYPE[type] ?? "img"; +} + +export type XArticleCaptureResult = { + record: ArticleCaptureRecord; + // Set when the run must stop here (a login wall, X refusing pages). + stop?: string; +}; + +export type XArticleCaptureOptions = { + now: () => Date; + onLog?: (line: string) => void; + signal?: AbortSignal; + // The gap between one image fetch and the next. + imageGap?: () => Promise<void>; +}; + +// Open the article, read it, shoot it, fetch its images, write its files into +// `dir`. The record says how it went; nothing is retried here. +export async function captureXArticle( + page: PageLike, + link: XArticleLink, + dir: string, + opts: XArticleCaptureOptions, +): Promise<XArticleCaptureResult> { + const { onLog } = opts; + const base = { articleId: link.articleId, url: link.url }; + const failed = (verdict: XPostVerdict): XArticleCaptureResult => ({ + record: { + ...base, + capturedAt: opts.now().toISOString(), + state: verdict.state, + blocks: 0, + files: [], + ...(verdict.error ? { error: verdict.error } : {}), + }, + ...(verdict.stop ? { stop: verdict.stop } : {}), + }); + + try { + await page.goto(link.url, { waitUntil: "domcontentloaded", timeout: 60_000 }); + } catch (err) { + return failed({ state: "error", error: `Could not load ${link.url}: ${firstLine(err)}` }); + } + await page + .waitForSelector([...ARTICLE_ROOT_MARKERS, ARTICLE_FALLBACK_ROOT].join(", "), { timeout: 20_000 }) + .catch(() => {}); + await page.waitForTimeout(1_500); + let snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; + const verdict = classifyXArticleSnapshot(snap); + if (verdict.state !== "captured") return failed(verdict); + if (snap.root?.marker === "fallback") { + onLog?.(`article ${link.articleId}: no article marker on the page — reading the page's post as the article.`); + } + + await page.evaluate(LOAD_LAZY_SCRIPT).catch(() => {}); + snap = (await page.evaluate(ARTICLE_SNAPSHOT_SCRIPT)) as XArticleSnapshot; + const html = String((await page.evaluate(ARTICLE_HTML_SCRIPT)) ?? ""); + const rect = snap.root?.rect; + if (!html || !rect || rect.width < 1 || rect.height < 1) { + return failed({ state: "error", error: "The article rendered with nothing to read." }); + } + const content = extractXArticle(html); + const capturedAt = opts.now().toISOString(); + + // The shot: the root, whole, up to the cap. + const trimmed = rect.height > ARTICLE_SHOT_MAX_HEIGHT; + let png: Uint8Array | undefined; + let error: string | undefined; + try { + png = await page.screenshot({ + type: "png", + fullPage: true, + clip: { + x: Math.max(0, Math.floor(rect.x)), + y: Math.max(0, Math.floor(rect.y)), + width: Math.ceil(rect.width), + height: Math.min(Math.ceil(rect.height), ARTICLE_SHOT_MAX_HEIGHT), + }, + }); + } catch (err) { + error = `The article's screenshot failed: ${firstLine(err)}`; + } + + // The images, each once, numbered in reading order. + await mkdir(dir, { recursive: true }); + const images = await fetchArticleImages(page, content.blocks, dir, opts); + if (images.errors.length) { + const why = `${images.errors.length} image(s) could not be fetched: ${images.errors.join("; ")}`; + error = error ? `${error}; ${why}` : why; + } + + const blocks: XArticleBlock[] = content.blocks.map((b) => + b.type === "image" && b.src && images.saved.has(b.src) + ? { ...b, file: images.saved.get(b.src)!.name } + : b, + ); + const article = { + version: 1, + ...base, + ...(content.title ? { title: content.title } : {}), + ...(content.author ? { author: content.author } : {}), + ...(content.handle ? { handle: content.handle } : {}), + ...(content.publishedAt ? { publishedAt: content.publishedAt } : {}), + capturedAt, + extraction: content.extraction, + ...(trimmed ? { trimmed: true } : {}), + blocks, + }; + await writeJsonAtomic(path.join(dir, ARTICLE_JSON_FILENAME), article, { mkdir: true }); + await writeFileAtomic( + path.join(dir, ARTICLE_MD_FILENAME), + xArticleMarkdown({ ...content, blocks, url: link.url }), + ); + await writeFileAtomic(path.join(dir, ARTICLE_HTML_FILENAME), html); + if (png) await writeFile(path.join(dir, ARTICLE_SHOT_FILENAME), png); + await removeStaleImages(dir, new Set([...images.saved.values()].map((f) => f.name))); + + const files: CapturedFile[] = [ + await describeCapturedFile(dir, ARTICLE_JSON_FILENAME, link.url), + await describeCapturedFile(dir, ARTICLE_MD_FILENAME, link.url), + await describeCapturedFile(dir, ARTICLE_HTML_FILENAME, link.url), + ...(png ? [await describeCapturedFile(dir, ARTICLE_SHOT_FILENAME, link.url)] : []), + ...images.saved.values(), + ]; + return { + record: { + ...base, + capturedAt, + state: "captured", + ...(content.title ? { title: content.title } : {}), + blocks: blocks.length, + extraction: content.extraction, + files, + ...(trimmed ? { trimmed: true } : {}), + ...(error ? { error } : {}), + }, + }; +} + +// Each distinct image src, fetched through the page's request context into +// `article-img-<n>.<ext>`: the full-size picture first, the src as rendered if +// that is refused. +async function fetchArticleImages( + page: PageLike, + blocks: ReadonlyArray<XArticleBlock>, + dir: string, + opts: XArticleCaptureOptions, +): Promise<{ saved: Map<string, CapturedFile>; errors: string[] }> { + const saved = new Map<string, CapturedFile>(); + const errors: string[] = []; + const srcs = [...new Set(blocks.flatMap((b) => (b.type === "image" && b.src ? [b.src] : [])))]; + if (srcs.length === 0) return { saved, errors }; + if (!page.request) { + errors.push("this browser page cannot fetch (no request context)"); + return { saved, errors }; + } + let n = 0; + for (const src of srcs) { + if (opts.signal?.aborted) { + errors.push("cancelled before the rest of the images"); + break; + } + if (n > 0) await opts.imageGap?.(); + n++; + const tries = [...new Set([fullSizeImageUrl(src), src])]; + let last = ""; + for (const url of tries) { + try { + const res = await page.request.get(url, { timeout: 30_000, failOnStatusCode: false }); + if (!res.ok()) { + last = `HTTP ${res.status()} for ${url}`; + continue; + } + const body = await res.body(); + const name = articleImageFilename(n, imageExtension(url, res.headers()["content-type"])); + await writeFile(path.join(dir, name), body); + saved.set(src, await describeCapturedFile(dir, name, url)); + last = ""; + break; + } catch (err) { + last = `${firstLine(err)} (${url})`; + } + } + if (last) errors.push(last); + } + if (saved.size) opts.onLog?.(`article: ${saved.size} image(s) saved.`); + return { saved, errors }; +} + +// A re-capture with fewer images leaves no older numbered file behind. +async function removeStaleImages(dir: string, keep: ReadonlySet<string>): Promise<void> { + let names: string[]; + try { + names = await readdir(dir); + } catch { + return; + } + for (const name of names) { + if (/^article-img-\d+\./.test(name) && !keep.has(name)) { + await rm(path.join(dir, name), { force: true }); + } + } +} + +function firstLine(err: unknown): string { + return ((err as Error)?.message ?? String(err)).split("\n")[0]; +} diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -41,6 +41,7 @@ import { } from "./xSessionBroker"; import type { XCookieSource } from "./xCookieSource"; import { listCapturedMediaFiles } from "./postCapture"; +import { xArticleLinkFromArchive } from "./xArticle"; import { captureXPosts, xStatusUrl, @@ -328,11 +329,17 @@ export async function captureXPostsByIds( input: PostCaptureInput, ): Promise<PostCaptureResult> { const paths = getPaths(); - if (input.shots ?? true) { + // The profile shoots the posts and opens the articles. A post is known to + // link to an article before any page only by its archived text; one found + // on a card is found by a shot, which already needs the profile. + const articles = + (input.articles ?? true) && + input.ids.some((id) => xArticleLinkFromArchive(input.archived?.get(id)) !== null); + if ((input.shots ?? true) || articles) { const status = await readXSessionStatus(paths); if (!status.hasProfile) { const why = - "No X session profile to shoot posts with: connect an X account on /settings, then run it again."; + "No X session profile to shoot posts or open articles with: connect an X account on /settings, then run it again."; input.onLog?.(`[auth] ${why}`); return { outcomes: [], needsCookies: true, stoppedEarly: why }; } @@ -355,9 +362,14 @@ export async function captureXPostsByIds( }, pauseMs: () => xRequestPauseMs(), pause, + articleImagePauseMs: () => ARTICLE_IMAGE_PAUSE_MS + Math.floor(Math.random() * ARTICLE_IMAGE_PAUSE_MS), }); } +// The gap between an article's images: 1–2 s. They are the page's own +// pictures, already loaded once by the browser, so not the full post gap. +const ARTICLE_IMAGE_PAUSE_MS = 1_000; + // --------------------------------------------------------------------------- // Search: the older-posts backfill // --------------------------------------------------------------------------- diff --git a/common/social/xPostCapture.ts b/common/social/xPostCapture.ts @@ -21,6 +21,10 @@ // - a protected, suspended or vanished account, or a withheld post: // unavailable. // +// A post that is an X Article (long-form) links to it — in its archived text, +// or on its rendered card — and the run then opens the article too, one more +// paced contact (xArticleCapture.ts), unless `articles` is off. +// // NOT VERIFIED AGAINST LIVE X. The page markers below are X's as of this // writing and are tested against recorded snapshots, never x.com; the first // real run is the check. @@ -41,11 +45,14 @@ import { readPostCapture, SHOT_FILENAME, writePostCapture, + type ArticleCaptureRecord, type CapturedFile, type CaptureMediaState, type PostCaptureRecord, type PostCaptureState, } from "./postCapture"; +import { xArticleLinkFromArchive, type XArticleLink } from "./xArticle"; +import { captureXArticle, xArticleLinkFromCard } from "./xArticleCapture"; export function xStatusUrl(id: string): string { return `https://x.com/i/status/${id}`; @@ -60,6 +67,8 @@ export type XPostSnapshot = { rect: { x: number; y: number; width: number; height: number }; // A "Show" / "View" button inside the post: a sensitive-media cover. sensitive: boolean; + // The hrefs inside the post that look like an X Article's: its card. + articleHrefs?: string[]; }; }; @@ -83,9 +92,14 @@ const SNAPSHOT_SCRIPT = (id: string) => `(() => { const sensitive = Array.from(own.querySelectorAll('button, [role="button"]')).some( (b) => /^(show|view)$/i.test((b.innerText || "").trim()), ); + const articleHrefs = Array.from(own.querySelectorAll('a[href*="/article/"]')) + .map((l) => l.getAttribute("href") || "") + .filter(Boolean) + .slice(0, 10); article = { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, sensitive, + articleHrefs, }; } return { path: location.pathname, text, article }; @@ -140,7 +154,12 @@ export type XPostVerdict = { }; // What a snapshot means. Pure, so every marker is testable without a browser. -export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { +// `noun` names what the page should have shown (an article page is read the +// same way, xArticleCapture.ts). +export function classifyXPostSnapshot( + s: Pick<XPostSnapshot, "path" | "text" | "article">, + noun = "post", +): XPostVerdict { if (/^\/(i\/flow\/login|login|i\/flow\/signup)\b/.test(s.path)) { return { state: "login-wall", @@ -153,7 +172,7 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { if (REFUSED_TEXT.test(s.text)) { return { state: "error", - error: "X answered “Something went wrong” instead of the post.", + error: `X answered “Something went wrong” instead of the ${noun}.`, stop: "X is refusing pages right now; stopping rather than asking again.", }; } @@ -162,19 +181,23 @@ export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { if (AGE_WALL_TEXT.test(s.text)) { return { state: "error", - error: "X shows this post only to an age-verified session.", + error: `X shows this ${noun} only to an age-verified session.`, }; } if (LOGGED_OUT_TEXT.test(s.text)) { return { state: "login-wall", - stop: "X showed its logged-out page instead of the post.", + stop: `X showed its logged-out page instead of the ${noun}.`, }; } - return { state: "error", error: "The post did not render." }; + return { state: "error", error: `The ${noun} did not render.` }; } -export type XShotResult = XPostVerdict & { shot?: CapturedFile }; +export type XShotResult = XPostVerdict & { + shot?: CapturedFile; + // The X Article the post's card links to, when it does. + articleLink?: XArticleLink; +}; // One post's screenshot: load, read the page, open a sensitive cover, shoot // the post's own article. Writes `shot.png` into `dir` only for a post that @@ -234,7 +257,8 @@ export async function shootXPost( await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, SHOT_FILENAME), png); const shot = await describeCapturedFile(dir, SHOT_FILENAME, url); - return { ...verdict, shot }; + const articleLink = xArticleLinkFromCard(snap.article?.articleHrefs); + return { ...verdict, shot, ...(articleLink ? { articleLink } : {}) }; } export type MediaDownloadResult = @@ -255,6 +279,8 @@ export type XCaptureDeps = { // The gap before each contact with X after the first. pauseMs: () => number; pause: (ms: number, signal: AbortSignal) => Promise<void>; + // The gap between one article image and the next (none when absent). + articleImagePauseMs?: () => number; now?: () => Date; }; @@ -274,6 +300,7 @@ export async function captureXPosts( shots: input.shots ?? true, media: input.media ?? true, force: input.force ?? false, + articles: input.articles ?? true, }; const now = deps.now ?? (() => new Date()); const outcomes: PostCaptureOutcome[] = []; @@ -296,7 +323,15 @@ export async function captureXPosts( const dir = postCaptureDir(input.outDir, id); const existing = await readPostCapture(dir); const work = captureWork(existing, wanted); - if (!work.shot && !work.media) { + // The article the post links to, as far as is known before any page: + // the archived text, or the last capture's record. + let articleLink: XArticleLink | null = wanted.articles + ? (xArticleLinkFromArchive(input.archived?.get(id)) ?? + (existing?.article ? { articleId: existing.article.articleId, url: existing.article.url } : null)) + : null; + const articleOwedNow = + work.article && !!articleLink && (!existing || work.shot || existing.state === "captured"); + if (!work.shot && !work.media && !articleOwedNow) { onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`); continue; } @@ -318,6 +353,7 @@ export async function captureXPosts( shot = res.shot; error = res.error; stop = res.stop; + if (wanted.articles && !articleLink && res.articleLink) articleLink = res.articleLink; } let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped"); @@ -347,6 +383,34 @@ export async function captureXPosts( } } + // The article: after the post and its media, one more paced load in the + // same page. Only for a post that is there, in a run not already + // stopping. + let article: ArticleCaptureRecord | undefined = existing?.article; + let articleFailed = false; + if (work.article && articleLink && (state === undefined || state === "captured") && !stop) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + browser ??= await deps.openPage(); + const got = await captureXArticle(browser.page, articleLink, dir, { + now, + onLog, + signal, + imageGap: deps.articleImagePauseMs + ? () => deps.pause(deps.articleImagePauseMs!(), signal) + : undefined, + }); + article = got.record; + articleFailed = article.state === "error"; + // A run that only read the article learns of the post only that it + // linked to a readable article. + state ??= article.state === "captured" ? "captured" : "error"; + if (got.stop) { + stop = got.stop; + if (article.state === "login-wall") needsCookies = true; + } + } + const finalState: PostCaptureState = state ?? "error"; const record: PostCaptureRecord = { version: 1, @@ -358,6 +422,7 @@ export async function captureXPosts( ...(shot ? { shot } : {}), mediaState, media, + ...(article ? { article } : {}), ...(error ? { error } : {}), }; await writePostCapture(dir, record); @@ -369,13 +434,14 @@ export async function captureXPosts( ...(work.shot && captureAvailability(finalState) ? { availability: captureAvailability(finalState) } : {}), - files: (shot ? 1 : 0) + media.length, + files: (shot ? 1 : 0) + media.length + (article?.files.length ?? 0), ...(error ? { error } : {}), }); onLog?.( `${id}: ${finalState}${sensitive ? " (behind a sensitive-media cover)" : ""}` + (shot ? ", shot" : "") + (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") + + (article && article !== existing?.article ? `, ${describeArticle(article)}` : "") + (error ? ` — ${error}` : ""), ); @@ -384,7 +450,7 @@ export async function captureXPosts( needsCookies: needsCookies || finalState === "login-wall", }); } - errorsInARow = finalState === "error" ? errorsInARow + 1 : 0; + errorsInARow = finalState === "error" || articleFailed ? errorsInARow + 1 : 0; if (errorsInARow >= STOP_AFTER_ERRORS) { return stopped( `${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`, @@ -397,6 +463,19 @@ export async function captureXPosts( } } +function describeArticle(a: ArticleCaptureRecord): string { + if (a.state !== "captured") { + return `article ${a.state}` + (a.error ? ` (${a.error})` : ""); + } + return ( + `article${a.title ? ` “${a.title}”` : ""} (${a.blocks} block(s), ${a.files.length} file(s)` + + (a.extraction === "fallback" ? ", read by the fallback" : "") + + (a.trimmed ? ", shot trimmed" : "") + + ")" + + (a.error ? ` — ${a.error}` : "") + ); +} + function firstLine(err: unknown): string { return ((err as Error)?.message ?? String(err)).split("\n")[0]; } diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Capturing an X post that is an Article also saves the article.** An X Article (a long-form post) is archived as nothing but its link, and gallery-dl cannot read its body. When `pnpm ops capture-posts` meets a post whose archived text, or whose card on the page, links to an article, it now opens the article in the same X profile, after the same 4–10 second pause, and saves beside the post's capture: `article.json` (the title, author, date and every heading, paragraph, quote, list item, image, link and embedded post in reading order, an embedded post by its URL), `article.md` (the same as readable text), `article.png` (the whole article as shown, cut off at 16,000 pixels tall and marked `trimmed` when longer), `article.html` (the article as X served it, so it can be read again without going back to X) and the article's pictures as `article-img-1.jpg`, `article-img-2.png`, … at full size, fetched through the same browser session. `capture.json` records the article's state, title, block count and every file's size and SHA-256. `"articles": false` leaves articles alone; `"shots": false, "media": false, "articles": true` reads only the articles. An article already captured, deleted or unavailable is not opened again unless `"force": true`; one that failed is tried on the next run. If X asks to log in on the article page, the job stops there as it does for a post. The post viewer's capture panel in the editor shows the article's title with a link to `article.md`. - **A report video can play a clip that has only sound, and a clip can be a file beside the manifest.** When a clip's source has no picture, `build-video.mjs` plays it under a poster: a card with the clip's channel, title and date, the size of the picture area, with the sound's waveform moving along its foot (`render.audioPoster.waveform: false` keeps it still). The segment matches every other one in size, frame rate and sound, and the header, footer and on-screen deck are drawn over it as over footage. A video's saved sound (`audio.mp3` and the like in its folder) is now a source the build can cut from, after every saved picture: before any download when the clip has no picture to fetch (`"audioOnly": true` on the clip, `"preferLocalAudio": true` in `render`, a podcast or feed record, or a record with no page). A clip that should have a picture is not quietly played from its sound: with `--no-network` or `--skip-fetch`, one whose picture is missing still stops the build, listed as needing a download with a note that its sound is on disk, so `--no-network` still proves every picture is there. Add `--audio-fallback` to play such clips from their sound under the poster instead; they are listed apart (and logged as `audio-fallback`), and clips that have no picture to fetch are listed as playing from audio only rather than refused. A clip may also give `"src"` (a video or audio file) and `"cues"` (its transcript, either a `transcript.cues.json` or a `parakeet-stitch` transcript), both relative to the manifest, instead of a channel and video: it plays the whole file unless `start`/`end` cut inside it, `resolve-windows.mjs` widens it with those cues, it gets no QR unless it has a `citeUrl`, and a path that leaves the manifest's folder (or an absolute one, without `"allowAbsoluteSrc": true` in `render`), a missing file or an unreadable transcript stops the build before anything runs, naming the clip. - **A report build cuts from media already on disk before it downloads anything, and `--no-network` makes sure it never does.** For each clip, `build-video.mjs` now looks, in order, in the project's own `out/clips-raw`, in the clip windows the editor fetched into the channel (`channels/<slug>/data/<id>/clips/`), and in a saved whole source video (through the saved-video store's pointer, or a `source-media` file still in the video's folder), and cuts from the first that holds the clip plus its fetch pad; only when none does is the window downloaded. A file that is a link to a drive that is not mounted counts as not there, and the next place is tried. The build prints one line per clip naming where its source came from (`raw-cache`, `corpus-window`, `saved-video`, or a network fetch). With `--no-network`, every clip's source is found before anything is rendered, and if any clip would need a download the build stops at once and lists each one (its position in the timeline, channel, video and the span it needs). umtool's clip bench reads the same three places, so a clip it shows as fetched is one the build cuts from without downloading. - **umtool's report videos can show a highlighted sentence from a saved article.** `node umtool/report-to-video/shoot-page.mjs --page <saved page.html> --quote "<sentence>" --out <shot.png>` opens a web page saved to disk, finds the sentence in its text, highlights it and saves a PNG of the paragraph that holds it, ready to be a report manifest's `image` entry. `--batch <items.json> --out <dir>` does a list of `{ id, page, quote, context? }` at once and writes `<id>.png` for each plus a `results.json` recording each shot's crop, the matched text and the block it shot. The page is opened offline: nothing is fetched except files saved beside it, and its own scripts do not run unless `--js` is given. The sentence is found whether its quotes and apostrophes are curly or straight, across links and emphasis, and through non-breaking spaces, soft hyphens and line breaks in the page's source. A sentence that is not on the page is listed in `results.json` and on the terminal, and the run ends with an error rather than leaving it out. `--color` sets the highlight; `context` picks one occurrence of a sentence that appears more than once. On a page where a whole post is one block of paragraphs separated by line breaks, `--crop mark` (or an item's `"crop": "mark"`) shoots only the sentence's own lines and one whole line above and below (`--context-lines` sets how many) instead of the whole post; `results.json` records which crop each shot used. diff --git a/editor/app/api/ops/capture-posts/route.test.ts b/editor/app/api/ops/capture-posts/route.test.ts @@ -0,0 +1,73 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +// Run with: +// pnpm -C editor exec tsx --test "app/api/ops/capture-posts/route.test.ts" +// +// The body's shape and the refusals that come before any job: every case here +// is answered from the disk alone, so a temp corpus with one X channel is the +// whole world — no job is queued and nothing reaches X. + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "capture-posts-route-")); +const SLUG = "demo-x"; +const CHANNEL = path.join(ROOT, "channels", SLUG); +// Set before the route (and getPaths, which caches) is first imported. +process.env.WORKER_TOKEN = "test-token"; +process.env.TRANSCRIPTS_DIR = ROOT; +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); +await mkdir(CHANNEL, { recursive: true }); +await writeFile( + path.join(CHANNEL, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + platform: "twitter", + postFetcher: "x-gallery-dl", + socialHandle: "example_user", + name: "Example (X)", + url: "https://x.com/example_user", + }), +); +await writeFile(path.join(CHANNEL, "posts-archive"), "twitter 111\n"); +const { POST } = await import("./route"); +test.after(() => rm(ROOT, { recursive: true, force: true })); + +async function post(body: Record<string, unknown>): Promise<{ status: number; error: string }> { + const res = await POST( + new Request("http://localhost/api/ops/capture-posts", { + method: "POST", + headers: { + authorization: "Bearer test-token", + "content-type": "application/json", + }, + body: JSON.stringify(body), + }), + ); + return { status: res.status, error: ((await res.json()) as { error?: string }).error ?? "" }; +} + +test("articles must be a boolean; the route names it among its keys", async () => { + const bad = await post({ slug: SLUG, ids: ["111"], articles: "yes" }); + assert.equal(bad.status, 400); + assert.match(bad.error, /"articles" must be a boolean/); + const unknown = await post({ slug: SLUG, ids: ["111"], article: true }); + assert.equal(unknown.status, 400); + assert.match(unknown.error, /unknown key\(s\): article — this route accepts .*articles/); +}); + +test("both halves off is nothing to do unless the articles are asked for by name", async () => { + for (const articles of [undefined, false]) { + const res = await post({ slug: SLUG, ids: ["111"], shots: false, media: false, articles }); + assert.equal(res.status, 400, String(articles)); + assert.match(res.error, /both the screenshot and the media are turned off/); + } + // Asked for: past that check, to the next refusal — an id not archived. + const stray = await post({ slug: SLUG, ids: ["999"], shots: false, media: false, articles: true }); + assert.equal(stray.status, 400); + assert.equal(stray.error, "1 id(s) not in demo-x's posts archive: 999"); + // No job was ever written. + assert.deepEqual(await readdir(path.join(ROOT, ".jobs")).catch(() => []), []); +}); diff --git a/editor/app/api/ops/capture-posts/route.ts b/editor/app/api/ops/capture-posts/route.ts @@ -10,20 +10,24 @@ import { export const dynamic = "force-dynamic"; -// POST { slug, ids, shots?, media?, force?, queueKey? } -> { ok: true, jobId } +// POST { slug, ids, shots?, media?, force?, articles?, queueKey? } -> { ok: true, jobId } // // Capture specific archived posts of a social channel: a screenshot of each // (`shots`, default true) and its attached media (`media`, default true), into -// the channel's posts-media/<id>/. Posts already captured are skipped unless -// `force`. The job runs on the platform's queue, as a post fetch does. +// the channel's posts-media/<id>/. A post that links to an X Article (its +// archived text, or its card) also gets the article — read into article.json, +// article.md and article.png, with its images — unless `articles` is false +// (default true). Posts already captured are skipped unless `force`. The job +// runs on the platform's queue, as a post fetch does. // -// Every refusal is the action's own sentence: both halves off, a channel that +// Every refusal is the action's own sentence: both halves off (an +// articles-only run names `articles: true`), a channel that // is not a social one, a fetcher that cannot capture, an id not in the // channel's posts archive. export async function POST(request: Request) { return ops( request, - ["slug", "ids", "shots", "media", "force", "queueKey"], + ["slug", "ids", "shots", "media", "force", "articles", "queueKey"], async (body) => { const slug = reqSlug(body, "slug"); return jobResponse( @@ -34,6 +38,7 @@ export async function POST(request: Request) { optBool(body, "shots"), optBool(body, "media"), optBool(body, "force"), + optBool(body, "articles"), ), ); }, diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts @@ -34,6 +34,7 @@ import { capturePosts, capturePostsProblem, NOTHING_TO_CAPTURE, + nothingToCapture, strayCaptureIds, strayIdsRefusal, } from "yt-dlp-transcript-common/controller/capturePosts"; @@ -211,7 +212,8 @@ export async function fetchPostsAction( // two never run against the same source at once. Refused HERE, before a job // exists: nothing asked for, no ids, a channel that is not social, a fetcher // that cannot capture, or an id that is not in the channel's posts archive -// (named). Posts already captured are skipped unless `force`. +// (named). Posts already captured are skipped unless `force`. A post that +// links to an X Article gets the article too unless `articles` is false. export async function capturePostsAction( slug: string, ids: string[], @@ -219,8 +221,9 @@ export async function capturePostsAction( shots?: boolean, media?: boolean, force?: boolean, + articles?: boolean, ): Promise<StreamActionResult> { - if (shots === false && media === false) return { ok: false, error: NOTHING_TO_CAPTURE }; + if (nothingToCapture({ shots, media, articles })) return { ok: false, error: NOTHING_TO_CAPTURE }; const wanted = [...new Set(ids)]; if (wanted.length === 0) return { ok: false, error: "No post ids to capture." }; const paths = getPaths(); @@ -249,7 +252,7 @@ export async function capturePostsAction( spec: { kind: "capture-posts", slug, - params: { queueKey, ids: wanted, shots, media, force }, + params: { queueKey, ids: wanted, shots, media, force, articles }, }, fn: async (onLog, signal, _progress, ctx) => { const result = await capturePosts({ @@ -260,6 +263,7 @@ export async function capturePostsAction( shots, media, force, + articles, onLog, signal, drain: ctx.drainSignal, diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -273,6 +273,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { bool(p.shots), bool(p.media), bool(p.force), + bool(p.articles), ); }, "download-missing-subs": (spec) => { diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -337,7 +337,10 @@ export function usage() { ' media through gallery-dl, into the channel\'s posts-media/<id>/:', ' {"slug", "ids": [...]}. Every id must be in the channel\'s posts archive.', ' "shots": false or "media": false skips that half; posts already captured', - ' are skipped unless "force": true. Paced like a post fetch, on its queue.', + ' are skipped unless "force": true. A post that links to an X Article also', + ' gets the article (article.json, .md, .png and its images) unless', + ' "articles": false; both halves off with "articles": true reads only the', + ' articles. Paced like a post fetch, on its queue.', "", 'persist-videos saves specific videos, across channels, to the saved-video', ' store: {"items": [{"slug", "id"}, ...]}. "format": "original" |', diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -404,6 +404,7 @@ test("capture-posts is a POST to its route, named in the usage", () => { assert.deepEqual(p.body, { slug: "example-x", ids: ["123"], media: false }); assert.match(usage(), /Actions:.*fetch-posts, capture-posts/); assert.match(usage(), /Every id must be in the channel's posts archive/); + assert.match(usage(), /unless\s+"articles": false/); }); test("persist-videos is a POST to its route, named in the usage", () => {