commit 195dbff2fb7b2cc2a94e8f741b6de2775f7b24fa
parent 959369a21fb7ee323d6bebb953d42fc9cc665068
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 20:09:08 -0400
common: read an X Article's HTML into blocks and markdown; find a post's article link
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 752 insertions(+), 0 deletions(-)
diff --git a/common/social/__fixtures__/x-article.html b/common/social/__fixtures__/x-article.html
@@ -0,0 +1 @@
+<div data-testid="twitterArticleReadView" class="css-175oi2r" data-archilyzer-article=""><div class="css-175oi2r"><div data-testid="tweetPhoto" class="css-175oi2r"><img alt="Image" draggable="true" src="https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" class="css-9pa8cd"></div><div data-testid="twitter-article-title" dir="auto" class="css-146c3p1"><span class="css-1jxf684">A Worked Example & Its Notes</span></div><div class="css-175oi2r"><div data-testid="UserAvatar-Container-demo_author" class="css-175oi2r"><img alt="" src="https://pbs.twimg.com/profile_images/1/avatar_normal.jpg"></div><div data-testid="User-Name" class="css-175oi2r"><div class="css-175oi2r"><a href="/demo_author" role="link"><div dir="ltr"><span>Demo Author</span></div></a></div><div class="css-175oi2r"><a href="/demo_author" role="link" tabindex="-1"><div dir="ltr"><span>@demo_author</span></div></a><div aria-hidden="true" dir="ltr"><span>ยท</span></div><a href="/demo_author/article/1234567890" role="link"><time datetime="2026-01-02T03:04:05.000Z">Jan 2</time></a></div></div><div class="css-175oi2r"><button role="button" type="button"><span>Follow</span></button></div></div></div><div data-testid="twitterArticleRichTextView" class="css-175oi2r"><div data-testid="longformRichTextComponent" class="css-175oi2r"><div class="DraftEditor-root"><div class="DraftEditor-editorContainer"><div aria-describedby="placeholder" class="public-DraftEditor-content" contenteditable="false" role="textbox" spellcheck="false"><div data-contents="true"><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="a-0-0"><div data-offset-key="a-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="a-0-0"><span data-text="true">The first paragraph, with a </span></span><a href="https://example.com/source" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="a-1-0"><span data-text="true">linked source</span></span></a><span data-offset-key="a-2-0"><span data-text="true"> and an emoji </span></span><img alt="๐" draggable="false" src="https://abs-0.twimg.com/emoji/v2/svg/1f642.svg" class="r-4qtqp9"><span data-offset-key="a-3-0"><span data-text="true">.</span></span></div></div><h2 class="longform-header-two" data-block="true" data-editor="ed1" data-offset-key="b-0-0"><div data-offset-key="b-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="b-0-0"><span data-text="true">A Section Heading</span></span></div></h2><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="c-0-0"><div data-offset-key="c-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="c-0-0"><span data-text="true">A second paragraph that runs</span></span><br data-text="true"><span data-offset-key="c-0-1"><span data-text="true">onto a second line.</span></span></div></div><blockquote class="longform-blockquote" data-block="true" data-editor="ed1" data-offset-key="d-0-0"><div data-offset-key="d-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="d-0-0"><span data-text="true">A quoted passage, set apart.</span></span></div></blockquote><ul class="public-DraftStyleDefault-ul" data-offset-key="e-0-0"><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-reset public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="e-0-0"><div data-offset-key="e-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="e-0-0"><span data-text="true">The first point</span></span></div></li><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="f-0-0"><div data-offset-key="f-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="f-0-0"><span data-text="true"># The second point</span></span></div></li></ul><section data-block="true" data-editor="ed1" data-offset-key="g-0-0"><div class="longform-media" contenteditable="false"><a href="/demo_author/article/1234567890/media/555" role="link"><div data-testid="tweetPhoto"><img alt="A chart of the numbers" draggable="true" src="https://pbs.twimg.com/media/BodyImg1?format=png&name=small"></div></a></div></section><section data-block="true" data-editor="ed1" data-offset-key="h-0-0"><div contenteditable="false"><div data-testid="simpleTweet" class="css-175oi2r"><article aria-labelledby="id1" role="article" tabindex="-1" data-testid="tweet"><div data-testid="User-Name"><a href="/other_user"><span>Other User</span></a><a href="/other_user"><span>@other_user</span></a><a href="/other_user/status/4444444444" role="link"><time datetime="2025-12-01T00:00:00.000Z">Dec 1</time></a></div><div data-testid="tweetText" dir="auto" lang="en"><span>An embedded post's own words.</span></div><div role="group" aria-label="12 replies"><button data-testid="reply"><span>12</span></button></div><a href="/other_user/status/4444444444/analytics">Views</a></article></div></div></section><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="i-0-0"><div data-offset-key="i-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><a href="https://example.com/further-reading" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="i-0-0"><span data-text="true">Further reading</span></span></a></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="j-0-0"><div data-offset-key="j-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="j-0-0"><br data-text="true"></span></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="k-0-0"><div data-offset-key="k-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="k-0-0"><span data-text="true">The last word: 3 < 4.</span></span></div></div></div></div></div></div></div></div><div role="group" aria-label="40 replies, 7 reposts, 300 likes" class="css-175oi2r"><button data-testid="reply" type="button"><span>40</span></button><button data-testid="like" type="button"><span>300</span></button></div><div class="css-175oi2r"><span>A trailing note outside the body.</span></div></div>
diff --git a/common/social/xArticle.test.ts b/common/social/xArticle.test.ts
@@ -0,0 +1,204 @@
+// X Articles, the pure half: the article link in a post's text or on its card,
+// the article root's HTML read into blocks (against a written fixture, never
+// x.com), and the markdown written from them.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticle.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ decodeEntities,
+ extractXArticle,
+ findXArticleLink,
+ parseHtml,
+ textOf,
+ xArticleLinkFromArchive,
+ xArticleMarkdown,
+} from "./xArticle";
+import { xArticleLinkFromCard } from "./xArticleCapture";
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8");
+
+// --- the link ----------------------------------------------------------------
+
+test("link: an article post's archived text, its expanded links, or nothing", () => {
+ assert.deepEqual(xArticleLinkFromArchive({ text: "https://x.com/i/article/1987654321" }), {
+ articleId: "1987654321",
+ url: "https://x.com/i/article/1987654321",
+ });
+ assert.deepEqual(
+ xArticleLinkFromArchive({ text: "Read it: https://t.co/abc", links: ["https://twitter.com/i/article/42"] }),
+ { articleId: "42", url: "https://x.com/i/article/42" },
+ );
+ assert.deepEqual(xArticleLinkFromArchive({ text: "x.com/demo_author/article/77 is up" }), {
+ articleId: "77",
+ url: "https://x.com/demo_author/article/77",
+ });
+ for (const text of [
+ "a plain post",
+ "https://x.com/demo_author/status/123",
+ "https://example.com/i/article/5",
+ "https://x.com/i/articles",
+ ]) {
+ assert.equal(xArticleLinkFromArchive({ text }), null, text);
+ }
+ assert.equal(xArticleLinkFromArchive(undefined), null);
+});
+
+test("link: a card's hrefs, relative as the page has them", () => {
+ assert.deepEqual(xArticleLinkFromCard(["/demo_author", "/demo_author/article/1234567890"]), {
+ articleId: "1234567890",
+ url: "https://x.com/demo_author/article/1234567890",
+ });
+ assert.deepEqual(xArticleLinkFromCard(["/i/article/99/media/1"]), {
+ articleId: "99",
+ url: "https://x.com/i/article/99",
+ });
+ assert.equal(xArticleLinkFromCard(["/demo_author/status/1/photo/1"]), null);
+ assert.equal(xArticleLinkFromCard([]), null);
+ assert.equal(xArticleLinkFromCard(undefined), null);
+ assert.equal(findXArticleLink([null, undefined, ""]), null);
+});
+
+// --- the HTML reader ---------------------------------------------------------
+
+test("html: elements, attributes, void and self-closing tags, entities, comments, raw text", () => {
+ const root = parseHtml(
+ '<!-- c --><div class="a" data-x=\'q\' hidden><img src="s?a=1&b=2"><br/>a <b> 'c' d' +
+ "<script>if (a < b) {}</script><span>e</span></p></div>tail",
+ );
+ const div = root.children[0] as { tag: string; attrs: Record<string, string>; children: unknown[] };
+ assert.equal(div.tag, "div");
+ assert.deepEqual(div.attrs, { class: "a", "data-x": "q", hidden: "" });
+ assert.equal((div.children[0] as { attrs: { src: string } }).attrs.src, "s?a=1&b=2");
+ // A stray end tag is ignored; the text after the div is the root's.
+ assert.equal(root.children[1], "tail");
+ assert.equal(textOf(div as never), "a <b> 'c' de");
+ assert.equal(decodeEntities("&&unknown;🙂"), "&&unknown;๐");
+});
+
+test("html: text reads as innerText would โ line breaks at <br> and blocks, script and svg silent", () => {
+ const root = parseHtml("<div><p>one <b>two</b></p><p>three<br>four</p><svg><text>no</text></svg><script>x</script></div>");
+ assert.equal(textOf(root), "one two\nthree\nfour");
+});
+
+// --- the extractor -------------------------------------------------------------
+
+test("extract: title, byline and the body's blocks in reading order, from the written fixture", () => {
+ const a = extractXArticle(FIXTURE);
+ assert.equal(a.extraction, "structured");
+ assert.equal(a.title, "A Worked Example & Its Notes");
+ assert.equal(a.author, "Demo Author");
+ assert.equal(a.handle, "@demo_author");
+ assert.equal(a.publishedAt, "2026-01-02T03:04:05.000Z");
+ assert.deepEqual(a.blocks, [
+ // The cover, above the body.
+ { type: "image", src: "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" },
+ { type: "paragraph", text: "The first paragraph, with a linked source and an emoji ๐." },
+ { type: "link", text: "linked source", href: "https://example.com/source" },
+ { type: "heading", text: "A Section Heading" },
+ { type: "paragraph", text: "A second paragraph that runs\nonto a second line." },
+ { type: "quote", text: "A quoted passage, set apart." },
+ { type: "list-item", text: "The first point" },
+ { type: "list-item", text: "# The second point" },
+ {
+ type: "image",
+ src: "https://pbs.twimg.com/media/BodyImg1?format=png&name=small",
+ text: "A chart of the numbers",
+ },
+ {
+ type: "embedded-post",
+ text: "An embedded post's own words.",
+ href: "https://x.com/other_user/status/4444444444",
+ },
+ // A paragraph that is only a link is the link.
+ { type: "link", text: "Further reading", href: "https://example.com/further-reading" },
+ { type: "paragraph", text: "The last word: 3 < 4." },
+ ]);
+});
+
+test("extract: no body marker โ the root's text split at block elements, and it says so", () => {
+ const a = extractXArticle(
+ '<article><h1>Plain Title</h1><div><p>One</p><p>Two <a href="/demo_author">a handle</a></p>' +
+ '<figure><img src="https://pbs.twimg.com/media/Fig?format=webp"></figure><ul><li>Item</li></ul>' +
+ '<blockquote>Said</blockquote></div><div role="group">40 likes</div></article>',
+ );
+ assert.equal(a.extraction, "fallback");
+ assert.equal(a.title, "Plain Title");
+ assert.deepEqual(a.blocks, [
+ { type: "paragraph", text: "One" },
+ { type: "paragraph", text: "Two a handle" },
+ { type: "link", text: "a handle", href: "https://x.com/demo_author" },
+ { type: "image", src: "https://pbs.twimg.com/media/Fig?format=webp" },
+ { type: "list-item", text: "Item" },
+ { type: "quote", text: "Said" },
+ ]);
+});
+
+test("extract: an empty root is an untitled article with no blocks", () => {
+ assert.deepEqual(extractXArticle(""), { extraction: "fallback", blocks: [] });
+});
+
+// --- markdown ----------------------------------------------------------------
+
+test("markdown: title, byline, the URL, then each block โ saved images linked locally, markdown in text escaped", () => {
+ const a = extractXArticle(FIXTURE);
+ const blocks = a.blocks.map((b) =>
+ b.src?.includes("BodyImg1") ? { ...b, file: "article-img-2.png" } : b,
+ );
+ const md = xArticleMarkdown({ ...a, blocks, url: "https://x.com/i/article/1234567890" });
+ assert.equal(
+ md,
+ [
+ "# A Worked Example & Its Notes",
+ "",
+ "By Demo Author @demo_author ยท 2026-01-02T03:04:05.000Z",
+ "",
+ "<https://x.com/i/article/1234567890>",
+ "",
+ "",
+ "",
+ "The first paragraph, with a linked source and an emoji ๐.",
+ "",
+ "[linked source](<https://example.com/source>)",
+ "",
+ "## A Section Heading",
+ "",
+ "A second paragraph that runs",
+ "onto a second line.",
+ "",
+ "> A quoted passage, set apart.",
+ "",
+ "- The first point",
+ "- \\# The second point",
+ "",
+ "",
+ "",
+ "> Embedded post: <https://x.com/other_user/status/4444444444>",
+ ">",
+ "> An embedded post's own words.",
+ "",
+ "[Further reading](<https://example.com/further-reading>)",
+ "",
+ "The last word: 3 < 4.",
+ "",
+ ].join("\n"),
+ );
+});
+
+test("markdown: an untitled article, no byline, a paragraph that would read as syntax", () => {
+ const md = xArticleMarkdown({
+ extraction: "fallback",
+ url: "https://x.com/i/article/1",
+ blocks: [{ type: "paragraph", text: "1. not a list\n> not a quote" }, { type: "link", href: "https://example.com/a b" }],
+ });
+ assert.equal(
+ md,
+ "# Untitled article\n\n<https://x.com/i/article/1>\n\n1\\. not a list\n\\> not a quote\n\n" +
+ "[https://example.com/a b](<https://example.com/a%20b>)\n",
+ );
+});
diff --git a/common/social/xArticle.ts b/common/social/xArticle.ts
@@ -0,0 +1,547 @@
+// X Articles (long-form posts), the pure half: finding a post's article link,
+// reading the article's HTML into blocks, and writing those blocks as
+// markdown. No browser, no network, no fs โ xArticleCapture.ts loads the page
+// and hands this module the article root's outerHTML.
+//
+// An article post archives as text that is only its link
+// (`https://x.com/i/article/<id>`): gallery-dl cannot read an article's body,
+// and the post's card shows a title and a preview. The article itself is a
+// page of its own, which the capture opens.
+//
+// WHY HTML AND NOT A WALK IN THE PAGE. The browser hands back the root's
+// outerHTML once, and everything after that is parsed and read here, so the
+// whole extractor runs in tests against a saved HTML fixture, and the HTML is
+// kept beside the capture (`article.html`) so a better reading later costs no
+// second visit to X.
+//
+// NOT VERIFIED AGAINST LIVE X. The markers below (data-testid names, the
+// Draft.js block classes) are X's as of this writing; every one is optional,
+// and an article whose body marker is missing is read by the fallback โ the
+// root's text split at block elements โ and says so (`extraction`).
+
+// --- the link ------------------------------------------------------------------
+
+export type XArticleLink = {
+ // The id X's article URL carries.
+ articleId: string;
+ url: string;
+};
+
+// `x.com/i/article/<id>`, or `x.com/<handle>/article/<id>` (the form a card
+// links to), absolute or as a path.
+const ARTICLE_URL_RE =
+ /(?:https?:\/\/)?(?:www\.|mobile\.)?(?:x|twitter)\.com\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/i;
+const ARTICLE_PATH_RE = /^\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/;
+
+function linkFrom(owner: string, articleId: string): XArticleLink {
+ return { articleId, url: `https://x.com/${owner}/article/${articleId}` };
+}
+
+// The first article link in these strings (a post's text, its expanded links,
+// a card's hrefs), or null.
+export function findXArticleLink(
+ candidates: ReadonlyArray<string | null | undefined>,
+): XArticleLink | null {
+ for (const c of candidates) {
+ if (!c) continue;
+ const m = ARTICLE_URL_RE.exec(c) ?? ARTICLE_PATH_RE.exec(c.trim());
+ if (m) return linkFrom(m[1].toLowerCase() === "i" ? "i" : m[1], m[2]);
+ }
+ return null;
+}
+
+// What the posts archive holds for a post, as far as its links go.
+export type ArchivedPostText = { text: string; links?: ReadonlyArray<string> };
+
+export function xArticleLinkFromArchive(
+ archived: ArchivedPostText | undefined,
+): XArticleLink | null {
+ if (!archived) return null;
+ return findXArticleLink([archived.text, ...(archived.links ?? [])]);
+}
+
+// --- a small HTML reader -------------------------------------------------------
+//
+// For HTML a browser serialised (outerHTML): attributes are double-quoted, void
+// elements are unclosed, text escapes only & < > and nbsp. It tolerates more
+// (single quotes, bare values, stray end tags), but it is not a general parser.
+
+export type HtmlElement = {
+ tag: string;
+ attrs: Record<string, string>;
+ children: HtmlNode[];
+};
+export type HtmlNode = HtmlElement | string;
+
+const VOID_TAGS = new Set([
+ "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta",
+ "param", "source", "track", "wbr",
+]);
+const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]);
+
+const NAMED_ENTITIES: Record<string, string> = {
+ amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0",
+ hellip: "โฆ", mdash: "โ", ndash: "โ", lsquo: "โ", rsquo: "โ", ldquo: "โ",
+ rdquo: "โ", copy: "ยฉ", reg: "ยฎ", trade: "โข",
+};
+
+export function decodeEntities(s: string): string {
+ return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => {
+ if (name[0] === "#") {
+ const code =
+ name[1] === "x" || name[1] === "X"
+ ? parseInt(name.slice(2), 16)
+ : parseInt(name.slice(1), 10);
+ return Number.isFinite(code) && code > 0 && code <= 0x10ffff
+ ? String.fromCodePoint(code)
+ : whole;
+ }
+ return NAMED_ENTITIES[name.toLowerCase()] ?? whole;
+ });
+}
+
+const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y;
+const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y;
+
+// The document's top-level nodes, under a synthetic root element.
+export function parseHtml(html: string): HtmlElement {
+ const root: HtmlElement = { tag: "#root", attrs: {}, children: [] };
+ const stack: HtmlElement[] = [root];
+ const top = () => stack[stack.length - 1];
+ let i = 0;
+ while (i < html.length) {
+ const lt = html.indexOf("<", i);
+ if (lt < 0) {
+ top().children.push(decodeEntities(html.slice(i)));
+ break;
+ }
+ if (lt > i) top().children.push(decodeEntities(html.slice(i, lt)));
+ i = lt;
+ if (html.startsWith("<!--", i)) {
+ const end = html.indexOf("-->", i + 4);
+ i = end < 0 ? html.length : end + 3;
+ continue;
+ }
+ if (html[i + 1] === "!" || html[i + 1] === "?") {
+ const end = html.indexOf(">", i);
+ i = end < 0 ? html.length : end + 1;
+ continue;
+ }
+ if (html[i + 1] === "/") {
+ const end = html.indexOf(">", i);
+ const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase();
+ i = end < 0 ? html.length : end + 1;
+ // Close up to the matching open element; a stray end tag is ignored.
+ for (let k = stack.length - 1; k > 0; k--) {
+ if (stack[k].tag === name) {
+ stack.length = k;
+ break;
+ }
+ }
+ continue;
+ }
+ TAG_NAME_RE.lastIndex = i + 1;
+ const nameMatch = TAG_NAME_RE.exec(html);
+ if (!nameMatch) {
+ // A "<" that opens no tag is text.
+ top().children.push("<");
+ i++;
+ continue;
+ }
+ const tag = nameMatch[0].toLowerCase();
+ let j = TAG_NAME_RE.lastIndex;
+ const attrs: Record<string, string> = {};
+ for (;;) {
+ ATTR_RE.lastIndex = j;
+ const a = ATTR_RE.exec(html);
+ if (!a || a[0].length === 0) break;
+ attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? "");
+ j = ATTR_RE.lastIndex;
+ }
+ const close = html.indexOf(">", j);
+ const selfClosing = close > 0 && html[close - 1] === "/";
+ i = close < 0 ? html.length : close + 1;
+ const el: HtmlElement = { tag, attrs, children: [] };
+ top().children.push(el);
+ if (RAW_TEXT_TAGS.has(tag)) {
+ const end = html.toLowerCase().indexOf(`</${tag}`, i);
+ const stop = end < 0 ? html.length : end;
+ if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop)));
+ const gt = end < 0 ? -1 : html.indexOf(">", end);
+ i = gt < 0 ? html.length : gt + 1;
+ continue;
+ }
+ if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el);
+ }
+ return root;
+}
+
+// --- reading the article ------------------------------------------------------
+
+export type XArticleBlockType =
+ | "heading"
+ | "paragraph"
+ | "quote"
+ | "list-item"
+ | "image"
+ | "embedded-post"
+ | "link";
+
+export type XArticleBlock = {
+ type: XArticleBlockType;
+ text?: string;
+ href?: string;
+ src?: string;
+ // An image saved beside the capture: its file name there.
+ file?: string;
+};
+
+export type XArticleContent = {
+ title?: string;
+ author?: string;
+ handle?: string;
+ publishedAt?: string;
+ // "structured": the body marker was found and read block by block;
+ // "fallback": it was not, and the root's text was split at block elements.
+ extraction: "structured" | "fallback";
+ blocks: XArticleBlock[];
+};
+
+const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string";
+const testid = (el: HtmlElement) => el.attrs["data-testid"] ?? "";
+const classes = (el: HtmlElement) => el.attrs["class"] ?? "";
+
+function findFirst(
+ el: HtmlElement,
+ pred: (e: HtmlElement) => boolean,
+ skip?: (e: HtmlElement) => boolean,
+): HtmlElement | null {
+ for (const c of el.children) {
+ if (!isEl(c)) continue;
+ if (pred(c)) return c;
+ if (skip?.(c)) continue;
+ const hit = findFirst(c, pred, skip);
+ if (hit) return hit;
+ }
+ return null;
+}
+
+function findAll(el: HtmlElement, pred: (e: HtmlElement) => boolean, out: HtmlElement[] = []): HtmlElement[] {
+ for (const c of el.children) {
+ if (!isEl(c)) continue;
+ if (pred(c)) out.push(c);
+ findAll(c, pred, out);
+ }
+ return out;
+}
+
+const SILENT_TAGS = new Set([
+ "script", "style", "svg", "noscript", "template", "button", "input", "select",
+ "video", "audio", "iframe", "canvas",
+]);
+
+// An element's text as innerText would give it, near enough: `<br>` and block
+// boundaries are line breaks (never a blank line), runs of other whitespace are
+// one space, an emoji drawn as an image is its alt.
+export function textOf(node: HtmlNode): string {
+ const parts: string[] = [];
+ const walk = (n: HtmlNode) => {
+ if (!isEl(n)) {
+ parts.push(n.replace(/[\s\u00a0]+/g, " "));
+ return;
+ }
+ if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true") return;
+ if (n.tag === "br") {
+ parts.push("\n");
+ return;
+ }
+ if (n.tag === "img") {
+ parts.push(emojiText(n));
+ return;
+ }
+ const block = BLOCK_TAGS.has(n.tag);
+ if (block) parts.push("\n");
+ for (const c of n.children) walk(c);
+ if (block) parts.push("\n");
+ };
+ walk(node);
+ return parts
+ .join("")
+ .split("\n")
+ .map((l) => l.replace(/ +/g, " ").trim())
+ .filter(Boolean)
+ .join("\n");
+}
+
+const BLOCK_TAGS = new Set([
+ "address", "article", "aside", "blockquote", "dd", "details", "div", "dl",
+ "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5",
+ "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section",
+ "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul",
+]);
+
+// The article's body: X's long-form rich-text view, else the Draft.js content
+// it is built on.
+const BODY_TESTIDS = new Set(["longformRichTextComponent", "twitterArticleRichTextView"]);
+const isBody = (el: HtmlElement) =>
+ BODY_TESTIDS.has(testid(el)) || el.attrs["data-contents"] === "true";
+
+const isTitle = (el: HtmlElement) => testid(el) === "twitter-article-title";
+const isByline = (el: HtmlElement) => testid(el) === "User-Name";
+// The chrome around an article that is not its text.
+const isChrome = (el: HtmlElement) =>
+ /^(UserAvatar|UserAvatar-Container|reply|retweet|like|bookmark|caret|app-text-transition-container)/.test(
+ testid(el),
+ ) || el.attrs["role"] === "group";
+
+const EMBED_TESTIDS = new Set(["tweet", "simpleTweet", "quoteTweet"]);
+const isEmbeddedPost = (el: HtmlElement) =>
+ EMBED_TESTIDS.has(testid(el)) ||
+ (el.tag === "blockquote" && /\btwitter-tweet\b/.test(classes(el)));
+
+const STATUS_RE =
+ /^(?:https?:\/\/(?:www\.|mobile\.)?(?:x|twitter)\.com)?\/([A-Za-z0-9_]{1,15})\/status\/(\d{1,25})/i;
+
+// An embedded post's status URL: the link around its timestamp, else its first
+// status link.
+function statusUrlIn(el: HtmlElement): string | undefined {
+ const links = findAll(el, (e) => e.tag === "a" && STATUS_RE.test(e.attrs.href ?? ""));
+ const best = links.find((a) => findFirst(a, (e) => e.tag === "time")) ?? links[0];
+ const m = best ? STATUS_RE.exec(best.attrs.href) : null;
+ return m ? `https://x.com/${m[1]}/status/${m[2]}` : undefined;
+}
+
+// An image that is the article's, not an emoji, an avatar or an icon.
+export function isContentImageSrc(src: string | undefined): src is string {
+ if (!src || src.startsWith("data:")) return false;
+ if (/\/emoji\/|\/hashflags\/|profile_images|profile_banners|\.svg(\?|$)/i.test(src)) return false;
+ return /^https?:\/\//i.test(src);
+}
+
+// X draws an emoji as an image whose alt is the emoji: its text.
+function emojiText(img: HtmlElement): string {
+ return /\/emoji\//.test(img.attrs.src ?? "") ? (img.attrs.alt ?? "") : "";
+}
+
+// An href as an absolute URL, or undefined for one that goes nowhere.
+function absoluteHref(href: string | undefined): string | undefined {
+ if (!href) return undefined;
+ const h = href.trim();
+ if (h === "" || h.startsWith("#") || /^(javascript|mailto|data):/i.test(h)) {
+ return /^mailto:/i.test(h) ? h : undefined;
+ }
+ if (/^https?:\/\//i.test(h)) return h;
+ if (h.startsWith("//")) return `https:${h}`;
+ if (h.startsWith("/")) return `https://x.com${h}`;
+ return undefined;
+}
+
+function blockTypeOf(el: HtmlElement): "heading" | "quote" | "list-item" | null {
+ const c = classes(el);
+ if (/^h[1-6]$/.test(el.tag) || /\blongform-header-/.test(c)) return "heading";
+ if (el.tag === "blockquote" || /\blongform-blockquote\b/.test(c)) return "quote";
+ if (el.tag === "li" || /\blongform-(un)?ordered-list-item\b/.test(c)) return "list-item";
+ return null;
+}
+
+// The blocks of `body` in reading order. Text outside any typed block gathers
+// into a paragraph that ends at the next block boundary; a link inside text is
+// kept as a link block after it (a paragraph that is only a link is just the
+// link).
+function readBlocks(body: HtmlElement, skip: (el: HtmlElement) => boolean): XArticleBlock[] {
+ const blocks: XArticleBlock[] = [];
+ let inline: string[] = [];
+ let links: XArticleBlock[] = [];
+
+ const pushLinks = (from: HtmlElement) => {
+ for (const a of findAll(from, (e) => e.tag === "a")) {
+ if (findFirst(a, (e) => e.tag === "img")) continue;
+ const href = absoluteHref(a.attrs.href);
+ const text = textOf(a);
+ if (href) blocks.push({ type: "link", ...(text ? { text } : {}), href });
+ }
+ };
+ const flush = () => {
+ const text = inline
+ .join("")
+ .split("\n")
+ .map((l) => l.replace(/[ \t\u00a0]+/g, " ").trim())
+ .filter(Boolean)
+ .join("\n");
+ inline = [];
+ const onlyLink = links.length === 1 && links[0].text === text;
+ if (text && !onlyLink) blocks.push({ type: "paragraph", text });
+ blocks.push(...links);
+ links = [];
+ };
+
+ const walk = (n: HtmlNode) => {
+ if (!isEl(n)) {
+ inline.push(n.replace(/[\s\u00a0]+/g, " "));
+ return;
+ }
+ if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true" || skip(n)) return;
+ if (n.tag === "br") {
+ inline.push("\n");
+ return;
+ }
+ if (isEmbeddedPost(n)) {
+ flush();
+ const textEl = findFirst(n, (e) => testid(e) === "tweetText");
+ const text = textEl ? textOf(textEl) : "";
+ const href = statusUrlIn(n);
+ blocks.push({ type: "embedded-post", ...(text ? { text } : {}), ...(href ? { href } : {}) });
+ return;
+ }
+ if (n.tag === "img") {
+ if (!isContentImageSrc(n.attrs.src)) {
+ inline.push(emojiText(n));
+ return;
+ }
+ flush();
+ const alt = (n.attrs.alt ?? "").trim();
+ blocks.push({ type: "image", src: n.attrs.src, ...(alt && alt.toLowerCase() !== "image" ? { text: alt } : {}) });
+ return;
+ }
+ const typed = blockTypeOf(n);
+ if (typed) {
+ flush();
+ const text = textOf(n);
+ if (text) blocks.push({ type: typed, text });
+ pushLinks(n);
+ // An image inside a typed block (rare) is still the article's.
+ for (const img of findAll(n, (e) => e.tag === "img" && isContentImageSrc(e.attrs.src))) {
+ blocks.push({ type: "image", src: img.attrs.src });
+ }
+ return;
+ }
+ if (n.tag === "a" && !findFirst(n, (e) => e.tag === "img")) {
+ const href = absoluteHref(n.attrs.href);
+ const text = textOf(n);
+ inline.push(text);
+ if (href) links.push({ type: "link", ...(text ? { text } : {}), href });
+ return;
+ }
+ const block = BLOCK_TAGS.has(n.tag);
+ if (block) flush();
+ for (const c of n.children) walk(c);
+ if (block) flush();
+ };
+ walk(body);
+ flush();
+ return blocks;
+}
+
+// The article root's HTML, read: title, byline, and the body's blocks.
+export function extractXArticle(html: string): XArticleContent {
+ const root = parseHtml(html);
+ const titleEl = findFirst(root, isTitle);
+ const fallbackTitle = titleEl ? null : findFirst(root, (e) => e.tag === "h1");
+ const title = textOf(titleEl ?? fallbackTitle ?? { tag: "#none", attrs: {}, children: [] }) || undefined;
+
+ const bylineEl = findFirst(root, isByline, isEmbeddedPost);
+ let author: string | undefined;
+ let handle: string | undefined;
+ if (bylineEl) {
+ const lines = textOf(bylineEl).split(/\n|ยท/).map((s) => s.trim()).filter(Boolean);
+ handle = lines.find((l) => /^@[A-Za-z0-9_]{1,15}$/.test(l));
+ author = lines.find((l) => !l.startsWith("@") && l !== handle);
+ }
+ const time = findFirst(root, (e) => e.tag === "time" && !!e.attrs.datetime, isEmbeddedPost);
+ const publishedAt = time?.attrs.datetime;
+
+ const body = findFirst(root, isBody);
+ const skipHeader = (el: HtmlElement) =>
+ el === titleEl || el === fallbackTitle || isByline(el) || isChrome(el);
+ let blocks: XArticleBlock[];
+ if (body) {
+ // A cover image sits above the body: the images before it are the
+ // article's; nothing after the body (the engagement bar, replies) is.
+ const cover: XArticleBlock[] = [];
+ const before = (el: HtmlElement): boolean => {
+ for (const c of el.children) {
+ if (!isEl(c)) continue;
+ if (c === body) return true;
+ if (isEmbeddedPost(c) || skipHeader(c)) continue;
+ if (c.tag === "img" && isContentImageSrc(c.attrs.src)) {
+ cover.push({ type: "image", src: c.attrs.src });
+ continue;
+ }
+ if (before(c)) return true;
+ }
+ return false;
+ };
+ before(root);
+ blocks = [...cover, ...readBlocks(body, isChrome)];
+ } else {
+ blocks = readBlocks(root, skipHeader);
+ }
+ return {
+ ...(title ? { title } : {}),
+ ...(author ? { author } : {}),
+ ...(handle ? { handle } : {}),
+ ...(publishedAt ? { publishedAt } : {}),
+ extraction: body ? "structured" : "fallback",
+ blocks,
+ };
+}
+
+// --- markdown ----------------------------------------------------------------
+
+// A line that would read as markdown syntax, escaped.
+function mdText(s: string): string {
+ return s
+ .split("\n")
+ .map((l) =>
+ l.replace(/^(\s*)([#>+\-*])(\s)/, "$1\\$2$3").replace(/^(\s*)(\d+)([.)])(\s)/, "$1$2\\$3$4"),
+ )
+ .join("\n");
+}
+const mdLabel = (s: string) => s.replace(/([\[\]\\])/g, "\\$1").replace(/\n+/g, " ");
+const mdDest = (u: string) => `<${u.replace(/[<>\s]/g, encodeURIComponent)}>`;
+
+// An image block links to its saved file when it has one, else to its src.
+export type XArticleMarkdownInput = XArticleContent & { url: string };
+
+export function xArticleMarkdown(a: XArticleMarkdownInput): string {
+ const oneLine = (t: string | undefined) => (t ?? "").replace(/\s*\n\s*/g, " ").trim();
+ let md = `# ${oneLine(a.title) || "Untitled article"}`;
+ const add = (chunk: string, tight = false) => {
+ md += (tight ? "\n" : "\n\n") + chunk;
+ };
+ const by = [a.author, a.handle].filter(Boolean).join(" ");
+ const meta = [by ? `By ${by}` : "", a.publishedAt ?? ""].filter(Boolean).join(" ยท ");
+ if (meta) add(mdText(meta));
+ add(mdDest(a.url));
+ let prev: XArticleBlockType | undefined;
+ for (const b of a.blocks) {
+ switch (b.type) {
+ case "heading":
+ add(`## ${oneLine(b.text)}`);
+ break;
+ case "quote":
+ add((b.text ?? "").split("\n").map((l) => `> ${l}`.trimEnd()).join("\n"));
+ break;
+ case "list-item":
+ // List items run together.
+ add(`- ${mdText(b.text ?? "").replace(/\n/g, "\n ")}`, prev === "list-item");
+ break;
+ case "image":
+ add(`})`);
+ break;
+ case "embedded-post":
+ add(
+ `> Embedded post: ${b.href ? mdDest(b.href) : "(no link)"}` +
+ (b.text ? "\n>\n" + b.text.split("\n").map((l) => `> ${l}`.trimEnd()).join("\n") : ""),
+ );
+ break;
+ case "link":
+ add(`[${mdLabel(b.text || b.href || "")}](${mdDest(b.href ?? "")})`);
+ break;
+ default:
+ add(mdText(b.text ?? ""));
+ }
+ prev = b.type;
+ }
+ return md + "\n";
+}