Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 195dbff2fb7b2cc2a94e8f741b6de2775f7b24fa
parent 959369a21fb7ee323d6bebb953d42fc9cc665068
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 20:09:08 -0400

common: read an X Article's HTML into blocks and markdown; find a post's article link

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/social/__fixtures__/x-article.html | 1+
Acommon/social/xArticle.test.ts | 204+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xArticle.ts | 547+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 752 insertions(+), 0 deletions(-)

diff --git a/common/social/__fixtures__/x-article.html b/common/social/__fixtures__/x-article.html @@ -0,0 +1 @@ +<div data-testid="twitterArticleReadView" class="css-175oi2r" data-archilyzer-article=""><div class="css-175oi2r"><div data-testid="tweetPhoto" class="css-175oi2r"><img alt="Image" draggable="true" src="https://pbs.twimg.com/media/CoverAbc?format=jpg&amp;name=small" class="css-9pa8cd"></div><div data-testid="twitter-article-title" dir="auto" class="css-146c3p1"><span class="css-1jxf684">A Worked Example &amp; Its Notes</span></div><div class="css-175oi2r"><div data-testid="UserAvatar-Container-demo_author" class="css-175oi2r"><img alt="" src="https://pbs.twimg.com/profile_images/1/avatar_normal.jpg"></div><div data-testid="User-Name" class="css-175oi2r"><div class="css-175oi2r"><a href="/demo_author" role="link"><div dir="ltr"><span>Demo Author</span></div></a></div><div class="css-175oi2r"><a href="/demo_author" role="link" tabindex="-1"><div dir="ltr"><span>@demo_author</span></div></a><div aria-hidden="true" dir="ltr"><span>ยท</span></div><a href="/demo_author/article/1234567890" role="link"><time datetime="2026-01-02T03:04:05.000Z">Jan 2</time></a></div></div><div class="css-175oi2r"><button role="button" type="button"><span>Follow</span></button></div></div></div><div data-testid="twitterArticleRichTextView" class="css-175oi2r"><div data-testid="longformRichTextComponent" class="css-175oi2r"><div class="DraftEditor-root"><div class="DraftEditor-editorContainer"><div aria-describedby="placeholder" class="public-DraftEditor-content" contenteditable="false" role="textbox" spellcheck="false"><div data-contents="true"><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="a-0-0"><div data-offset-key="a-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="a-0-0"><span data-text="true">The first paragraph, with a </span></span><a href="https://example.com/source" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="a-1-0"><span data-text="true">linked source</span></span></a><span data-offset-key="a-2-0"><span data-text="true"> and an&nbsp;emoji </span></span><img alt="๐Ÿ™‚" draggable="false" src="https://abs-0.twimg.com/emoji/v2/svg/1f642.svg" class="r-4qtqp9"><span data-offset-key="a-3-0"><span data-text="true">.</span></span></div></div><h2 class="longform-header-two" data-block="true" data-editor="ed1" data-offset-key="b-0-0"><div data-offset-key="b-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="b-0-0"><span data-text="true">A Section Heading</span></span></div></h2><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="c-0-0"><div data-offset-key="c-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="c-0-0"><span data-text="true">A second paragraph that runs</span></span><br data-text="true"><span data-offset-key="c-0-1"><span data-text="true">onto a second line.</span></span></div></div><blockquote class="longform-blockquote" data-block="true" data-editor="ed1" data-offset-key="d-0-0"><div data-offset-key="d-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="d-0-0"><span data-text="true">A quoted passage, set apart.</span></span></div></blockquote><ul class="public-DraftStyleDefault-ul" data-offset-key="e-0-0"><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-reset public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="e-0-0"><div data-offset-key="e-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="e-0-0"><span data-text="true">The first point</span></span></div></li><li class="longform-unordered-list-item public-DraftStyleDefault-unorderedListItem public-DraftStyleDefault-depth0 public-DraftStyleDefault-listLTR" data-block="true" data-editor="ed1" data-offset-key="f-0-0"><div data-offset-key="f-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="f-0-0"><span data-text="true"># The second point</span></span></div></li></ul><section data-block="true" data-editor="ed1" data-offset-key="g-0-0"><div class="longform-media" contenteditable="false"><a href="/demo_author/article/1234567890/media/555" role="link"><div data-testid="tweetPhoto"><img alt="A chart of the numbers" draggable="true" src="https://pbs.twimg.com/media/BodyImg1?format=png&amp;name=small"></div></a></div></section><section data-block="true" data-editor="ed1" data-offset-key="h-0-0"><div contenteditable="false"><div data-testid="simpleTweet" class="css-175oi2r"><article aria-labelledby="id1" role="article" tabindex="-1" data-testid="tweet"><div data-testid="User-Name"><a href="/other_user"><span>Other User</span></a><a href="/other_user"><span>@other_user</span></a><a href="/other_user/status/4444444444" role="link"><time datetime="2025-12-01T00:00:00.000Z">Dec 1</time></a></div><div data-testid="tweetText" dir="auto" lang="en"><span>An embedded post's own words.</span></div><div role="group" aria-label="12 replies"><button data-testid="reply"><span>12</span></button></div><a href="/other_user/status/4444444444/analytics">Views</a></article></div></div></section><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="i-0-0"><div data-offset-key="i-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><a href="https://example.com/further-reading" rel="noopener noreferrer nofollow" target="_blank"><span data-offset-key="i-0-0"><span data-text="true">Further reading</span></span></a></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="j-0-0"><div data-offset-key="j-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="j-0-0"><br data-text="true"></span></div></div><div class="longform-unstyled" data-block="true" data-editor="ed1" data-offset-key="k-0-0"><div data-offset-key="k-0-0" class="public-DraftStyleDefault-block public-DraftStyleDefault-ltr"><span data-offset-key="k-0-0"><span data-text="true">The last word: 3 &lt; 4.</span></span></div></div></div></div></div></div></div></div><div role="group" aria-label="40 replies, 7 reposts, 300 likes" class="css-175oi2r"><button data-testid="reply" type="button"><span>40</span></button><button data-testid="like" type="button"><span>300</span></button></div><div class="css-175oi2r"><span>A trailing note outside the body.</span></div></div> diff --git a/common/social/xArticle.test.ts b/common/social/xArticle.test.ts @@ -0,0 +1,204 @@ +// X Articles, the pure half: the article link in a post's text or on its card, +// the article root's HTML read into blocks (against a written fixture, never +// x.com), and the markdown written from them. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticle.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + decodeEntities, + extractXArticle, + findXArticleLink, + parseHtml, + textOf, + xArticleLinkFromArchive, + xArticleMarkdown, +} from "./xArticle"; +import { xArticleLinkFromCard } from "./xArticleCapture"; + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8"); + +// --- the link ---------------------------------------------------------------- + +test("link: an article post's archived text, its expanded links, or nothing", () => { + assert.deepEqual(xArticleLinkFromArchive({ text: "https://x.com/i/article/1987654321" }), { + articleId: "1987654321", + url: "https://x.com/i/article/1987654321", + }); + assert.deepEqual( + xArticleLinkFromArchive({ text: "Read it: https://t.co/abc", links: ["https://twitter.com/i/article/42"] }), + { articleId: "42", url: "https://x.com/i/article/42" }, + ); + assert.deepEqual(xArticleLinkFromArchive({ text: "x.com/demo_author/article/77 is up" }), { + articleId: "77", + url: "https://x.com/demo_author/article/77", + }); + for (const text of [ + "a plain post", + "https://x.com/demo_author/status/123", + "https://example.com/i/article/5", + "https://x.com/i/articles", + ]) { + assert.equal(xArticleLinkFromArchive({ text }), null, text); + } + assert.equal(xArticleLinkFromArchive(undefined), null); +}); + +test("link: a card's hrefs, relative as the page has them", () => { + assert.deepEqual(xArticleLinkFromCard(["/demo_author", "/demo_author/article/1234567890"]), { + articleId: "1234567890", + url: "https://x.com/demo_author/article/1234567890", + }); + assert.deepEqual(xArticleLinkFromCard(["/i/article/99/media/1"]), { + articleId: "99", + url: "https://x.com/i/article/99", + }); + assert.equal(xArticleLinkFromCard(["/demo_author/status/1/photo/1"]), null); + assert.equal(xArticleLinkFromCard([]), null); + assert.equal(xArticleLinkFromCard(undefined), null); + assert.equal(findXArticleLink([null, undefined, ""]), null); +}); + +// --- the HTML reader --------------------------------------------------------- + +test("html: elements, attributes, void and self-closing tags, entities, comments, raw text", () => { + const root = parseHtml( + '<!-- c --><div class="a" data-x=\'q\' hidden><img src="s?a=1&amp;b=2"><br/>a &lt;b&gt; &#39;c&#x27; &nbsp;d' + + "<script>if (a < b) {}</script><span>e</span></p></div>tail", + ); + const div = root.children[0] as { tag: string; attrs: Record<string, string>; children: unknown[] }; + assert.equal(div.tag, "div"); + assert.deepEqual(div.attrs, { class: "a", "data-x": "q", hidden: "" }); + assert.equal((div.children[0] as { attrs: { src: string } }).attrs.src, "s?a=1&b=2"); + // A stray end tag is ignored; the text after the div is the root's. + assert.equal(root.children[1], "tail"); + assert.equal(textOf(div as never), "a <b> 'c' de"); + assert.equal(decodeEntities("&amp;&unknown;&#128578;"), "&&unknown;๐Ÿ™‚"); +}); + +test("html: text reads as innerText would โ€” line breaks at <br> and blocks, script and svg silent", () => { + const root = parseHtml("<div><p>one <b>two</b></p><p>three<br>four</p><svg><text>no</text></svg><script>x</script></div>"); + assert.equal(textOf(root), "one two\nthree\nfour"); +}); + +// --- the extractor ------------------------------------------------------------- + +test("extract: title, byline and the body's blocks in reading order, from the written fixture", () => { + const a = extractXArticle(FIXTURE); + assert.equal(a.extraction, "structured"); + assert.equal(a.title, "A Worked Example & Its Notes"); + assert.equal(a.author, "Demo Author"); + assert.equal(a.handle, "@demo_author"); + assert.equal(a.publishedAt, "2026-01-02T03:04:05.000Z"); + assert.deepEqual(a.blocks, [ + // The cover, above the body. + { type: "image", src: "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" }, + { type: "paragraph", text: "The first paragraph, with a linked source and an emoji ๐Ÿ™‚." }, + { type: "link", text: "linked source", href: "https://example.com/source" }, + { type: "heading", text: "A Section Heading" }, + { type: "paragraph", text: "A second paragraph that runs\nonto a second line." }, + { type: "quote", text: "A quoted passage, set apart." }, + { type: "list-item", text: "The first point" }, + { type: "list-item", text: "# The second point" }, + { + type: "image", + src: "https://pbs.twimg.com/media/BodyImg1?format=png&name=small", + text: "A chart of the numbers", + }, + { + type: "embedded-post", + text: "An embedded post's own words.", + href: "https://x.com/other_user/status/4444444444", + }, + // A paragraph that is only a link is the link. + { type: "link", text: "Further reading", href: "https://example.com/further-reading" }, + { type: "paragraph", text: "The last word: 3 < 4." }, + ]); +}); + +test("extract: no body marker โ€” the root's text split at block elements, and it says so", () => { + const a = extractXArticle( + '<article><h1>Plain Title</h1><div><p>One</p><p>Two <a href="/demo_author">a handle</a></p>' + + '<figure><img src="https://pbs.twimg.com/media/Fig?format=webp"></figure><ul><li>Item</li></ul>' + + '<blockquote>Said</blockquote></div><div role="group">40 likes</div></article>', + ); + assert.equal(a.extraction, "fallback"); + assert.equal(a.title, "Plain Title"); + assert.deepEqual(a.blocks, [ + { type: "paragraph", text: "One" }, + { type: "paragraph", text: "Two a handle" }, + { type: "link", text: "a handle", href: "https://x.com/demo_author" }, + { type: "image", src: "https://pbs.twimg.com/media/Fig?format=webp" }, + { type: "list-item", text: "Item" }, + { type: "quote", text: "Said" }, + ]); +}); + +test("extract: an empty root is an untitled article with no blocks", () => { + assert.deepEqual(extractXArticle(""), { extraction: "fallback", blocks: [] }); +}); + +// --- markdown ---------------------------------------------------------------- + +test("markdown: title, byline, the URL, then each block โ€” saved images linked locally, markdown in text escaped", () => { + const a = extractXArticle(FIXTURE); + const blocks = a.blocks.map((b) => + b.src?.includes("BodyImg1") ? { ...b, file: "article-img-2.png" } : b, + ); + const md = xArticleMarkdown({ ...a, blocks, url: "https://x.com/i/article/1234567890" }); + assert.equal( + md, + [ + "# A Worked Example & Its Notes", + "", + "By Demo Author @demo_author ยท 2026-01-02T03:04:05.000Z", + "", + "<https://x.com/i/article/1234567890>", + "", + "![](<https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small>)", + "", + "The first paragraph, with a linked source and an emoji ๐Ÿ™‚.", + "", + "[linked source](<https://example.com/source>)", + "", + "## A Section Heading", + "", + "A second paragraph that runs", + "onto a second line.", + "", + "> A quoted passage, set apart.", + "", + "- The first point", + "- \\# The second point", + "", + "![A chart of the numbers](<article-img-2.png>)", + "", + "> Embedded post: <https://x.com/other_user/status/4444444444>", + ">", + "> An embedded post's own words.", + "", + "[Further reading](<https://example.com/further-reading>)", + "", + "The last word: 3 < 4.", + "", + ].join("\n"), + ); +}); + +test("markdown: an untitled article, no byline, a paragraph that would read as syntax", () => { + const md = xArticleMarkdown({ + extraction: "fallback", + url: "https://x.com/i/article/1", + blocks: [{ type: "paragraph", text: "1. not a list\n> not a quote" }, { type: "link", href: "https://example.com/a b" }], + }); + assert.equal( + md, + "# Untitled article\n\n<https://x.com/i/article/1>\n\n1\\. not a list\n\\> not a quote\n\n" + + "[https://example.com/a b](<https://example.com/a%20b>)\n", + ); +}); diff --git a/common/social/xArticle.ts b/common/social/xArticle.ts @@ -0,0 +1,547 @@ +// X Articles (long-form posts), the pure half: finding a post's article link, +// reading the article's HTML into blocks, and writing those blocks as +// markdown. No browser, no network, no fs โ€” xArticleCapture.ts loads the page +// and hands this module the article root's outerHTML. +// +// An article post archives as text that is only its link +// (`https://x.com/i/article/<id>`): gallery-dl cannot read an article's body, +// and the post's card shows a title and a preview. The article itself is a +// page of its own, which the capture opens. +// +// WHY HTML AND NOT A WALK IN THE PAGE. The browser hands back the root's +// outerHTML once, and everything after that is parsed and read here, so the +// whole extractor runs in tests against a saved HTML fixture, and the HTML is +// kept beside the capture (`article.html`) so a better reading later costs no +// second visit to X. +// +// NOT VERIFIED AGAINST LIVE X. The markers below (data-testid names, the +// Draft.js block classes) are X's as of this writing; every one is optional, +// and an article whose body marker is missing is read by the fallback โ€” the +// root's text split at block elements โ€” and says so (`extraction`). + +// --- the link ------------------------------------------------------------------ + +export type XArticleLink = { + // The id X's article URL carries. + articleId: string; + url: string; +}; + +// `x.com/i/article/<id>`, or `x.com/<handle>/article/<id>` (the form a card +// links to), absolute or as a path. +const ARTICLE_URL_RE = + /(?:https?:\/\/)?(?:www\.|mobile\.)?(?:x|twitter)\.com\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/i; +const ARTICLE_PATH_RE = /^\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/; + +function linkFrom(owner: string, articleId: string): XArticleLink { + return { articleId, url: `https://x.com/${owner}/article/${articleId}` }; +} + +// The first article link in these strings (a post's text, its expanded links, +// a card's hrefs), or null. +export function findXArticleLink( + candidates: ReadonlyArray<string | null | undefined>, +): XArticleLink | null { + for (const c of candidates) { + if (!c) continue; + const m = ARTICLE_URL_RE.exec(c) ?? ARTICLE_PATH_RE.exec(c.trim()); + if (m) return linkFrom(m[1].toLowerCase() === "i" ? "i" : m[1], m[2]); + } + return null; +} + +// What the posts archive holds for a post, as far as its links go. +export type ArchivedPostText = { text: string; links?: ReadonlyArray<string> }; + +export function xArticleLinkFromArchive( + archived: ArchivedPostText | undefined, +): XArticleLink | null { + if (!archived) return null; + return findXArticleLink([archived.text, ...(archived.links ?? [])]); +} + +// --- a small HTML reader ------------------------------------------------------- +// +// For HTML a browser serialised (outerHTML): attributes are double-quoted, void +// elements are unclosed, text escapes only & < > and nbsp. It tolerates more +// (single quotes, bare values, stray end tags), but it is not a general parser. + +export type HtmlElement = { + tag: string; + attrs: Record<string, string>; + children: HtmlNode[]; +}; +export type HtmlNode = HtmlElement | string; + +const VOID_TAGS = new Set([ + "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", + "param", "source", "track", "wbr", +]); +const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]); + +const NAMED_ENTITIES: Record<string, string> = { + amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0", + hellip: "โ€ฆ", mdash: "โ€”", ndash: "โ€“", lsquo: "โ€˜", rsquo: "โ€™", ldquo: "โ€œ", + rdquo: "โ€", copy: "ยฉ", reg: "ยฎ", trade: "โ„ข", +}; + +export function decodeEntities(s: string): string { + return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => { + if (name[0] === "#") { + const code = + name[1] === "x" || name[1] === "X" + ? parseInt(name.slice(2), 16) + : parseInt(name.slice(1), 10); + return Number.isFinite(code) && code > 0 && code <= 0x10ffff + ? String.fromCodePoint(code) + : whole; + } + return NAMED_ENTITIES[name.toLowerCase()] ?? whole; + }); +} + +const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y; +const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y; + +// The document's top-level nodes, under a synthetic root element. +export function parseHtml(html: string): HtmlElement { + const root: HtmlElement = { tag: "#root", attrs: {}, children: [] }; + const stack: HtmlElement[] = [root]; + const top = () => stack[stack.length - 1]; + let i = 0; + while (i < html.length) { + const lt = html.indexOf("<", i); + if (lt < 0) { + top().children.push(decodeEntities(html.slice(i))); + break; + } + if (lt > i) top().children.push(decodeEntities(html.slice(i, lt))); + i = lt; + if (html.startsWith("<!--", i)) { + const end = html.indexOf("-->", i + 4); + i = end < 0 ? html.length : end + 3; + continue; + } + if (html[i + 1] === "!" || html[i + 1] === "?") { + const end = html.indexOf(">", i); + i = end < 0 ? html.length : end + 1; + continue; + } + if (html[i + 1] === "/") { + const end = html.indexOf(">", i); + const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase(); + i = end < 0 ? html.length : end + 1; + // Close up to the matching open element; a stray end tag is ignored. + for (let k = stack.length - 1; k > 0; k--) { + if (stack[k].tag === name) { + stack.length = k; + break; + } + } + continue; + } + TAG_NAME_RE.lastIndex = i + 1; + const nameMatch = TAG_NAME_RE.exec(html); + if (!nameMatch) { + // A "<" that opens no tag is text. + top().children.push("<"); + i++; + continue; + } + const tag = nameMatch[0].toLowerCase(); + let j = TAG_NAME_RE.lastIndex; + const attrs: Record<string, string> = {}; + for (;;) { + ATTR_RE.lastIndex = j; + const a = ATTR_RE.exec(html); + if (!a || a[0].length === 0) break; + attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? ""); + j = ATTR_RE.lastIndex; + } + const close = html.indexOf(">", j); + const selfClosing = close > 0 && html[close - 1] === "/"; + i = close < 0 ? html.length : close + 1; + const el: HtmlElement = { tag, attrs, children: [] }; + top().children.push(el); + if (RAW_TEXT_TAGS.has(tag)) { + const end = html.toLowerCase().indexOf(`</${tag}`, i); + const stop = end < 0 ? html.length : end; + if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop))); + const gt = end < 0 ? -1 : html.indexOf(">", end); + i = gt < 0 ? html.length : gt + 1; + continue; + } + if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el); + } + return root; +} + +// --- reading the article ------------------------------------------------------ + +export type XArticleBlockType = + | "heading" + | "paragraph" + | "quote" + | "list-item" + | "image" + | "embedded-post" + | "link"; + +export type XArticleBlock = { + type: XArticleBlockType; + text?: string; + href?: string; + src?: string; + // An image saved beside the capture: its file name there. + file?: string; +}; + +export type XArticleContent = { + title?: string; + author?: string; + handle?: string; + publishedAt?: string; + // "structured": the body marker was found and read block by block; + // "fallback": it was not, and the root's text was split at block elements. + extraction: "structured" | "fallback"; + blocks: XArticleBlock[]; +}; + +const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string"; +const testid = (el: HtmlElement) => el.attrs["data-testid"] ?? ""; +const classes = (el: HtmlElement) => el.attrs["class"] ?? ""; + +function findFirst( + el: HtmlElement, + pred: (e: HtmlElement) => boolean, + skip?: (e: HtmlElement) => boolean, +): HtmlElement | null { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) return c; + if (skip?.(c)) continue; + const hit = findFirst(c, pred, skip); + if (hit) return hit; + } + return null; +} + +function findAll(el: HtmlElement, pred: (e: HtmlElement) => boolean, out: HtmlElement[] = []): HtmlElement[] { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) out.push(c); + findAll(c, pred, out); + } + return out; +} + +const SILENT_TAGS = new Set([ + "script", "style", "svg", "noscript", "template", "button", "input", "select", + "video", "audio", "iframe", "canvas", +]); + +// An element's text as innerText would give it, near enough: `<br>` and block +// boundaries are line breaks (never a blank line), runs of other whitespace are +// one space, an emoji drawn as an image is its alt. +export function textOf(node: HtmlNode): string { + const parts: string[] = []; + const walk = (n: HtmlNode) => { + if (!isEl(n)) { + parts.push(n.replace(/[\s\u00a0]+/g, " ")); + return; + } + if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true") return; + if (n.tag === "br") { + parts.push("\n"); + return; + } + if (n.tag === "img") { + parts.push(emojiText(n)); + return; + } + const block = BLOCK_TAGS.has(n.tag); + if (block) parts.push("\n"); + for (const c of n.children) walk(c); + if (block) parts.push("\n"); + }; + walk(node); + return parts + .join("") + .split("\n") + .map((l) => l.replace(/ +/g, " ").trim()) + .filter(Boolean) + .join("\n"); +} + +const BLOCK_TAGS = new Set([ + "address", "article", "aside", "blockquote", "dd", "details", "div", "dl", + "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", + "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section", + "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", +]); + +// The article's body: X's long-form rich-text view, else the Draft.js content +// it is built on. +const BODY_TESTIDS = new Set(["longformRichTextComponent", "twitterArticleRichTextView"]); +const isBody = (el: HtmlElement) => + BODY_TESTIDS.has(testid(el)) || el.attrs["data-contents"] === "true"; + +const isTitle = (el: HtmlElement) => testid(el) === "twitter-article-title"; +const isByline = (el: HtmlElement) => testid(el) === "User-Name"; +// The chrome around an article that is not its text. +const isChrome = (el: HtmlElement) => + /^(UserAvatar|UserAvatar-Container|reply|retweet|like|bookmark|caret|app-text-transition-container)/.test( + testid(el), + ) || el.attrs["role"] === "group"; + +const EMBED_TESTIDS = new Set(["tweet", "simpleTweet", "quoteTweet"]); +const isEmbeddedPost = (el: HtmlElement) => + EMBED_TESTIDS.has(testid(el)) || + (el.tag === "blockquote" && /\btwitter-tweet\b/.test(classes(el))); + +const STATUS_RE = + /^(?:https?:\/\/(?:www\.|mobile\.)?(?:x|twitter)\.com)?\/([A-Za-z0-9_]{1,15})\/status\/(\d{1,25})/i; + +// An embedded post's status URL: the link around its timestamp, else its first +// status link. +function statusUrlIn(el: HtmlElement): string | undefined { + const links = findAll(el, (e) => e.tag === "a" && STATUS_RE.test(e.attrs.href ?? "")); + const best = links.find((a) => findFirst(a, (e) => e.tag === "time")) ?? links[0]; + const m = best ? STATUS_RE.exec(best.attrs.href) : null; + return m ? `https://x.com/${m[1]}/status/${m[2]}` : undefined; +} + +// An image that is the article's, not an emoji, an avatar or an icon. +export function isContentImageSrc(src: string | undefined): src is string { + if (!src || src.startsWith("data:")) return false; + if (/\/emoji\/|\/hashflags\/|profile_images|profile_banners|\.svg(\?|$)/i.test(src)) return false; + return /^https?:\/\//i.test(src); +} + +// X draws an emoji as an image whose alt is the emoji: its text. +function emojiText(img: HtmlElement): string { + return /\/emoji\//.test(img.attrs.src ?? "") ? (img.attrs.alt ?? "") : ""; +} + +// An href as an absolute URL, or undefined for one that goes nowhere. +function absoluteHref(href: string | undefined): string | undefined { + if (!href) return undefined; + const h = href.trim(); + if (h === "" || h.startsWith("#") || /^(javascript|mailto|data):/i.test(h)) { + return /^mailto:/i.test(h) ? h : undefined; + } + if (/^https?:\/\//i.test(h)) return h; + if (h.startsWith("//")) return `https:${h}`; + if (h.startsWith("/")) return `https://x.com${h}`; + return undefined; +} + +function blockTypeOf(el: HtmlElement): "heading" | "quote" | "list-item" | null { + const c = classes(el); + if (/^h[1-6]$/.test(el.tag) || /\blongform-header-/.test(c)) return "heading"; + if (el.tag === "blockquote" || /\blongform-blockquote\b/.test(c)) return "quote"; + if (el.tag === "li" || /\blongform-(un)?ordered-list-item\b/.test(c)) return "list-item"; + return null; +} + +// The blocks of `body` in reading order. Text outside any typed block gathers +// into a paragraph that ends at the next block boundary; a link inside text is +// kept as a link block after it (a paragraph that is only a link is just the +// link). +function readBlocks(body: HtmlElement, skip: (el: HtmlElement) => boolean): XArticleBlock[] { + const blocks: XArticleBlock[] = []; + let inline: string[] = []; + let links: XArticleBlock[] = []; + + const pushLinks = (from: HtmlElement) => { + for (const a of findAll(from, (e) => e.tag === "a")) { + if (findFirst(a, (e) => e.tag === "img")) continue; + const href = absoluteHref(a.attrs.href); + const text = textOf(a); + if (href) blocks.push({ type: "link", ...(text ? { text } : {}), href }); + } + }; + const flush = () => { + const text = inline + .join("") + .split("\n") + .map((l) => l.replace(/[ \t\u00a0]+/g, " ").trim()) + .filter(Boolean) + .join("\n"); + inline = []; + const onlyLink = links.length === 1 && links[0].text === text; + if (text && !onlyLink) blocks.push({ type: "paragraph", text }); + blocks.push(...links); + links = []; + }; + + const walk = (n: HtmlNode) => { + if (!isEl(n)) { + inline.push(n.replace(/[\s\u00a0]+/g, " ")); + return; + } + if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true" || skip(n)) return; + if (n.tag === "br") { + inline.push("\n"); + return; + } + if (isEmbeddedPost(n)) { + flush(); + const textEl = findFirst(n, (e) => testid(e) === "tweetText"); + const text = textEl ? textOf(textEl) : ""; + const href = statusUrlIn(n); + blocks.push({ type: "embedded-post", ...(text ? { text } : {}), ...(href ? { href } : {}) }); + return; + } + if (n.tag === "img") { + if (!isContentImageSrc(n.attrs.src)) { + inline.push(emojiText(n)); + return; + } + flush(); + const alt = (n.attrs.alt ?? "").trim(); + blocks.push({ type: "image", src: n.attrs.src, ...(alt && alt.toLowerCase() !== "image" ? { text: alt } : {}) }); + return; + } + const typed = blockTypeOf(n); + if (typed) { + flush(); + const text = textOf(n); + if (text) blocks.push({ type: typed, text }); + pushLinks(n); + // An image inside a typed block (rare) is still the article's. + for (const img of findAll(n, (e) => e.tag === "img" && isContentImageSrc(e.attrs.src))) { + blocks.push({ type: "image", src: img.attrs.src }); + } + return; + } + if (n.tag === "a" && !findFirst(n, (e) => e.tag === "img")) { + const href = absoluteHref(n.attrs.href); + const text = textOf(n); + inline.push(text); + if (href) links.push({ type: "link", ...(text ? { text } : {}), href }); + return; + } + const block = BLOCK_TAGS.has(n.tag); + if (block) flush(); + for (const c of n.children) walk(c); + if (block) flush(); + }; + walk(body); + flush(); + return blocks; +} + +// The article root's HTML, read: title, byline, and the body's blocks. +export function extractXArticle(html: string): XArticleContent { + const root = parseHtml(html); + const titleEl = findFirst(root, isTitle); + const fallbackTitle = titleEl ? null : findFirst(root, (e) => e.tag === "h1"); + const title = textOf(titleEl ?? fallbackTitle ?? { tag: "#none", attrs: {}, children: [] }) || undefined; + + const bylineEl = findFirst(root, isByline, isEmbeddedPost); + let author: string | undefined; + let handle: string | undefined; + if (bylineEl) { + const lines = textOf(bylineEl).split(/\n|ยท/).map((s) => s.trim()).filter(Boolean); + handle = lines.find((l) => /^@[A-Za-z0-9_]{1,15}$/.test(l)); + author = lines.find((l) => !l.startsWith("@") && l !== handle); + } + const time = findFirst(root, (e) => e.tag === "time" && !!e.attrs.datetime, isEmbeddedPost); + const publishedAt = time?.attrs.datetime; + + const body = findFirst(root, isBody); + const skipHeader = (el: HtmlElement) => + el === titleEl || el === fallbackTitle || isByline(el) || isChrome(el); + let blocks: XArticleBlock[]; + if (body) { + // A cover image sits above the body: the images before it are the + // article's; nothing after the body (the engagement bar, replies) is. + const cover: XArticleBlock[] = []; + const before = (el: HtmlElement): boolean => { + for (const c of el.children) { + if (!isEl(c)) continue; + if (c === body) return true; + if (isEmbeddedPost(c) || skipHeader(c)) continue; + if (c.tag === "img" && isContentImageSrc(c.attrs.src)) { + cover.push({ type: "image", src: c.attrs.src }); + continue; + } + if (before(c)) return true; + } + return false; + }; + before(root); + blocks = [...cover, ...readBlocks(body, isChrome)]; + } else { + blocks = readBlocks(root, skipHeader); + } + return { + ...(title ? { title } : {}), + ...(author ? { author } : {}), + ...(handle ? { handle } : {}), + ...(publishedAt ? { publishedAt } : {}), + extraction: body ? "structured" : "fallback", + blocks, + }; +} + +// --- markdown ---------------------------------------------------------------- + +// A line that would read as markdown syntax, escaped. +function mdText(s: string): string { + return s + .split("\n") + .map((l) => + l.replace(/^(\s*)([#>+\-*])(\s)/, "$1\\$2$3").replace(/^(\s*)(\d+)([.)])(\s)/, "$1$2\\$3$4"), + ) + .join("\n"); +} +const mdLabel = (s: string) => s.replace(/([\[\]\\])/g, "\\$1").replace(/\n+/g, " "); +const mdDest = (u: string) => `<${u.replace(/[<>\s]/g, encodeURIComponent)}>`; + +// An image block links to its saved file when it has one, else to its src. +export type XArticleMarkdownInput = XArticleContent & { url: string }; + +export function xArticleMarkdown(a: XArticleMarkdownInput): string { + const oneLine = (t: string | undefined) => (t ?? "").replace(/\s*\n\s*/g, " ").trim(); + let md = `# ${oneLine(a.title) || "Untitled article"}`; + const add = (chunk: string, tight = false) => { + md += (tight ? "\n" : "\n\n") + chunk; + }; + const by = [a.author, a.handle].filter(Boolean).join(" "); + const meta = [by ? `By ${by}` : "", a.publishedAt ?? ""].filter(Boolean).join(" ยท "); + if (meta) add(mdText(meta)); + add(mdDest(a.url)); + let prev: XArticleBlockType | undefined; + for (const b of a.blocks) { + switch (b.type) { + case "heading": + add(`## ${oneLine(b.text)}`); + break; + case "quote": + add((b.text ?? "").split("\n").map((l) => `> ${l}`.trimEnd()).join("\n")); + break; + case "list-item": + // List items run together. + add(`- ${mdText(b.text ?? "").replace(/\n/g, "\n ")}`, prev === "list-item"); + break; + case "image": + add(`![${mdLabel(b.text ?? "")}](${mdDest(b.file ?? b.src ?? "")})`); + break; + case "embedded-post": + add( + `> Embedded post: ${b.href ? mdDest(b.href) : "(no link)"}` + + (b.text ? "\n>\n" + b.text.split("\n").map((l) => `> ${l}`.trimEnd()).join("\n") : ""), + ); + break; + case "link": + add(`[${mdLabel(b.text || b.href || "")}](${mdDest(b.href ?? "")})`); + break; + default: + add(mdText(b.text ?? "")); + } + prev = b.type; + } + return md + "\n"; +}