// X Articles (long-form posts), the pure half: finding a post's article link, // reading the article's HTML into blocks, and writing those blocks as // markdown. No browser, no network, no fs — xArticleCapture.ts loads the page // and hands this module the article root's outerHTML. // // An article post archives as text that is only its link // (`https://x.com/i/article/`): gallery-dl cannot read an article's body, // and the post's card shows a title and a preview. The article itself is a // page of its own, which the capture opens. // // WHY HTML AND NOT A WALK IN THE PAGE. The browser hands back the root's // outerHTML once, and everything after that is parsed and read here, so the // whole extractor runs in tests against a saved HTML fixture, and the HTML is // kept beside the capture (`article.html`) so a better reading later costs no // second visit to X. // // NOT VERIFIED AGAINST LIVE X. The markers below (data-testid names, the // Draft.js block classes) are X's as of this writing; every one is optional, // and an article whose body marker is missing is read by the fallback — the // root's text split at block elements — and says so (`extraction`). import { decodeEntities, parseHtml, type HtmlElement, type HtmlNode, } from "./htmlReader"; // --- the link ------------------------------------------------------------------ export type XArticleLink = { // The id X's article URL carries. articleId: string; url: string; }; // `x.com/i/article/`, or `x.com//article/` (the form a card // links to), absolute or as a path. const ARTICLE_URL_RE = /(?:https?:\/\/)?(?:www\.|mobile\.)?(?:x|twitter)\.com\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/i; const ARTICLE_PATH_RE = /^\/(i|[A-Za-z0-9_]{1,15})\/article\/(\d{1,25})(?!\d)/; function linkFrom(owner: string, articleId: string): XArticleLink { return { articleId, url: `https://x.com/${owner}/article/${articleId}` }; } // The first article link in these strings (a post's text, its expanded links, // a card's hrefs), or null. export function findXArticleLink( candidates: ReadonlyArray, ): XArticleLink | null { for (const c of candidates) { if (!c) continue; const m = ARTICLE_URL_RE.exec(c) ?? ARTICLE_PATH_RE.exec(c.trim()); if (m) return linkFrom(m[1].toLowerCase() === "i" ? "i" : m[1], m[2]); } return null; } // What the posts archive holds for a post, as far as its links go. export type ArchivedPostText = { text: string; links?: ReadonlyArray }; export function xArticleLinkFromArchive( archived: ArchivedPostText | undefined, ): XArticleLink | null { if (!archived) return null; return findXArticleLink([archived.text, ...(archived.links ?? [])]); } // --- the HTML reader ----------------------------------------------------------- // // Shared with the forum-thread parser (xenforoParse.ts): it lives in // htmlReader.ts and is re-exported here for the callers that import it from // this module. export { decodeEntities, parseHtml, type HtmlElement, type HtmlNode }; // --- reading the article ------------------------------------------------------ export type XArticleBlockType = | "heading" | "paragraph" | "quote" | "list-item" | "image" | "embedded-post" | "link"; export type XArticleBlock = { type: XArticleBlockType; text?: string; href?: string; src?: string; // An image saved beside the capture: its file name there. file?: string; }; export type XArticleContent = { title?: string; author?: string; handle?: string; publishedAt?: string; // "structured": the body marker was found and read block by block; // "fallback": it was not, and the root's text was split at block elements. extraction: "structured" | "fallback"; blocks: XArticleBlock[]; }; const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string"; const testid = (el: HtmlElement) => el.attrs["data-testid"] ?? ""; const classes = (el: HtmlElement) => el.attrs["class"] ?? ""; function findFirst( el: HtmlElement, pred: (e: HtmlElement) => boolean, skip?: (e: HtmlElement) => boolean, ): HtmlElement | null { for (const c of el.children) { if (!isEl(c)) continue; if (pred(c)) return c; if (skip?.(c)) continue; const hit = findFirst(c, pred, skip); if (hit) return hit; } return null; } function findAll(el: HtmlElement, pred: (e: HtmlElement) => boolean, out: HtmlElement[] = []): HtmlElement[] { for (const c of el.children) { if (!isEl(c)) continue; if (pred(c)) out.push(c); findAll(c, pred, out); } return out; } const SILENT_TAGS = new Set([ "script", "style", "svg", "noscript", "template", "button", "input", "select", "video", "audio", "iframe", "canvas", ]); // An element's text as innerText would give it, near enough: `
` and block // boundaries are line breaks (never a blank line), runs of other whitespace are // one space, an emoji drawn as an image is its alt. export function textOf(node: HtmlNode): string { const parts: string[] = []; const walk = (n: HtmlNode) => { if (!isEl(n)) { parts.push(n.replace(/[\s\u00a0]+/g, " ")); return; } if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true") return; if (n.tag === "br") { parts.push("\n"); return; } if (n.tag === "img") { parts.push(emojiText(n)); return; } const block = BLOCK_TAGS.has(n.tag); if (block) parts.push("\n"); for (const c of n.children) walk(c); if (block) parts.push("\n"); }; walk(node); return parts .join("") .split("\n") .map((l) => l.replace(/ +/g, " ").trim()) .filter(Boolean) .join("\n"); } const BLOCK_TAGS = new Set([ "address", "article", "aside", "blockquote", "dd", "details", "div", "dl", "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section", "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", ]); // The article's body: X's long-form rich-text view, else the Draft.js content // it is built on. const BODY_TESTIDS = new Set(["longformRichTextComponent", "twitterArticleRichTextView"]); const isBody = (el: HtmlElement) => BODY_TESTIDS.has(testid(el)) || el.attrs["data-contents"] === "true"; const isTitle = (el: HtmlElement) => testid(el) === "twitter-article-title"; const isByline = (el: HtmlElement) => testid(el) === "User-Name"; // The chrome around an article that is not its text. const isChrome = (el: HtmlElement) => /^(UserAvatar|UserAvatar-Container|reply|retweet|like|bookmark|caret|app-text-transition-container)/.test( testid(el), ) || el.attrs["role"] === "group"; const EMBED_TESTIDS = new Set(["tweet", "simpleTweet", "quoteTweet"]); const isEmbeddedPost = (el: HtmlElement) => EMBED_TESTIDS.has(testid(el)) || (el.tag === "blockquote" && /\btwitter-tweet\b/.test(classes(el))); const STATUS_RE = /^(?:https?:\/\/(?:www\.|mobile\.)?(?:x|twitter)\.com)?\/([A-Za-z0-9_]{1,15})\/status\/(\d{1,25})/i; // An embedded post's status URL: the link around its timestamp, else its first // status link. function statusUrlIn(el: HtmlElement): string | undefined { const links = findAll(el, (e) => e.tag === "a" && STATUS_RE.test(e.attrs.href ?? "")); const best = links.find((a) => findFirst(a, (e) => e.tag === "time")) ?? links[0]; const m = best ? STATUS_RE.exec(best.attrs.href) : null; return m ? `https://x.com/${m[1]}/status/${m[2]}` : undefined; } // An image that is the article's, not an emoji, an avatar or an icon. export function isContentImageSrc(src: string | undefined): src is string { if (!src || src.startsWith("data:")) return false; if (/\/emoji\/|\/hashflags\/|profile_images|profile_banners|\.svg(\?|$)/i.test(src)) return false; return /^https?:\/\//i.test(src); } // X draws an emoji as an image whose alt is the emoji: its text. function emojiText(img: HtmlElement): string { return /\/emoji\//.test(img.attrs.src ?? "") ? (img.attrs.alt ?? "") : ""; } // An href as an absolute URL, or undefined for one that goes nowhere. function absoluteHref(href: string | undefined): string | undefined { if (!href) return undefined; const h = href.trim(); if (h === "" || h.startsWith("#") || /^(javascript|mailto|data):/i.test(h)) { return /^mailto:/i.test(h) ? h : undefined; } if (/^https?:\/\//i.test(h)) return h; if (h.startsWith("//")) return `https:${h}`; if (h.startsWith("/")) return `https://x.com${h}`; return undefined; } function blockTypeOf(el: HtmlElement): "heading" | "quote" | "list-item" | null { const c = classes(el); if (/^h[1-6]$/.test(el.tag) || /\blongform-header-/.test(c)) return "heading"; if (el.tag === "blockquote" || /\blongform-blockquote\b/.test(c)) return "quote"; if (el.tag === "li" || /\blongform-(un)?ordered-list-item\b/.test(c)) return "list-item"; return null; } // The blocks of `body` in reading order. Text outside any typed block gathers // into a paragraph that ends at the next block boundary; a link inside text is // kept as a link block after it (a paragraph that is only a link is just the // link). function readBlocks(body: HtmlElement, skip: (el: HtmlElement) => boolean): XArticleBlock[] { const blocks: XArticleBlock[] = []; let inline: string[] = []; let links: XArticleBlock[] = []; const pushLinks = (from: HtmlElement) => { for (const a of findAll(from, (e) => e.tag === "a")) { if (findFirst(a, (e) => e.tag === "img")) continue; const href = absoluteHref(a.attrs.href); const text = textOf(a); if (href) blocks.push({ type: "link", ...(text ? { text } : {}), href }); } }; const flush = () => { const text = inline .join("") .split("\n") .map((l) => l.replace(/[ \t\u00a0]+/g, " ").trim()) .filter(Boolean) .join("\n"); inline = []; const onlyLink = links.length === 1 && links[0].text === text; if (text && !onlyLink) blocks.push({ type: "paragraph", text }); blocks.push(...links); links = []; }; const walk = (n: HtmlNode) => { if (!isEl(n)) { inline.push(n.replace(/[\s\u00a0]+/g, " ")); return; } if (SILENT_TAGS.has(n.tag) || n.attrs["aria-hidden"] === "true" || skip(n)) return; if (n.tag === "br") { inline.push("\n"); return; } if (isEmbeddedPost(n)) { flush(); const textEl = findFirst(n, (e) => testid(e) === "tweetText"); const text = textEl ? textOf(textEl) : ""; const href = statusUrlIn(n); blocks.push({ type: "embedded-post", ...(text ? { text } : {}), ...(href ? { href } : {}) }); return; } if (n.tag === "img") { if (!isContentImageSrc(n.attrs.src)) { inline.push(emojiText(n)); return; } flush(); const alt = (n.attrs.alt ?? "").trim(); blocks.push({ type: "image", src: n.attrs.src, ...(alt && alt.toLowerCase() !== "image" ? { text: alt } : {}) }); return; } const typed = blockTypeOf(n); if (typed) { flush(); const text = textOf(n); if (text) blocks.push({ type: typed, text }); pushLinks(n); // An image inside a typed block (rare) is still the article's. for (const img of findAll(n, (e) => e.tag === "img" && isContentImageSrc(e.attrs.src))) { blocks.push({ type: "image", src: img.attrs.src }); } return; } if (n.tag === "a" && !findFirst(n, (e) => e.tag === "img")) { const href = absoluteHref(n.attrs.href); const text = textOf(n); inline.push(text); if (href) links.push({ type: "link", ...(text ? { text } : {}), href }); return; } const block = BLOCK_TAGS.has(n.tag); if (block) flush(); for (const c of n.children) walk(c); if (block) flush(); }; walk(body); flush(); return blocks; } // The article root's HTML, read: title, byline, and the body's blocks. export function extractXArticle(html: string): XArticleContent { const root = parseHtml(html); const titleEl = findFirst(root, isTitle); const fallbackTitle = titleEl ? null : findFirst(root, (e) => e.tag === "h1"); const title = textOf(titleEl ?? fallbackTitle ?? { tag: "#none", attrs: {}, children: [] }) || undefined; const bylineEl = findFirst(root, isByline, isEmbeddedPost); let author: string | undefined; let handle: string | undefined; if (bylineEl) { const lines = textOf(bylineEl).split(/\n|·/).map((s) => s.trim()).filter(Boolean); handle = lines.find((l) => /^@[A-Za-z0-9_]{1,15}$/.test(l)); author = lines.find((l) => !l.startsWith("@") && l !== handle); } const time = findFirst(root, (e) => e.tag === "time" && !!e.attrs.datetime, isEmbeddedPost); const publishedAt = time?.attrs.datetime; const body = findFirst(root, isBody); const skipHeader = (el: HtmlElement) => el === titleEl || el === fallbackTitle || isByline(el) || isChrome(el); let blocks: XArticleBlock[]; if (body) { // A cover image sits above the body: the images before it are the // article's; nothing after the body (the engagement bar, replies) is. const cover: XArticleBlock[] = []; const before = (el: HtmlElement): boolean => { for (const c of el.children) { if (!isEl(c)) continue; if (c === body) return true; if (isEmbeddedPost(c) || skipHeader(c)) continue; if (c.tag === "img" && isContentImageSrc(c.attrs.src)) { cover.push({ type: "image", src: c.attrs.src }); continue; } if (before(c)) return true; } return false; }; before(root); blocks = [...cover, ...readBlocks(body, isChrome)]; } else { blocks = readBlocks(root, skipHeader); } return { ...(title ? { title } : {}), ...(author ? { author } : {}), ...(handle ? { handle } : {}), ...(publishedAt ? { publishedAt } : {}), extraction: body ? "structured" : "fallback", blocks, }; } // --- markdown ---------------------------------------------------------------- // A line that would read as markdown syntax, escaped. function mdText(s: string): string { return s .split("\n") .map((l) => l.replace(/^(\s*)([#>+\-*])(\s)/, "$1\\$2$3").replace(/^(\s*)(\d+)([.)])(\s)/, "$1$2\\$3$4"), ) .join("\n"); } const mdLabel = (s: string) => s.replace(/([\[\]\\])/g, "\\$1").replace(/\n+/g, " "); const mdDest = (u: string) => `<${u.replace(/[<>\s]/g, encodeURIComponent)}>`; // An image block links to its saved file when it has one, else to its src. export type XArticleMarkdownInput = XArticleContent & { url: string }; export function xArticleMarkdown(a: XArticleMarkdownInput): string { const oneLine = (t: string | undefined) => (t ?? "").replace(/\s*\n\s*/g, " ").trim(); let md = `# ${oneLine(a.title) || "Untitled article"}`; const add = (chunk: string, tight = false) => { md += (tight ? "\n" : "\n\n") + chunk; }; const by = [a.author, a.handle].filter(Boolean).join(" "); const meta = [by ? `By ${by}` : "", a.publishedAt ?? ""].filter(Boolean).join(" · "); if (meta) add(mdText(meta)); add(mdDest(a.url)); let prev: XArticleBlockType | undefined; for (const b of a.blocks) { switch (b.type) { case "heading": add(`## ${oneLine(b.text)}`); break; case "quote": add((b.text ?? "").split("\n").map((l) => `> ${l}`.trimEnd()).join("\n")); break; case "list-item": // List items run together. add(`- ${mdText(b.text ?? "").replace(/\n/g, "\n ")}`, prev === "list-item"); break; case "image": add(`![${mdLabel(b.text ?? "")}](${mdDest(b.file ?? b.src ?? "")})`); break; case "embedded-post": add( `> Embedded post: ${b.href ? mdDest(b.href) : "(no link)"}` + (b.text ? "\n>\n" + b.text.split("\n").map((l) => `> ${l}`.trimEnd()).join("\n") : ""), ); break; case "link": add(`[${mdLabel(b.text || b.href || "")}](${mdDest(b.href ?? "")})`); break; default: add(mdText(b.text ?? "")); } prev = b.type; } return md + "\n"; }