// X Articles, the pure half: the article link in a post's text or on its card, // the article root's HTML read into blocks (against a written fixture, never // x.com), and the markdown written from them. // // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticle.test.ts import { test } from "node:test"; import assert from "node:assert/strict"; import { readFile } from "node:fs/promises"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { decodeEntities, extractXArticle, findXArticleLink, parseHtml, textOf, xArticleLinkFromArchive, xArticleMarkdown, } from "./xArticle"; import { xArticleLinkFromCard } from "./xArticleCapture"; const HERE = path.dirname(fileURLToPath(import.meta.url)); const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8"); // --- the link ---------------------------------------------------------------- test("link: an article post's archived text, its expanded links, or nothing", () => { assert.deepEqual(xArticleLinkFromArchive({ text: "https://x.com/i/article/1987654321" }), { articleId: "1987654321", url: "https://x.com/i/article/1987654321", }); assert.deepEqual( xArticleLinkFromArchive({ text: "Read it: https://t.co/abc", links: ["https://twitter.com/i/article/42"] }), { articleId: "42", url: "https://x.com/i/article/42" }, ); assert.deepEqual(xArticleLinkFromArchive({ text: "x.com/demo_author/article/77 is up" }), { articleId: "77", url: "https://x.com/demo_author/article/77", }); for (const text of [ "a plain post", "https://x.com/demo_author/status/123", "https://example.com/i/article/5", "https://x.com/i/articles", ]) { assert.equal(xArticleLinkFromArchive({ text }), null, text); } assert.equal(xArticleLinkFromArchive(undefined), null); }); test("link: a card's hrefs, relative as the page has them", () => { assert.deepEqual(xArticleLinkFromCard(["/demo_author", "/demo_author/article/1234567890"]), { articleId: "1234567890", url: "https://x.com/demo_author/article/1234567890", }); assert.deepEqual(xArticleLinkFromCard(["/i/article/99/media/1"]), { articleId: "99", url: "https://x.com/i/article/99", }); assert.equal(xArticleLinkFromCard(["/demo_author/status/1/photo/1"]), null); assert.equal(xArticleLinkFromCard([]), null); assert.equal(xArticleLinkFromCard(undefined), null); assert.equal(findXArticleLink([null, undefined, ""]), null); }); // --- the HTML reader --------------------------------------------------------- test("html: elements, attributes, void and self-closing tags, entities, comments, raw text", () => { const root = parseHtml( 'tail", ); const div = root.children[0] as { tag: string; attrs: Record; children: unknown[] }; assert.equal(div.tag, "div"); assert.deepEqual(div.attrs, { class: "a", "data-x": "q", hidden: "" }); assert.equal((div.children[0] as { attrs: { src: string } }).attrs.src, "s?a=1&b=2"); // A stray end tag is ignored; the text after the div is the root's. assert.equal(root.children[1], "tail"); assert.equal(textOf(div as never), "a 'c' de"); assert.equal(decodeEntities("&&unknown;🙂"), "&&unknown;๐Ÿ™‚"); }); test("html: text reads as innerText would โ€” line breaks at
and blocks, script and svg silent", () => { const root = parseHtml("

one two

three
four

no
"); assert.equal(textOf(root), "one two\nthree\nfour"); }); // --- the extractor ------------------------------------------------------------- test("extract: title, byline and the body's blocks in reading order, from the written fixture", () => { const a = extractXArticle(FIXTURE); assert.equal(a.extraction, "structured"); assert.equal(a.title, "A Worked Example & Its Notes"); assert.equal(a.author, "Demo Author"); assert.equal(a.handle, "@demo_author"); assert.equal(a.publishedAt, "2026-01-02T03:04:05.000Z"); assert.deepEqual(a.blocks, [ // The cover, above the body. { type: "image", src: "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" }, { type: "paragraph", text: "The first paragraph, with a linked source and an emoji ๐Ÿ™‚." }, { type: "link", text: "linked source", href: "https://example.com/source" }, { type: "heading", text: "A Section Heading" }, { type: "paragraph", text: "A second paragraph that runs\nonto a second line." }, { type: "quote", text: "A quoted passage, set apart." }, { type: "list-item", text: "The first point" }, { type: "list-item", text: "# The second point" }, { type: "image", src: "https://pbs.twimg.com/media/BodyImg1?format=png&name=small", text: "A chart of the numbers", }, { type: "embedded-post", text: "An embedded post's own words.", href: "https://x.com/other_user/status/4444444444", }, // A paragraph that is only a link is the link. { type: "link", text: "Further reading", href: "https://example.com/further-reading" }, { type: "paragraph", text: "The last word: 3 < 4." }, ]); }); test("extract: no body marker โ€” the root's text split at block elements, and it says so", () => { const a = extractXArticle( '

Plain Title

One

Two a handle

' + '
  • Item
' + '
Said
40 likes
', ); assert.equal(a.extraction, "fallback"); assert.equal(a.title, "Plain Title"); assert.deepEqual(a.blocks, [ { type: "paragraph", text: "One" }, { type: "paragraph", text: "Two a handle" }, { type: "link", text: "a handle", href: "https://x.com/demo_author" }, { type: "image", src: "https://pbs.twimg.com/media/Fig?format=webp" }, { type: "list-item", text: "Item" }, { type: "quote", text: "Said" }, ]); }); test("extract: an empty root is an untitled article with no blocks", () => { assert.deepEqual(extractXArticle(""), { extraction: "fallback", blocks: [] }); }); // --- markdown ---------------------------------------------------------------- test("markdown: title, byline, the URL, then each block โ€” saved images linked locally, markdown in text escaped", () => { const a = extractXArticle(FIXTURE); const blocks = a.blocks.map((b) => b.src?.includes("BodyImg1") ? { ...b, file: "article-img-2.png" } : b, ); const md = xArticleMarkdown({ ...a, blocks, url: "https://x.com/i/article/1234567890" }); assert.equal( md, [ "# A Worked Example & Its Notes", "", "By Demo Author @demo_author ยท 2026-01-02T03:04:05.000Z", "", "", "", "![]()", "", "The first paragraph, with a linked source and an emoji ๐Ÿ™‚.", "", "[linked source]()", "", "## A Section Heading", "", "A second paragraph that runs", "onto a second line.", "", "> A quoted passage, set apart.", "", "- The first point", "- \\# The second point", "", "![A chart of the numbers]()", "", "> Embedded post: ", ">", "> An embedded post's own words.", "", "[Further reading]()", "", "The last word: 3 < 4.", "", ].join("\n"), ); }); test("markdown: an untitled article, no byline, a paragraph that would read as syntax", () => { const md = xArticleMarkdown({ extraction: "fallback", url: "https://x.com/i/article/1", blocks: [{ type: "paragraph", text: "1. not a list\n> not a quote" }, { type: "link", href: "https://example.com/a b" }], }); assert.equal( md, "# Untitled article\n\n\n\n1\\. not a list\n\\> not a quote\n\n" + "[https://example.com/a b]()\n", ); });