// X Articles, the pure half: the article link in a post's text or on its card,
// the article root's HTML read into blocks (against a written fixture, never
// x.com), and the markdown written from them.
//
// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xArticle.test.ts
import { test } from "node:test";
import assert from "node:assert/strict";
import { readFile } from "node:fs/promises";
import path from "node:path";
import { fileURLToPath } from "node:url";
import {
decodeEntities,
extractXArticle,
findXArticleLink,
parseHtml,
textOf,
xArticleLinkFromArchive,
xArticleMarkdown,
} from "./xArticle";
import { xArticleLinkFromCard } from "./xArticleCapture";
const HERE = path.dirname(fileURLToPath(import.meta.url));
const FIXTURE = await readFile(path.join(HERE, "__fixtures__", "x-article.html"), "utf8");
// --- the link ----------------------------------------------------------------
test("link: an article post's archived text, its expanded links, or nothing", () => {
assert.deepEqual(xArticleLinkFromArchive({ text: "https://x.com/i/article/1987654321" }), {
articleId: "1987654321",
url: "https://x.com/i/article/1987654321",
});
assert.deepEqual(
xArticleLinkFromArchive({ text: "Read it: https://t.co/abc", links: ["https://twitter.com/i/article/42"] }),
{ articleId: "42", url: "https://x.com/i/article/42" },
);
assert.deepEqual(xArticleLinkFromArchive({ text: "x.com/demo_author/article/77 is up" }), {
articleId: "77",
url: "https://x.com/demo_author/article/77",
});
for (const text of [
"a plain post",
"https://x.com/demo_author/status/123",
"https://example.com/i/article/5",
"https://x.com/i/articles",
]) {
assert.equal(xArticleLinkFromArchive({ text }), null, text);
}
assert.equal(xArticleLinkFromArchive(undefined), null);
});
test("link: a card's hrefs, relative as the page has them", () => {
assert.deepEqual(xArticleLinkFromCard(["/demo_author", "/demo_author/article/1234567890"]), {
articleId: "1234567890",
url: "https://x.com/demo_author/article/1234567890",
});
assert.deepEqual(xArticleLinkFromCard(["/i/article/99/media/1"]), {
articleId: "99",
url: "https://x.com/i/article/99",
});
assert.equal(xArticleLinkFromCard(["/demo_author/status/1/photo/1"]), null);
assert.equal(xArticleLinkFromCard([]), null);
assert.equal(xArticleLinkFromCard(undefined), null);
assert.equal(findXArticleLink([null, undefined, ""]), null);
});
// --- the HTML reader ---------------------------------------------------------
test("html: elements, attributes, void and self-closing tags, entities, comments, raw text", () => {
const root = parseHtml(
'

a <b> 'c' d' +
"
e tail",
);
const div = root.children[0] as { tag: string; attrs: Record; children: unknown[] };
assert.equal(div.tag, "div");
assert.deepEqual(div.attrs, { class: "a", "data-x": "q", hidden: "" });
assert.equal((div.children[0] as { attrs: { src: string } }).attrs.src, "s?a=1&b=2");
// A stray end tag is ignored; the text after the div is the root's.
assert.equal(root.children[1], "tail");
assert.equal(textOf(div as never), "a 'c' de");
assert.equal(decodeEntities("&&unknown;🙂"), "&&unknown;๐");
});
test("html: text reads as innerText would โ line breaks at
and blocks, script and svg silent", () => {
const root = parseHtml("");
assert.equal(textOf(root), "one two\nthree\nfour");
});
// --- the extractor -------------------------------------------------------------
test("extract: title, byline and the body's blocks in reading order, from the written fixture", () => {
const a = extractXArticle(FIXTURE);
assert.equal(a.extraction, "structured");
assert.equal(a.title, "A Worked Example & Its Notes");
assert.equal(a.author, "Demo Author");
assert.equal(a.handle, "@demo_author");
assert.equal(a.publishedAt, "2026-01-02T03:04:05.000Z");
assert.deepEqual(a.blocks, [
// The cover, above the body.
{ type: "image", src: "https://pbs.twimg.com/media/CoverAbc?format=jpg&name=small" },
{ type: "paragraph", text: "The first paragraph, with a linked source and an emoji ๐." },
{ type: "link", text: "linked source", href: "https://example.com/source" },
{ type: "heading", text: "A Section Heading" },
{ type: "paragraph", text: "A second paragraph that runs\nonto a second line." },
{ type: "quote", text: "A quoted passage, set apart." },
{ type: "list-item", text: "The first point" },
{ type: "list-item", text: "# The second point" },
{
type: "image",
src: "https://pbs.twimg.com/media/BodyImg1?format=png&name=small",
text: "A chart of the numbers",
},
{
type: "embedded-post",
text: "An embedded post's own words.",
href: "https://x.com/other_user/status/4444444444",
},
// A paragraph that is only a link is the link.
{ type: "link", text: "Further reading", href: "https://example.com/further-reading" },
{ type: "paragraph", text: "The last word: 3 < 4." },
]);
});
test("extract: no body marker โ the root's text split at block elements, and it says so", () => {
const a = extractXArticle(
'Plain Title
40 likes
',
);
assert.equal(a.extraction, "fallback");
assert.equal(a.title, "Plain Title");
assert.deepEqual(a.blocks, [
{ type: "paragraph", text: "One" },
{ type: "paragraph", text: "Two a handle" },
{ type: "link", text: "a handle", href: "https://x.com/demo_author" },
{ type: "image", src: "https://pbs.twimg.com/media/Fig?format=webp" },
{ type: "list-item", text: "Item" },
{ type: "quote", text: "Said" },
]);
});
test("extract: an empty root is an untitled article with no blocks", () => {
assert.deepEqual(extractXArticle(""), { extraction: "fallback", blocks: [] });
});
// --- markdown ----------------------------------------------------------------
test("markdown: title, byline, the URL, then each block โ saved images linked locally, markdown in text escaped", () => {
const a = extractXArticle(FIXTURE);
const blocks = a.blocks.map((b) =>
b.src?.includes("BodyImg1") ? { ...b, file: "article-img-2.png" } : b,
);
const md = xArticleMarkdown({ ...a, blocks, url: "https://x.com/i/article/1234567890" });
assert.equal(
md,
[
"# A Worked Example & Its Notes",
"",
"By Demo Author @demo_author ยท 2026-01-02T03:04:05.000Z",
"",
"",
"",
"![]()",
"",
"The first paragraph, with a linked source and an emoji ๐.",
"",
"[linked source]()",
"",
"## A Section Heading",
"",
"A second paragraph that runs",
"onto a second line.",
"",
"> A quoted passage, set apart.",
"",
"- The first point",
"- \\# The second point",
"",
"![A chart of the numbers]()",
"",
"> Embedded post: ",
">",
"> An embedded post's own words.",
"",
"[Further reading]()",
"",
"The last word: 3 < 4.",
"",
].join("\n"),
);
});
test("markdown: an untitled article, no byline, a paragraph that would read as syntax", () => {
const md = xArticleMarkdown({
extraction: "fallback",
url: "https://x.com/i/article/1",
blocks: [{ type: "paragraph", text: "1. not a list\n> not a quote" }, { type: "link", href: "https://example.com/a b" }],
});
assert.equal(
md,
"# Untitled article\n\n\n\n1\\. not a list\n\\> not a quote\n\n" +
"[https://example.com/a b]()\n",
);
});