// The XenForo thread parser over SYNTHETIC pages (__fixtures__/xenforoPages.ts): // posts, quotes, links, media, edits, the page nav, a browser-saved page, and // the pages that stand in a thread's way. // // Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xenforoParse.test.ts import { test } from "node:test"; import assert from "node:assert/strict"; import { classifyForumPage, embedTarget, parseXenforoThreadPage, parseXenforoThreadUrl, resolveForumUrl, xenforoPageUrl, xenforoThreadHandle, } from "./xenforoParse"; import { parsePost } from "../lib/posts"; import { CAPTCHA_PAGE, CHALLENGE_PAGE, LOGIN_PAGE, NOT_FOUND_PAGE, ORIGIN, RICH_BODY, THREAD_URL, threadPage, type FakePost, } from "./__fixtures__/xenforoPages"; const T0 = 1_760_000_000; // epoch seconds function post(id: number, position: number, body = `Post number ${position}.`, extra: Partial = {}): FakePost { return { id, author: `Member${id % 5}`, userId: 100 + (id % 5), ts: T0 + position * 60, position, body, ...extra }; } test("thread URLs: friendly, paged, prefixed and non-friendly forms", () => { const t = parseXenforoThreadUrl(`${THREAD_URL}page-7`); assert.deepEqual(t, { origin: ORIGIN, host: "forum.example", base: THREAD_URL, threadId: "4242", page: 7, }); assert.equal(parseXenforoThreadUrl(`${THREAD_URL}post-99`)?.base, THREAD_URL); assert.equal(parseXenforoThreadUrl("https://forum.example/community/threads/x.12/")?.base, "https://forum.example/community/threads/x.12/"); assert.equal(parseXenforoThreadUrl("https://forum.example/index.php?threads/x.12/")?.threadId, "12"); assert.equal(parseXenforoThreadUrl("https://forum.example/threads/12/")?.threadId, "12"); assert.equal(parseXenforoThreadUrl("https://forum.example/members/someone.3/"), null); assert.equal(parseXenforoThreadUrl("not a url"), null); assert.equal(xenforoPageUrl({ base: THREAD_URL }, 1), THREAD_URL); assert.equal(xenforoPageUrl({ base: THREAD_URL }, 3), `${THREAD_URL}page-3`); assert.equal(xenforoThreadHandle(`${THREAD_URL}page-2`), "the-teapot-collectors-thread.4242"); }); test("a page: thread metadata, page numbers and every post in page order", () => { const html = threadPage({ page: 3, last: 9, posts: [post(301, 41), post(302, 42), post(303, 43)] }); const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); assert.equal(page.origin, ORIGIN); assert.equal(page.host, "forum.example"); assert.equal(page.threadId, "4242"); assert.equal(page.threadTitle, "The Teapot Collectors Thread"); assert.equal(page.threadUrl, THREAD_URL); assert.equal(page.page, 3); assert.equal(page.lastPage, 9); assert.deepEqual(page.posts.map((p) => p.id), ["301", "302", "303"]); const p = page.posts[1]; assert.equal(p.slug, "teapots/302"); assert.equal(p.platform, "xenforo"); assert.equal(p.author, "Member2"); assert.equal(p.createdAt, new Date((T0 + 42 * 60) * 1000).toISOString()); assert.equal(p.uploadDate.length, 8); assert.equal(p.url, `${ORIGIN}/posts/302/`); assert.equal(p.text, "Post number 42."); assert.equal(p.isReply, true); assert.equal(p.isRepost, false); assert.deepEqual(p.forum, { host: "forum.example", threadId: "4242", page: 3, threadTitle: "The Teapot Collectors Thread", threadUrl: THREAD_URL, position: 42, authorId: "102", }); // Round-trips through the stored-record validator unchanged. assert.deepEqual(parsePost(JSON.parse(JSON.stringify(p))), p); }); test("positions with thousands separators, the first post, and a single-page thread", () => { const html = threadPage({ page: 1, last: 1, posts: [post(1, 1, "Opening post."), post(2, 2101)] }); const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); assert.equal(page.page, 1); assert.equal(page.lastPage, 1); assert.equal(page.posts[0].forum?.position, 1); assert.equal(page.posts[0].isReply, false); assert.equal(page.posts[1].forum?.position, 2101); }); test("the body: quotes, mentions, links, smilies, images, embeds, spoilers, video, unfurls, lists, attachments", () => { const html = threadPage({ page: 2, last: 2, posts: [post(500, 30, RICH_BODY)] }); const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; const lines = p.text.split("\n"); // The quote: a header naming the quoted member and post, then its lines. assert.deepEqual(lines.slice(0, 3), [ "> Quoting Marigold (post 1001):", "> The blue one is a reproduction.", "> Look at the glaze.", ]); assert.ok(!p.text.includes("Click to expand"), "the expand link is chrome"); assert.ok(!p.text.includes("Marigold said:"), "the quote title is replaced by the header"); assert.ok(p.text.includes("I disagree, @Marigold."), "a mention is its text"); // A blank line survives

. assert.ok(/@Marigold\.\n\nHere is the catalogue/.test(p.text)); assert.ok(p.text.includes("Here is the catalogue: https://archive.example/AbCd1")); assert.ok(p.text.includes("museum page (https://museum.example/teapots?id=9) :)")); assert.ok(p.text.includes("[image]")); assert.ok(p.text.includes("[embed: https://www.youtube.com/watch?v=AbCdEfGhIjK]")); assert.ok(p.text.includes("[embed: https://x.com/i/status/1234567890123456789]")); assert.ok(p.text.includes("[Spoiler: the ending]\nThe lid was glued on.\n[/Spoiler]")); assert.ok(p.text.includes("[video]")); assert.ok(p.text.includes("https://news.example/teapot-auction")); assert.ok(!p.text.includes("An auction report"), "an unfurl card is its URL, not its blurb"); assert.ok(p.text.includes("- first point\n- second point")); assert.ok(!/ /.test(p.text), "no runs of spaces"); assert.deepEqual(p.forum?.quotes, [{ postId: "1001", author: "Marigold", authorId: "7" }]); assert.deepEqual(p.quoted, { platform: "xenforo", id: "1001", url: `${ORIGIN}/posts/1001/`, author: "Marigold", }); assert.deepEqual(p.replyTo, p.quoted); assert.deepEqual(p.links, [ "https://archive.example/AbCd1", "https://museum.example/teapots?id=9", "https://www.youtube.com/watch?v=AbCdEfGhIjK", "https://x.com/i/status/1234567890123456789", "https://news.example/teapot-auction", ]); assert.deepEqual(p.media, [ { kind: "image", url: "https://images.example/teapot.jpg", name: "teapot.jpg" }, { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK", provider: "youtube" }, { kind: "embed", url: "https://x.com/i/status/1234567890123456789", provider: "twitter" }, { kind: "video", url: `${ORIGIN}/data/video/12/12345-abc.mp4` }, { kind: "link-card", url: "https://news.example/teapot-auction" }, { kind: "attachment", url: `${ORIGIN}/attachments/receipt-png.555/` }, { kind: "image", url: `${ORIGIN}/data/attachments/0/555-receipt.jpg`, name: "receipt.png" }, ]); assert.equal(p.mediaCount, p.media?.length); }); test("an edited post records when, and listed attachments are media", () => { const html = threadPage({ page: 1, last: 1, posts: [ post(600, 5, "Edited words.", { editedTs: T0 + 9999, attachments: [{ href: "/attachments/scan-jpg.900/", name: "scan.jpg" }], }), ], }); const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; assert.equal(p.forum?.editedAt, new Date((T0 + 9999) * 1000).toISOString()); // The edit time is not the post time. assert.equal(p.createdAt, new Date((T0 + 5 * 60) * 1000).toISOString()); assert.deepEqual(p.media, [{ kind: "attachment", url: `${ORIGIN}/attachments/scan-jpg.900/`, name: "scan.jpg" }]); }); test("a browser-saved page: no canonical link, rewritten assets, the saved-from comment names it", () => { const html = threadPage({ page: 4, last: 6, saved: true, posts: [post(700, 61), post(701, 62)] }); assert.equal(classifyForumPage(html), null); const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); assert.equal(page.origin, ORIGIN); assert.equal(page.threadId, "4242"); assert.equal(page.page, 4); assert.equal(page.lastPage, 6); assert.equal(page.posts.length, 2); assert.equal(page.posts[0].url, `${ORIGIN}/posts/700/`); // With no saved-from comment either, the caller's URL is the fallback, and // the content key still names the thread. const bare = html.replace(//, ""); const fromFallback = parseXenforoThreadPage(bare, { channelSlug: "teapots", pageUrl: THREAD_URL }); assert.equal(fromFallback.origin, ORIGIN); const noUrl = parseXenforoThreadPage(bare, { channelSlug: "teapots" }); assert.equal(noUrl.threadId, "4242"); assert.equal(noUrl.posts.length, 2); }); test("URL resolution keeps saved-asset paths and absolutises forum paths", () => { assert.equal(resolveForumUrl("/attachments/a.1/", ORIGIN), `${ORIGIN}/attachments/a.1/`); assert.equal(resolveForumUrl("./Thread_files/a.jpg", ORIGIN), "./Thread_files/a.jpg"); assert.equal(resolveForumUrl("Thread_files/a.jpg", ORIGIN), "Thread_files/a.jpg"); assert.equal(resolveForumUrl("//cdn.example/x.png", ORIGIN), "https://cdn.example/x.png"); assert.equal(resolveForumUrl("data:image/png;base64,AAAA", ORIGIN), undefined); assert.equal(resolveForumUrl("#post-3", ORIGIN), undefined); assert.deepEqual(embedTarget("https://rumble.com/embed/v1abc/?pub=4"), { url: "https://rumble.com/embed/v1abc/", provider: "rumble", }); }); test("pages in the way: a browser check, a captcha, a login, a missing thread, a refusal", () => { const thread = threadPage({ page: 1, last: 1, posts: [post(1, 1)] }); assert.equal(classifyForumPage(thread), null); assert.equal(classifyForumPage(thread, 200), null); const challenge = classifyForumPage(CHALLENGE_PAGE, 403); assert.equal(challenge?.kind, "challenge"); assert.equal(challenge?.title, "Checking your browser"); assert.equal(classifyForumPage(CAPTCHA_PAGE)?.kind, "captcha"); assert.equal(classifyForumPage(LOGIN_PAGE)?.kind, "login"); assert.equal(classifyForumPage(NOT_FOUND_PAGE)?.kind, "not-found"); assert.equal(classifyForumPage("nothing", 429)?.kind, "blocked"); assert.equal(classifyForumPage("nothing", 404)?.kind, "not-found"); assert.equal(classifyForumPage("You have been banned.")?.kind, "blocked"); assert.equal(classifyForumPage("nothing")?.kind, "unknown"); // A check page that names a captcha in passing is still a check (waited // out), not a captcha (stopped at once). assert.equal( classifyForumPage(CHALLENGE_PAGE.replace("", "

No captcha needed.

"))?.kind, "challenge", ); }); test("an article thread's first post, atop a later page with no #N, is post 1 of page 1", () => { const html = threadPage({ page: 3, last: 4, article: post(1, 1, "The article."), posts: [post(41, 41), post(42, 42)] }); const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); assert.deepEqual(page.posts.map((p) => [p.id, p.forum?.position, p.forum?.page]), [ ["1", 1, 1], ["41", 41, 3], ["42", 42, 3], ]); assert.equal(page.posts[0].isReply, false); }); test("a message with no id or no date is not a post", () => { const html = threadPage({ page: 1, last: 1, posts: [post(1, 1)] }) .replace('data-timestamp="', 'data-x="') .replace(/datetime="[^"]*"/, ""); assert.equal(parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts.length, 0); }); test("a Kiwi Farms player upload is video media named by its file, not a bare duration", () => { const player = `
` + `` + `3:29
`; const html = threadPage({ page: 1, last: 1, posts: [post(700, 7, `Before it goes:${player}Watch it.`)] }); const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; assert.deepEqual(p.media, [ { kind: "video", url: "https://uploads.kiwifarms.st/data/video/9619/9619135-10fa.mp4?hash=lZ6", name: "Teapot Intro [aLfKTn4x7q8].mp4" }, ]); assert.ok(p.text.includes("[video: Teapot Intro [aLfKTn4x7q8].mp4 (3:29)]"), p.text); assert.ok(!/^3:29$/m.test(p.text), "the duration label is not a line of its own"); });