// XenForo 2 thread pages → `Post`s. The ONE parser: the live fetcher hands it // the page a headless browser loaded (xenforoFetcher.ts), the import hands it // a page the operator saved from their own browser (controller/importForumPages.ts), and // both get the same posts. // // Pure: no browser, no network, no fs. The HTML is read with the small shared // reader (htmlReader.ts), so every marker below runs in tests against // synthetic pages. // // THE MARKUP (XenForo 2, as served): // article.message[data-author][data-content="post-N"]#js-post-N // (.message--article: an article thread's first post, repeated atop // every page with no "#N") // .message-user … a.username[data-user-id] the author // header.message-attribution // .message-attribution-main time.u-dt[data-timestamp] posted // .message-attribution-opposite a "#N" position // .bbWrapper the body // blockquote.bbCodeBlock--quote[data-quote][data-source="post: N"] // .bbCodeSpoiler, .bbCodeBlock--code, .bbImageWrapper, iframe, video… // .message-attachments li.file a[href] attachments // .message-lastEdit time.u-dt last edited // h1.p-title-value thread title // .pageNav-page(--current) page / last page // // A page SAVED from a browser has rewritten asset URLs (`./Thread_files/…`); // the original absolute URL is kept wherever the markup still carries it // (data-url, data-src, an anchor's href), else what is there is kept. import { postPermalink, uploadDateFromCreatedAt, type ForumPostInfo, type Post, type PostMedia, type PostRef, } from "../lib/posts"; import { XENFORO_THREAD_PATH_RE } from "../lib/detectPlatform.mjs"; import { parseHtml, type HtmlElement, type HtmlNode } from "./htmlReader"; // --- thread URLs --------------------------------------------------------------- export type XenforoThreadUrl = { origin: string; // "https://forum.example" host: string; // The thread's URL with no page: "https://forum.example/threads/a-title.123/" base: string; threadId: string; // The page the URL names, when it names one (/page-N). page?: number; }; // A XenForo thread URL, read; null when the URL is not one. Accepts the // friendly form (/threads/./, under any path prefix) with an // optional /page-N and /post-N, and the non-friendly index.php?threads/… form. export function parseXenforoThreadUrl(url: string): XenforoThreadUrl | null { let u: URL; try { u = new URL(url.trim()); } catch { return null; } if (u.protocol !== "http:" && u.protocol !== "https:") return null; const where = u.pathname + u.search; const m = XENFORO_THREAD_PATH_RE.exec(where); if (!m) return null; const threadId = m[1]; // Everything up to and including the thread segment (and its slash). const end = m.index + m[0].length; let basePath = where.slice(0, end); if (!basePath.endsWith("/")) basePath += "/"; const pageM = /(?:^|\/)page-(\d+)(?:\/|$|[?#])/.exec(where.slice(end)); const out: XenforoThreadUrl = { origin: u.origin, host: u.hostname.toLowerCase(), base: `${u.origin}${basePath}`, threadId, }; if (pageM) out.page = Number(pageM[1]); return out; } // The URL of page `page` of a thread (page 1 is the bare thread URL). export function xenforoPageUrl(t: Pick, page: number): string { return page <= 1 ? t.base : `${t.base}page-${page}`; } // The thread key a channel uses as its handle: "." (or the id). export function xenforoThreadHandle(url: string): string | null { const t = parseXenforoThreadUrl(url); if (!t) return null; const seg = /threads\/([^/?#]+)\/?$/.exec(t.base); return seg ? decodeURIComponent(seg[1]) : t.threadId; } // --- the page's state: a thread, or something in its way ------------------------- export type ForumBlockKind = // A JavaScript check (KiwiFlare's proof of work, a "just a moment" page) // that a real browser normally clears by itself. | "challenge" // A captcha: a person has to answer it. | "captcha" // Refused outright: 403 / 429 / a ban or access-denied page. | "blocked" // The thread is only shown to a logged-in member. | "login" // The forum says the thread (or post) does not exist. | "not-found" // Not a thread page, and nothing above recognised. | "unknown"; export type ForumBlock = { kind: ForumBlockKind; // What was recognised, for the log ("KiwiFlare marker", "HTTP 429"). detail: string; title?: string; }; const POST_MARKER_RE = /data-content="post-\d+"|id="js-post-\d+"/; const THREAD_TEMPLATE_RE = /data-template="thread_view"/; const CAPTCHA_MARKERS: [RegExp, string][] = [ [/h-captcha|hcaptcha\.com/i, "hCaptcha"], [/g-recaptcha|google\.com\/recaptcha|recaptcha\/api/i, "reCAPTCHA"], [/cf-turnstile|challenges\.cloudflare\.com\/turnstile/i, "Turnstile"], ]; // The bare word is weaker than a widget: a browser check's own page may name // it, so it is read only after the check markers. const CAPTCHA_WORD = /\bcaptcha\b/i; const CHALLENGE_MARKERS: [RegExp, string][] = [ [/kiwiflare/i, "KiwiFlare"], [/\/\.sssg\/|\bsssg[_-]/i, "KiwiFlare (sssg)"], [/proof[- ]of[- ]work/i, "a proof-of-work check"], [/checking your browser/i, "a browser check"], [/just a moment\.\.\.|cf-chl|challenge-platform|cf_chl_/i, "Cloudflare's browser check"], [/ddos-guard/i, "DDoS-Guard"], [/please wait while (your request|we) .{0,40}verif/i, "a verification page"], ]; const BLOCK_TEXT: [RegExp, string][] = [ [/you have been banned|your (ip|access) (has been|is) (banned|blocked)/i, "a ban page"], [/access denied|error 1020|403 forbidden/i, "an access-denied page"], [/too many requests|rate limit/i, "a rate-limit page"], ]; const LOGIN_TEXT = /you must be logged[- ]in to do that|you do not have permission to view this page|data-template="login"/i; const NOT_FOUND_TEXT = /the requested (thread|post|page) could not be found|data-template="error"[^>]*>[\s\S]{0,4000}could not be found/i; function titleOf(html: string): string | undefined { const m = /]*>([\s\S]*?)<\/title>/i.exec(html); const t = m?.[1].replace(/\s+/g, " ").trim(); return t || undefined; } // What stands between this page and the thread, or null when it IS a thread // page (it carries posts, or XenForo's thread template). `status` is the HTTP // status the page came with, when known. export function classifyForumPage(html: string, status?: number): ForumBlock | null { if (POST_MARKER_RE.test(html) || THREAD_TEMPLATE_RE.test(html)) return null; const title = titleOf(html); const withTitle = (b: Omit): ForumBlock => (title ? { ...b, title } : b); for (const [re, what] of CAPTCHA_MARKERS) { if (re.test(html)) return withTitle({ kind: "captcha", detail: what }); } for (const [re, what] of CHALLENGE_MARKERS) { if (re.test(html)) return withTitle({ kind: "challenge", detail: what }); } if (CAPTCHA_WORD.test(html)) return withTitle({ kind: "captcha", detail: "a captcha" }); if (LOGIN_TEXT.test(html)) return withTitle({ kind: "login", detail: "the forum asks for a login" }); if (NOT_FOUND_TEXT.test(html) || status === 404) { return withTitle({ kind: "not-found", detail: status === 404 ? "HTTP 404" : "the forum says it could not be found" }); } if (status === 401 || status === 403 || status === 429 || status === 503) { return withTitle({ kind: "blocked", detail: `HTTP ${status}` }); } for (const [re, what] of BLOCK_TEXT) { if (re.test(html)) return withTitle({ kind: "blocked", detail: what }); } return withTitle({ kind: "unknown", detail: "no forum posts on the page" }); } // --- tree helpers ------------------------------------------------------------------ const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string"; const classList = (el: HtmlElement): string[] => (el.attrs.class ?? "").split(/\s+/).filter(Boolean); const hasClass = (el: HtmlElement, c: string) => classList(el).includes(c); const hasClassPrefix = (el: HtmlElement, p: string) => classList(el).some((c) => c.startsWith(p)); function findAll( el: HtmlElement, pred: (e: HtmlElement) => boolean, out: HtmlElement[] = [], skip?: (e: HtmlElement) => boolean, ): HtmlElement[] { for (const c of el.children) { if (!isEl(c)) continue; if (pred(c)) out.push(c); if (skip?.(c)) continue; findAll(c, pred, out, skip); } return out; } function findFirst( el: HtmlElement, pred: (e: HtmlElement) => boolean, skip?: (e: HtmlElement) => boolean, ): HtmlElement | null { for (const c of el.children) { if (!isEl(c)) continue; if (pred(c)) return c; if (skip?.(c)) continue; const hit = findFirst(c, pred, skip); if (hit) return hit; } return null; } // Plain text of an element: whitespace collapsed, scripts dropped. function plainText(node: HtmlNode): string { if (!isEl(node)) return node; if (node.tag === "script" || node.tag === "style" || node.tag === "template") return ""; return node.children.map(plainText).join(""); } const squash = (s: string) => s.replace(/[\s\u00a0]+/g, " ").trim(); // --- URLs ------------------------------------------------------------------------ // A URL from the markup, absolute against the forum's origin. A path a browser // wrote when it SAVED the page (`./Thread_files/x.jpg`, a `file:` URL) is kept // as it is: it names nothing on the forum. export function resolveForumUrl(raw: string | undefined, origin: string | undefined): string | undefined { const v = raw?.trim(); if (!v || v.startsWith("data:") || v.startsWith("javascript:") || v.startsWith("#")) return undefined; if (/^[a-z][a-z0-9+.-]*:/i.test(v)) return v; if (v.startsWith("//")) return `https:${v}`; if (v.startsWith("./") || v.startsWith("../") || /_files\//.test(v)) return v; if (!origin) return v; try { return new URL(v, `${origin}/`).href; } catch { return v; } } // The canonical page URL an embed's iframe stands for, and its provider. export function embedTarget(src: string, provider?: string): { url: string; provider?: string } { const yt = /youtube(?:-nocookie)?\.com\/embed\/([A-Za-z0-9_-]{6,})/.exec(src); if (yt) return { url: `https://www.youtube.com/watch?v=${yt[1]}`, provider: "youtube" }; // s9e's media embeds: an iframe page with the item id in the fragment. const s9e = /s9e\.github\.io\/iframe\/\d+\/([a-z0-9]+)(?:\.min)?\.html#([^&?]+)/i.exec(src); if (s9e) { const kind = s9e[1].toLowerCase(); const id = decodeURIComponent(s9e[2]); if (kind === "twitter" && /^\d+$/.test(id)) { return { url: `https://x.com/i/status/${id}`, provider: "twitter" }; } if (kind === "youtube") return { url: `https://www.youtube.com/watch?v=${id}`, provider: "youtube" }; return { url: src, provider: provider ?? kind }; } const tw = /platform\.twitter\.com\/embed\/.*[?&]id=(\d+)/.exec(src); if (tw) return { url: `https://x.com/i/status/${tw[1]}`, provider: "twitter" }; const rumble = /rumble\.com\/embed\/([A-Za-z0-9]+)/.exec(src); if (rumble) return { url: `https://rumble.com/embed/${rumble[1]}/`, provider: "rumble" }; return provider ? { url: src, provider } : { url: src }; } // --- the body → text ------------------------------------------------------------------ type BodyCtx = { origin?: string; links: string[]; media: PostMedia[]; quotes: NonNullable; }; const BLOCK_TAGS = new Set([ "address", "article", "aside", "blockquote", "dd", "details", "div", "dl", "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section", "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", ]); const SILENT_TAGS = new Set(["script", "style", "noscript", "template", "svg", "button", "input", "select", "canvas"]); function addLink(ctx: BodyCtx, url: string | undefined) { if (url && /^https?:\/\//i.test(url) && !ctx.links.includes(url)) ctx.links.push(url); } function addMedia(ctx: BodyCtx, m: PostMedia) { if (!ctx.media.some((x) => x.url === m.url && x.kind === m.kind)) ctx.media.push(m); } const isQuote = (el: HtmlElement) => el.tag === "blockquote" && (hasClass(el, "bbCodeBlock--quote") || "data-quote" in el.attrs); const isSpoiler = (el: HtmlElement) => hasClass(el, "bbCodeSpoiler") || hasClass(el, "bbCodeInlineSpoiler"); const isUnfurl = (el: HtmlElement) => hasClass(el, "bbCodeBlock--unfurl") || (el.attrs["data-unfurl"] === "true" && !!el.attrs["data-url"]); const isImageWrapper = (el: HtmlElement) => hasClass(el, "bbImageWrapper"); const isSmilie = (el: HtmlElement) => el.tag === "img" && (hasClass(el, "smilie") || hasClassPrefix(el, "smilie--") || "data-shortname" in el.attrs); function imageUrl(el: HtmlElement, ctx: BodyCtx, wrapper?: HtmlElement): string | undefined { return resolveForumUrl( el.attrs["data-url"] || wrapper?.attrs["data-src"] || el.attrs["data-src"] || el.attrs.src, ctx.origin, ); } // A quote's source post id: data-source="post: 123", else the jump link. function quoteSource(el: HtmlElement): { postId?: string; author?: string; authorId?: string } { const out: { postId?: string; author?: string; authorId?: string } = {}; const src = /post:\s*(\d+)/.exec(el.attrs["data-source"] ?? ""); if (src) out.postId = src[1]; if (!out.postId) { const jump = findFirst(el, (e) => e.tag === "a" && hasClass(e, "bbCodeBlock-sourceJump")); const m = /[?&]id=(\d+)/.exec(jump?.attrs.href ?? "") ?? /post-(\d+)/.exec(jump?.attrs["data-content-selector"] ?? "") ?? /\/posts\/(\d+)/.exec(jump?.attrs.href ?? ""); if (m) out.postId = m[1]; } const author = squash(el.attrs["data-quote"] ?? ""); if (author) out.author = author; const member = /member:\s*(\d+)/.exec(el.attrs["data-attributes"] ?? ""); if (member) out.authorId = member[1]; return out; } // A block boundary: a line break that merges with any break beside it, where // "\n" (a
) is a break of its own — so

is a blank line and a // list item is not. const SOFT = "\u0000"; // Render one body subtree. Text is emitted with "\n" for
and SOFT around // blocks; `finish` turns the run into lines. function render(node: HtmlNode, ctx: BodyCtx, out: string[]): void { if (!isEl(node)) { out.push(node.replace(/[\s\u00a0]+/g, " ")); return; } const el = node; if (SILENT_TAGS.has(el.tag) && !(el.tag === "button" && isSpoilerButton(el))) return; if (hasClass(el, "bbCodeBlock-expandLink") || hasClass(el, "bbCodeBlock-title")) return; if (el.tag === "br") { out.push("\n"); return; } if (isQuote(el)) { const src = quoteSource(el); ctx.quotes.push(src); const inner: string[] = []; const content = findFirst(el, (e) => hasClass(e, "bbCodeBlock-content")) ?? el; for (const c of content.children) render(c, ctx, inner); const body = finish(inner); const head = `Quoting ${src.author ?? "an earlier post"}` + (src.postId ? ` (post ${src.postId})` : "") + ":"; const lines = [head, ...(body ? body.split("\n") : [])]; out.push(SOFT, lines.map((l) => (l ? `> ${l}` : ">")).join("\n"), SOFT); return; } if (isSpoiler(el)) { const titleEl = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-button-title")); const title = titleEl ? squash(plainText(titleEl)) : ""; const content = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-content") || hasClass(e, "bbCodeBlock-content")) ?? el; const inner: string[] = []; for (const c of content.children) render(c, ctx, inner); const inline = hasClass(el, "bbCodeInlineSpoiler"); if (inline) { out.push(`[spoiler: ${finish(inner).replace(/\n+/g, " ")}]`); return; } out.push(SOFT, `[Spoiler${title && !/^spoiler$/i.test(title) ? `: ${title}` : ""}]`, "\n", finish(inner), "\n", "[/Spoiler]", SOFT); return; } if (isUnfurl(el)) { const url = resolveForumUrl(el.attrs["data-url"], ctx.origin); if (url) { addLink(ctx, url); addMedia(ctx, { kind: "link-card", url }); out.push(SOFT, url, SOFT); } return; } if (isImageWrapper(el)) { const img = findFirst(el, (e) => e.tag === "img"); const url = img ? imageUrl(img, ctx, el) : resolveForumUrl(el.attrs["data-src"], ctx.origin); if (url) { const name = el.attrs.title || img?.attrs.alt; addMedia(ctx, { kind: "image", url, ...(name ? { name } : {}) }); } out.push("[image]"); return; } if (el.tag === "img") { if (isSmilie(el)) { out.push(el.attrs.alt ?? ""); return; } const url = imageUrl(el, ctx); if (url) addMedia(ctx, { kind: "image", url, ...(el.attrs.alt ? { name: el.attrs.alt } : {}) }); out.push("[image]"); return; } // Kiwi Farms' own player ("ephyra"): a div carrying the upload as data // attributes, its