// A XenForo forum THREAD as a posts source (platform "xenforo"): the channel is // one thread, each forum post a Post. Kiwi Farms is the first host; any // XenForo 2 forum whose thread pages a browser can open works the same way. // // THE WALK, NEWEST FIRST. A thread grows at its end, so a run starts on the // last page (asked for with a page number past the end, which XenForo // redirects to the last page) and steps back a page at a time. It stops at the // first page holding an already-archived post (the incremental case), at the // watermark, at page 1, or at a cap — `pages` ("the latest N pages") or // `limit` (posts), checked after a whole page so a page is never half-read. // The cursor is the next page to read, so a capped or drained run resumes // there — after catching up from the last page first, so a run always starts // with the newest posts; past the cursor the walk goes on to page 1 (pages // shift as posts are deleted, so meeting an archived post there does not end // it). // // PACED, SERIAL, ONE BROWSER. One page at a time with a jittered pause between // loads (about 10–20 s by default; `pagePauseMs` from the channel), in one // headless Chromium on the host's persistent profile (forumSession.ts). Never // parallel, never faster on a slow run. // // A BROWSER CHECK (KiwiFlare's proof of work and the like) is waited out by the // loader, up to a minute; the profile keeps the clearance for later pages and // runs. A check that does not clear, a captcha, a login wall, a ban or a // 403/429 STOPS the run with a typed failure and keeps the cursor — never a // retry loop, never an attempt at a captcha. `needsCookies` is set where // Connect (a headed window on the same profile) is the way through. // // One parser for everything: xenforoParse.ts reads the live page here and the // operator's saved pages in the import (controller/importForumPages.ts). import { mkdir, writeFile } from "node:fs/promises"; import path from "node:path"; import { getPaths } from "../lib/paths"; import type { Post, PostMedia } from "../lib/posts"; import { registerSocialFetcher, type PostCaptureInput, type PostCaptureOutcome, type PostCaptureResult, type PostFetchInput, type PostFetchResult, type SocialFetcher, type SocialFetcherProbe, } from "./fetchers"; import { browserForumLoader, forumProfileDir, type ForumPageLoader, } from "./forumSession"; import { captureAvailability, captureWork, describeCapturedFile, postCaptureDir, readPostCapture, SHOT_FILENAME, writePostCapture, type CapturedFile, type CaptureMediaState, type PostCaptureRecord, type PostCaptureState, } from "./postCapture"; import type { PageLike } from "./playwrightRuntime"; import { classifyForumPage, parseXenforoThreadPage, parseXenforoThreadUrl, xenforoPageUrl, xenforoThreadHandle, type ForumBlock, type XenforoThreadUrl, } from "./xenforoParse"; // A page number past any real thread's end: XenForo redirects it to the last // page, which saves reading page 1 just to learn where the end is. export const LAST_PAGE_PROBE = 1_000_000; export const DEFAULT_PAGE_PAUSE_MS = 12_000; // The floor a configured pause is held to. export const MIN_PAGE_PAUSE_MS = 5_000; // The pause before a load: `base` jittered to [0.85, 1.65) of itself — 12 s // gives 10.2–19.8 s. export function jitteredPause(base: number, random: () => number): number { const b = Math.max(MIN_PAGE_PAUSE_MS, base); return Math.round(b * (0.85 + random() * 0.8)); } export function sleepUnlessAborted(ms: number, signal: AbortSignal): Promise { return new Promise((resolve) => { if (signal.aborted) return resolve(); const timer = setTimeout(done, ms); function done() { clearTimeout(timer); signal.removeEventListener("abort", done); resolve(); } signal.addEventListener("abort", done, { once: true }); }); } // What a stop on a forum block says, and whether Connect is the way through. export function forumBlockFailure( block: ForumBlock, host: string, ): { error: string; needsCookies: boolean } { const connect = `Use “Connect forum session” on the channel page: it opens a browser window on this host's profile ` + `at the thread — clear it there, wait for the thread to show, close the window, and fetch again.`; const shown = block.title ? ` (page title “${block.title}”)` : ""; switch (block.kind) { case "challenge": return { error: `${host} kept its browser check (${block.detail}) up past the wait${shown}; the headless browser could not clear it. ${connect}`, needsCookies: true, }; case "captcha": return { error: `${host} asked for a captcha (${block.detail})${shown}, which only a person answers. ${connect}`, needsCookies: true, }; case "login": return { error: `${host} shows this thread only to a logged-in member${shown}. ${connect} (log in in that window).`, needsCookies: true, }; case "blocked": return { error: `${host} refused the page (${block.detail})${shown}. Stopped; the next run resumes here — do not run it again at once.`, needsCookies: false, }; case "not-found": return { error: `${host} says the thread could not be found (${block.detail}).`, needsCookies: false }; default: return { error: `${host} answered with a page that is not a thread page (${block.detail})${shown}. Stopped.`, needsCookies: false, }; } } export type ThreadWalkDeps = { loader: ForumPageLoader; // The gap before each load after the run's first. pauseMs: () => number; sleep: (ms: number, signal: AbortSignal) => Promise; }; // THE WALK. Pure over its deps: the loader hands back HTML, the parser reads it. // // Two modes. "new": from the last page down, stopping at the first archived // post (or the watermark) — what a run is for. "backfill": from a stored // cursor down to page 1, where an archived post means nothing (pages shift as // posts are deleted). A run with a cursor does both, newest first: it catches // up from the last page, then continues the older pages at the cursor — so // "the latest N pages" are always the newest, and an unfinished history is // never forgotten. export async function walkXenforoThread( input: PostFetchInput, deps: ThreadWalkDeps, ): Promise { const { channelSlug, seenIds, signal, onLog } = input; const thread = parseXenforoThreadUrl(input.accountUrl); if (!thread) { return { posts: [], complete: false, error: `${input.accountUrl} is not a XenForo thread URL (…/threads/.<id>/).` }; } let resumeAt: number | undefined; if (input.cursor) { resumeAt = Number(input.cursor); if (!Number.isInteger(resumeAt) || resumeAt < 1) { return { posts: [], complete: false, error: `The stored resume point "${input.cursor}" is not a page number.` }; } } const stopAtKnown = input.stopAtKnown ?? true; // A full re-walk (stopAtKnown false) reads every page from the end; a // cursor means nothing to it. if (!stopAtKnown) resumeAt = undefined; const watermark = input.since; const posts: Post[] = []; const taken = new Set<string>(); let returnedCount = 0; let loads = 0; let pagesRead = 0; const fetchPage = async (url: string) => { if (loads++ > 0) await deps.sleep(deps.pauseMs(), signal); return deps.loader.load(url, signal); }; let mode: "new" | "backfill" = "new"; let page = LAST_PAGE_PROBE; if (resumeAt) onLog?.(`An earlier run stopped at page ${resumeAt}: catching up from the last page first, then continuing there.`); let lastPage: number | undefined; const visited = new Set<number>(); const cursorOf = (n: number) => (n === LAST_PAGE_PROBE ? (resumeAt ? String(resumeAt) : undefined) : String(n)); try { for (;;) { if (signal.aborted) { const cursor = cursorOf(page); return { posts, complete: false, ...(cursor ? { cursor } : {}) }; } const url = xenforoPageUrl(thread, page); let load; try { load = await fetchPage(url); } catch (err) { const cursor = cursorOf(page); return { posts, complete: false, ...(cursor ? { cursor } : {}), error: (err as Error).message }; } if (signal.aborted) { const cursor = cursorOf(page); return { posts, complete: false, ...(cursor ? { cursor } : {}) }; } const block = classifyForumPage(load.html, load.status); if (block) { const why = forumBlockFailure(block, thread.host); const cursor = cursorOf(page); onLog?.(`Page ${page === LAST_PAGE_PROBE ? "(last)" : page}: ${block.kind} — ${block.detail}.`); return { posts, complete: false, ...(cursor ? { cursor } : {}), error: why.error, ...(why.needsCookies ? { needsCookies: true } : {}), }; } const parsed = parseXenforoThreadPage(load.html, { channelSlug, pageUrl: load.url }); if (visited.has(parsed.page)) { return { posts, complete: false, ...(cursorOf(page) ? { cursor: cursorOf(page) } : {}), error: `Asked for page ${page} and was shown page ${parsed.page} again; stopped rather than loop.`, }; } visited.add(parsed.page); page = parsed.page; lastPage = Math.max(lastPage ?? 0, parsed.lastPage); pagesRead++; // Catching up has reached the pages the backfill has yet to read. if (mode === "new" && resumeAt && page <= resumeAt) mode = "backfill"; let sawKnown = false; let sawOld = false; const fresh: Post[] = []; // Newest first within the page too, so the stops read in time order. for (const post of [...parsed.posts].reverse()) { // A post that belongs to another page — an article thread's first // post, shown atop every page — says nothing about where this page // stands: it never stops the walk, and is taken once if new. const repeated = post.forum?.page !== undefined && post.forum.page !== parsed.page; if (repeated) { if (!seenIds.has(post.id) && !taken.has(post.id)) { taken.add(post.id); fresh.push(post); } continue; } if (seenIds.has(post.id)) { if (mode === "new" && stopAtKnown) sawKnown = true; continue; } if (mode === "new" && watermark && post.createdAt <= watermark) { sawOld = true; continue; } if (taken.has(post.id)) continue; taken.add(post.id); fresh.push(post); } fresh.reverse(); onLog?.( `Page ${page}/${lastPage}: ${parsed.posts.length} post(s) parsed, ${fresh.length} new` + (load.challengeMs !== undefined ? ` (browser check cleared in ${Math.round(load.challengeMs / 1000)} s)` : "") + (sawKnown ? ", reached already-archived posts" : "") + (sawOld ? ", reached the watermark" : "") + ".", ); if (parsed.posts.length === 0 && page > 1) { // A thread page with no posts in it is not a page to step past // silently: it is most likely markup this parser does not know. return { posts, complete: false, cursor: String(page), error: `Page ${page} of the thread parsed to no posts; stopped rather than walking on.`, }; } let next = page - 1; if (sawKnown || sawOld) { if (!resumeAt) { if (input.onCheckpoint && fresh.length > 0) await input.onCheckpoint({ posts: fresh, cursor: String(Math.max(1, next)) }); else posts.push(...fresh); return { posts, complete: true }; } // Caught up: on to the older pages an earlier run left. mode = "backfill"; next = Math.min(resumeAt, next); onLog?.(`Caught up with the archive; continuing the older pages at page ${next}.`); } returnedCount += fresh.length; if (input.onCheckpoint && fresh.length > 0) { await input.onCheckpoint({ posts: fresh, cursor: String(Math.max(1, next)) }); } else { posts.push(...fresh); } if (next < 1) return { posts, complete: true }; if (input.pages && pagesRead >= input.pages) { onLog?.(`Read ${pagesRead} page(s), the cap for this run; the next run continues at page ${next}.`); return { posts, complete: false, cursor: String(next) }; } if (input.limit && returnedCount >= input.limit) { return { posts, complete: false, cursor: String(next) }; } if (input.drain?.aborted) { onLog?.(`Drained; the next run continues at page ${next}.`); return { posts, complete: false, cursor: String(next), drained: true }; } page = next; } } finally { await deps.loader.close().catch(() => {}); } } // --- capture ------------------------------------------------------------------------- // What the page shows for one post: the post's own article (its box, in // document coordinates), and whatever the page says instead. type ForumPostSnapshot = { rect: { x: number; y: number; width: number; height: number } | null; }; const SNAPSHOT_SCRIPT = (id: string) => `(() => { const el = document.querySelector('article.message[data-content="post-${id}"]') || document.getElementById('js-post-${id}'); if (!el) return { rect: null }; el.scrollIntoView({ block: "center" }); const r = el.getBoundingClientRect(); return { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height } }; })()`; // Let the post's images load (each capped), and open its spoilers, so the shot // shows what the post holds. const PREPARE_SCRIPT = (id: string) => `(() => { const el = document.querySelector('article.message[data-content="post-${id}"]') || document.getElementById('js-post-${id}'); if (!el) return 0; for (const s of el.querySelectorAll('.bbCodeSpoiler')) s.classList.add('is-active'); for (const c of el.querySelectorAll('.bbCodeBlock--expandable')) c.classList.add('is-expanded'); const imgs = Array.from(el.querySelectorAll('img')).filter((i) => !i.complete); return Promise.all(imgs.map((i) => new Promise((r) => { i.addEventListener('load', r, { once: true }); i.addEventListener('error', r, { once: true }); setTimeout(r, 5000); }))).then(() => imgs.length); })()`; export type ForumShotResult = { state: PostCaptureState; shot?: CapturedFile; error?: string; stop?: string; // The post's media as its page reads NOW, parsed from the page just loaded. // An archived record keeps what the parser of its day saw: before the // player fix, a forum-hosted video was a bare duration and no media. liveMedia?: PostMedia[]; }; export async function shootForumPost( load: (url: string) => Promise<{ html: string; status?: number }>, page: PageLike, postUrl: string, id: string, dir: string, host: string, ): Promise<ForumShotResult> { let got; try { got = await load(postUrl); } catch (err) { return { state: "error", error: (err as Error).message }; } const block = classifyForumPage(got.html, got.status); if (block) { if (block.kind === "not-found") return { state: "deleted" }; const why = forumBlockFailure(block, host); return { state: block.kind === "challenge" || block.kind === "captcha" || block.kind === "login" ? "login-wall" : "error", error: why.error, stop: why.error, }; } let liveMedia: PostMedia[] | undefined; try { liveMedia = parseXenforoThreadPage(got.html, { channelSlug: "capture", pageUrl: postUrl }).posts.find( (p) => p.id === id, )?.media; } catch { liveMedia = undefined; } await page.waitForTimeout(1_000); await page.evaluate(PREPARE_SCRIPT(id)).catch(() => {}); const snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as ForumPostSnapshot; if (!snap.rect) { // The thread rendered without this post: XenForo shows a deleted post to // nobody but moderators, and a moved post redirects elsewhere. return { state: "deleted" }; } const r = snap.rect; if (r.width < 1 || r.height < 1) return { state: "error", error: "The post rendered with no size to shoot." }; let png: Uint8Array; try { png = await page.screenshot({ type: "png", fullPage: true, clip: { x: Math.max(0, Math.floor(r.x)), y: Math.max(0, Math.floor(r.y)), width: Math.ceil(r.width), height: Math.ceil(r.height) }, }); } catch (err) { return { state: "error", error: `The screenshot failed: ${((err as Error).message ?? "").split("\n")[0]}` }; } await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, SHOT_FILENAME), png); return { state: "captured", shot: await describeCapturedFile(dir, SHOT_FILENAME, postUrl), ...(liveMedia ? { liveMedia } : {}) }; } const EXT_BY_TYPE: Record<string, string> = { "image/jpeg": "jpg", "image/png": "png", "image/gif": "gif", "image/webp": "webp", "image/avif": "avif", "video/mp4": "mp4", "video/webm": "webm", "application/pdf": "pdf", }; function extFor(url: string, contentType: string | undefined): string { const ct = (contentType ?? "").split(";")[0].trim().toLowerCase(); if (EXT_BY_TYPE[ct]) return EXT_BY_TYPE[ct]; const m = /\.([a-z0-9]{2,5})(?:[?#/]|$)/i.exec(new URL(url).pathname.replace(/-([a-z0-9]{2,5})\.\d+\/?$/i, ".$1")); return m ? m[1].toLowerCase() : "bin"; } // The media a capture downloads: images, attachments and videos the post // carries as files. Embeds and link cards are pages, not files. export function downloadableMedia( media: ReadonlyArray<{ kind: string; url: string }> | undefined, ...more: ReadonlyArray<ReadonlyArray<{ kind: string; url: string }> | undefined> ): { kind: string; url: string }[] { const out: { kind: string; url: string }[] = []; for (const list of [media, ...more]) { for (const m of list ?? []) { if (!(m.kind === "image" || m.kind === "attachment" || m.kind === "video")) continue; if (!/^https?:\/\//.test(m.url) || out.some((x) => x.url === m.url)) continue; out.push(m); } } return out; } // A video can be hundreds of megabytes; an image is not. const MEDIA_TIMEOUT_MS = { video: 600_000, other: 60_000 }; export type ForumCaptureDeps = { loader: ForumPageLoader & { page(): Promise<PageLike> }; pauseMs: () => number; sleep: (ms: number, signal: AbortSignal) => Promise<void>; now?: () => Date; }; const STOP_AFTER_ERRORS = 3; // The capture loop for forum posts: per id, the post's page (its permalink, // which XenForo resolves to the post in its thread), the shot of its article, // then its media fetched through the same browser profile (so a clearance // cookie covers the downloads too). Every contact after the first is paced. export async function captureForumPosts( input: PostCaptureInput, deps: ForumCaptureDeps, ): Promise<PostCaptureResult> { const { signal, onLog } = input; const thread = parseXenforoThreadUrl(input.accountUrl ?? ""); if (!thread) { return { outcomes: [], stoppedEarly: `${input.accountUrl ?? "(no URL)"} is not a XenForo thread URL.` }; } const wanted = { shots: input.shots ?? true, media: input.media ?? true, force: input.force ?? false }; const now = deps.now ?? (() => new Date()); const outcomes: PostCaptureOutcome[] = []; let contacts = 0; let errorsInARow = 0; const contact = async () => { if (contacts++ > 0) await deps.sleep(deps.pauseMs(), signal); }; const stopped = (why: string, extra: Partial<PostCaptureResult> = {}): PostCaptureResult => { onLog?.(why); return { outcomes, stoppedEarly: why, ...extra }; }; try { for (const [i, id] of input.ids.entries()) { if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); if (input.drain?.aborted) return stopped("Drained; the rest are left for a later run."); const dir = postCaptureDir(input.outDir, id); const existing = await readPostCapture(dir); const work = captureWork(existing, { ...wanted, articles: false }); if (!work.shot && !work.media) { onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`); continue; } onLog?.(`[${i + 1}/${input.ids.length}] ${id}`); const archived = input.archived?.get(id); const postUrl = archived?.url && /^https?:\/\//.test(archived.url) ? archived.url : `${thread.origin}/posts/${id}/`; let state: PostCaptureState | undefined = work.shot ? undefined : existing?.state; let liveMedia: PostMedia[] | undefined; let shot = work.shot ? undefined : existing?.shot; let error: string | undefined; let stop: string | undefined; if (work.shot) { await contact(); if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); const page = await deps.loader.page(); const res = await shootForumPost( (url) => deps.loader.load(url, signal), page, postUrl, id, dir, thread.host, ); state = res.state; shot = res.shot; error = res.error; stop = res.stop; liveMedia = res.liveMedia; } let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped"); let media: CapturedFile[] = work.media ? [] : (existing?.media ?? []); const postIsThere = state === undefined || state === "captured"; if (work.media && postIsThere && !stop) { // What the page holds now first, then anything only the archive kept. const files = downloadableMedia(liveMedia, archived?.media); if (files.length === 0) { mediaState = "none"; state ??= "captured"; } else { const page = await deps.loader.page(); let failed = 0; for (const [n, m] of files.entries()) { await contact(); if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); try { const timeout = m.kind === "video" ? MEDIA_TIMEOUT_MS.video : MEDIA_TIMEOUT_MS.other; const res = await page.request!.get(m.url, { timeout, failOnStatusCode: false }); if (!res.ok()) { failed++; onLog?.(`${id}: media ${n + 1} answered HTTP ${res.status()}.`); continue; } const name = `media-${n + 1}.${extFor(m.url, res.headers()["content-type"])}`; await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, name), await res.body()); media.push(await describeCapturedFile(dir, name, m.url)); } catch (err) { failed++; onLog?.(`${id}: media ${n + 1} failed: ${((err as Error).message ?? "").split("\n")[0]}`); } } mediaState = failed > 0 ? "error" : "ok"; if (failed > 0) error = error ? `${error}; ${failed} media file(s) failed` : `${failed} media file(s) failed`; state ??= "captured"; } } const finalState: PostCaptureState = state ?? "error"; const record: PostCaptureRecord = { version: 1, id, url: postUrl, capturedAt: now().toISOString(), state: finalState, ...(shot ? { shot } : {}), mediaState, media, ...(error ? { error } : {}), }; await writePostCapture(dir, record); outcomes.push({ id, state: finalState, ...(work.shot && captureAvailability(finalState) ? { availability: captureAvailability(finalState) } : {}), files: (shot ? 1 : 0) + media.length, ...(error ? { error } : {}), }); onLog?.( `${id}: ${finalState}` + (shot ? ", shot" : "") + (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") + (error ? ` — ${error}` : ""), ); if (stop) { return stopped(`${stop} Stopped at ${id}; the rest are left for a later run.`, { needsCookies: finalState === "login-wall", }); } errorsInARow = finalState === "error" ? errorsInARow + 1 : 0; if (errorsInARow >= STOP_AFTER_ERRORS) { return stopped(`${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`); } } return { outcomes }; } finally { await deps.loader.close().catch(() => {}); } } // --- the fetcher ---------------------------------------------------------------------- function threadOrThrow(url: string): XenforoThreadUrl { const t = parseXenforoThreadUrl(url); if (!t) throw new Error(`${url} is not a XenForo thread URL (…/threads/<title>.<id>/).`); return t; } const livePause = (base?: number) => () => jitteredPause(base ?? DEFAULT_PAGE_PAUSE_MS, Math.random); export const xenforoFetcher: SocialFetcher = { id: "xenforo-thread", label: "Forum thread (XenForo, headless browser)", platform: "xenforo", fields: { limit: true }, detect(url: string): boolean { return parseXenforoThreadUrl(url) !== null; }, // Offline: a probe that loaded the thread would be a paced page load (and a // browser check) just to fill a form. The URL names the thread. async probe(url: string): Promise<SocialFetcherProbe> { const t = parseXenforoThreadUrl(url); if (!t) return { ok: false, error: "Not a XenForo thread URL (…/threads/<title>.<id>/)." }; const handle = xenforoThreadHandle(url) ?? t.threadId; const words = handle.replace(/\.\d+$/, "").replace(/[-_]+/g, " ").trim(); return { ok: true, name: words ? words.replace(/\b\w/g, (c) => c.toUpperCase()) : `Thread ${t.threadId}`, handle, url: t.base, }; }, async fetch(input: PostFetchInput): Promise<PostFetchResult> { const thread = threadOrThrow(input.accountUrl); const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog }); return walkXenforoThread(input, { loader, pauseMs: livePause(input.pagePauseMs), sleep: sleepUnlessAborted, }); }, async captureByIds(input: PostCaptureInput): Promise<PostCaptureResult> { const thread = threadOrThrow(input.accountUrl ?? ""); const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog }); return captureForumPosts(input, { loader, pauseMs: livePause(input.pagePauseMs), sleep: sleepUnlessAborted, }); }, }; registerSocialFetcher(xenforoFetcher);