commit 18582324729636b2e40cef5a77a2c7e713e0b12a
parent a38ef94914b2d5ecfc92a5799231fcc96dd28a52
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 14:55:41 -0400
posts: forum threads (XenForo) as a posts source — parser, paced walk, import, capture
A XenForo thread is a posts channel (platform "xenforo"), each forum post a
Post. One parser (social/xenforoParse.ts) reads live pages and browser-saved
pages; the headless walk (social/xenforoFetcher.ts) goes newest first from the
last page, paced and resumable, waits out a browser check on a persistent
per-host profile (social/forumSession.ts) and stops typed when one will not
clear. `archilyzer posts import-html` merges saved pages (upsertPosts: an
edited post is updated, append-only). capture-posts shoots a forum post's
article and downloads its media through the same profile. Post gains optional
`media` and `forum`; the HTML reader moves to social/htmlReader.ts.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
31 files changed, 3934 insertions(+), 147 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -333,7 +333,26 @@ export const COMMANDS: Command[] = [
script(["duplicates"], "duplicate-shorts.ts",
"[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192),
script(["posts", "fetch"], "fetch-posts.ts",
- "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post)"),
+ "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post; --pages: a forum thread's latest N pages)"),
+ {
+ path: ["posts", "import-html"],
+ usage:
+ "<slug> <file-or-dir>… [--dry-run] import forum thread pages saved from a browser (\"Save page as\", .html) into a forum-thread channel: new posts appended, edited ones updated, nothing fetched",
+ flags: { "dry-run": "boolean" },
+ maxPositionals: 1000,
+ run: async ({ positionals, flags }) => {
+ const [slug, ...inputs] = positionals;
+ if (!slug || inputs.length === 0) {
+ console.error("posts import-html: pass the channel slug, then one or more saved pages or directories.");
+ return 2;
+ }
+ return (await import("./posts-import-html")).main({
+ slug,
+ inputs,
+ dryRun: flags["dry-run"] === true,
+ });
+ },
+ },
script(["posts", "check"], "check-post-availability.ts",
"--slug <channel> [--mode stale|unchecked|all] [--limit N] which archived posts were deleted at the source"),
script(["diarize", "backfill"], "diarize-backfill.ts",
diff --git a/common/bin/fetch-posts.ts b/common/bin/fetch-posts.ts
@@ -2,7 +2,11 @@
// Fetch social posts for one channel into its on-disk posts corpus.
//
// pnpm --filter yt-dlp-transcript-common exec tsx bin/fetch-posts.ts \
-// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N]
+// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N]
+//
+// --pages caps how many pages one run reads, for a source read page by page (a
+// forum thread: its latest N pages, newest first; the next run continues where
+// this one stopped).
//
// --full re-walks the account's whole history instead of stopping at the
// stored watermark. New posts are still deduped against the posts-archive, so
@@ -24,7 +28,7 @@ const flags = parseFlags(process.argv.slice(2));
const slug = flags.slug;
if (!slug) {
console.error(
- "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N]",
+ "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N]",
);
process.exit(2);
}
@@ -35,6 +39,12 @@ const limit =
? Math.floor(limitRaw)
: undefined;
+const pagesRaw = flags.pages ? Number(flags.pages) : undefined;
+const pages =
+ typeof pagesRaw === "number" && Number.isFinite(pagesRaw) && pagesRaw > 0
+ ? Math.floor(pagesRaw)
+ : undefined;
+
fetchPosts({
paths: getPaths(),
slug,
@@ -44,6 +54,7 @@ fetchPosts({
floor: flags.floor,
force: flags.force === "true",
limit,
+ pages,
onLog: (line) => console.log(line),
})
.then((result) => {
diff --git a/common/bin/posts-import-html.ts b/common/bin/posts-import-html.ts
@@ -0,0 +1,35 @@
+// `archilyzer posts import-html <slug> <file-or-dir>… [--dry-run]` — forum
+// thread pages the operator saved from a browser, imported into a forum-thread
+// channel (controller/importForumPages.ts). Offline: nothing is fetched. New
+// posts are appended, a post saved again after an edit is updated, and an
+// unchanged one is left alone. The index picks them up on its next build.
+
+import { isValidChannelSlug } from "../controller/channels";
+import { importForumPages } from "../controller/importForumPages";
+import { getPaths } from "../lib/paths";
+
+export async function main(opts: {
+ slug: string;
+ inputs: string[];
+ dryRun: boolean;
+}): Promise<number> {
+ if (!isValidChannelSlug(opts.slug)) {
+ console.error(`posts import-html: "${opts.slug}" is not a channel slug`);
+ return 2;
+ }
+ const result = await importForumPages({
+ paths: getPaths(),
+ slug: opts.slug,
+ inputs: opts.inputs,
+ dryRun: opts.dryRun,
+ onLog: (line) => console.log(line),
+ });
+ if (!result.ok) {
+ console.error(`posts import-html: ${result.error}`);
+ return 1;
+ }
+ for (const p of result.pages) {
+ if (p.skipped) console.log(`skipped ${p.file}: ${p.skipped}`);
+ }
+ return 0;
+}
diff --git a/common/components/PostModal.tsx b/common/components/PostModal.tsx
@@ -14,7 +14,7 @@ import { XIcon } from "lucide-react";
import { Button } from "./ui/button";
import { useUrlParams, writeUrlParams } from "./urlState";
import { fetchPost, fetchThread } from "./postsCache";
-import type { Post } from "../lib/posts";
+import { postPlatformLabel, type Post } from "../lib/posts";
function formatWhen(iso: string): string {
const ms = Date.parse(iso);
@@ -30,7 +30,7 @@ function formatWhen(iso: string): string {
}
function platformLabel(platform: Post["platform"]): string {
- return platform === "twitter" ? "X" : "Bluesky";
+ return postPlatformLabel(platform);
}
// A post's capture (a screenshot and its attached media, taken by the
@@ -185,7 +185,15 @@ function PostBody({
<span>·</span>
<time dateTime={post.createdAt}>{formatWhen(post.createdAt)}</time>
{post.isRepost && <Badge>repost</Badge>}
- {post.isReply && <Badge>reply</Badge>}
+ {/* Every forum post after the first answers the thread; the badge
+ would be on all of them. Its place in the thread says more. */}
+ {post.isReply && !post.forum && <Badge>reply</Badge>}
+ {post.forum?.position && <Badge>#{post.forum.position}</Badge>}
+ {post.forum?.editedAt && (
+ <span title={`Last edited ${formatWhen(post.forum.editedAt)}`}>
+ <Badge>edited</Badge>
+ </span>
+ )}
{/* The whole point of archiving: this post no longer exists upstream. */}
{post.isDeleted && (
<span
@@ -207,6 +215,36 @@ function PostBody({
{capture && <PostCapturePanel capture={capture} />}
+ {post.forum?.threadTitle && !compact && (
+ <p className="mt-1 text-xs text-muted-foreground">
+ In thread: {post.forum.threadTitle}
+ {post.forum.page ? ` · page ${post.forum.page}` : ""}
+ </p>
+ )}
+
+ {post.media && post.media.length > 0 && !capturedMedia && (
+ <ul data-post-media="" className="mt-2 flex flex-col gap-1">
+ {post.media.map((m) => (
+ <li key={`${m.kind}:${m.url}`} className="text-xs text-muted-foreground">
+ {m.kind}
+ {m.provider ? ` (${m.provider})` : ""}:{" "}
+ {/^https?:\/\//.test(m.url) ? (
+ <a
+ href={m.url}
+ target="_blank"
+ rel="noopener noreferrer"
+ className="text-primary underline break-all"
+ >
+ {m.name || m.url}
+ </a>
+ ) : (
+ <span className="break-all">{m.name || m.url}</span>
+ )}
+ </li>
+ ))}
+ </ul>
+ )}
+
{links.length > 0 && (
<ul className="mt-2 flex flex-col gap-1">
{links.map((href) => (
diff --git a/common/components/SearchResults.tsx b/common/components/SearchResults.tsx
@@ -27,6 +27,7 @@ import {
type DuplicateLookup,
} from "./duplicatesCache";
import type { DuplicateVideoRef } from "../lib/duplicates";
+import { postPlatformLabel } from "../lib/posts";
import { splitId } from "./originId";
import { usePlayer } from "./PlayerProvider";
import { formatTimestamp } from "../lib/vtt";
@@ -962,7 +963,7 @@ function sectionScope(
function PostBadge({ platform }: { platform: string }) {
return (
<span className="shrink-0 text-[10px] uppercase tracking-wide font-medium px-1.5 py-0.5 rounded bg-muted text-muted-foreground self-center">
- {platform === "twitter" ? "X" : "Bluesky"}
+ {postPlatformLabel(platform)}
</span>
);
}
diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts
@@ -5,7 +5,12 @@
// federating hub can hold several origins' posts at once without collisions.
import { useQueries, useQuery } from "@tanstack/react-query";
-import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts";
+import {
+ postConversation,
+ type ChannelPostsManifest,
+ type Post,
+ type PostsManifest,
+} from "../lib/posts";
import { PromiseMap } from "../lib/archive/reader";
import { channelRef, readerFor } from "../lib/archive/readers";
import { makeId, splitId } from "./originId";
@@ -102,22 +107,20 @@ export async function fetchThread(id: string): Promise<Post[]> {
const channelSlug = slug.slice(0, slashIdx);
const post = await fetchPost(id);
const threadId = post.threadId || post.id;
+ const forum = post.platform === "xenforo";
// A thread can straddle pages, so scan every page of the channel. Pages are
- // cached, and a channel's page count is small (byte-capped shards).
+ // cached, and a channel's page count is small (byte-capped shards). A forum
+ // post's thread is its conversation (postConversation): the channel IS the
+ // forum thread, so every page is a candidate.
const manifest = await fetchChannelPostsManifest(channelSlug, origin);
- const thread: Post[] = [];
+ const candidates: Post[] = [];
for (let i = 0; i < manifest.pageCount; i++) {
for (const entry of await fetchPostsPage(channelSlug, i, origin)) {
- if ((entry.threadId || entry.id) === threadId) thread.push(entry);
+ if (forum || (entry.threadId || entry.id) === threadId) candidates.push(entry);
}
}
- thread.sort((a, b) =>
- a.createdAt === b.createdAt
- ? a.id.localeCompare(b.id)
- : a.createdAt.localeCompare(b.createdAt),
- );
- return thread.length > 0 ? thread : [post];
+ return postConversation(post, candidates);
}
export function usePostsManifest(origin = "") {
diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts
@@ -42,6 +42,7 @@ import "../social/blueskyFetcher";
import "../social/xGalleryDlFetcher";
import "../social/xPlaywrightFetcher";
import "../social/xNitterFetcher";
+import "../social/xenforoFetcher";
import {
resolveXCookieSourceFor,
type XLoginSettings,
@@ -136,13 +137,27 @@ export async function capturePosts(
const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot));
if (stray) return fail(strayIdsRefusal(slug, stray));
- // The archived text of each id, for the article links in it (X only).
- let archived: Map<string, { text: string; links: string[] }> | undefined;
- if (opts.articles !== false && fetcher.platform === "twitter") {
+ // The archived record of each id: its text and links for the article links
+ // in it (X), its URL and media for a forum post (the capture opens the one
+ // and downloads the other).
+ type Archived = {
+ text: string;
+ links: string[];
+ url?: string;
+ media?: { kind: string; url: string; name?: string }[];
+ };
+ let archived: Map<string, Archived> | undefined;
+ const forum = fetcher.platform === "xenforo";
+ if ((opts.articles !== false && fetcher.platform === "twitter") || forum) {
const wanted = new Set(ids);
archived = new Map();
for (const p of await readAllPosts(channelRoot)) {
- if (wanted.has(p.id)) archived.set(p.id, { text: p.text, links: p.links ?? [] });
+ if (!wanted.has(p.id)) continue;
+ archived.set(p.id, {
+ text: p.text,
+ links: p.links ?? [],
+ ...(forum ? { url: p.url, ...(p.media ? { media: p.media } : {}) } : {}),
+ });
}
}
@@ -169,6 +184,10 @@ export async function capturePosts(
result = await fetcher.captureByIds({
ids,
handle,
+ accountUrl,
+ ...(config.postPagePauseSeconds
+ ? { pagePauseMs: config.postPagePauseSeconds * 1000 }
+ : {}),
outDir: postsMediaDir(channelRoot),
shots: opts.shots,
media: opts.media,
@@ -222,7 +241,9 @@ export async function capturePosts(
ok: false,
outcomes: result.outcomes,
needsCookies: true,
- error: result.stoppedEarly ?? "The capture needs an X login.",
+ error:
+ result.stoppedEarly ??
+ (forum ? "The capture needs the forum session (Connect)." : "The capture needs an X login."),
};
}
const stoppedByOperator =
diff --git a/common/controller/checkPostAvailability.ts b/common/controller/checkPostAvailability.ts
@@ -33,6 +33,7 @@ import {
resolveSocialFetcher,
} from "../social/fetchers";
import "../social/blueskyFetcher";
+import "../social/xenforoFetcher";
import "../social/xGalleryDlFetcher";
import "../social/xPlaywrightFetcher";
import "../social/xNitterFetcher";
diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts
@@ -51,6 +51,8 @@ import "../social/xGalleryDlFetcher";
// only ever used when a channel opts into it via postFetcher: "x-playwright".
import "../social/xPlaywrightFetcher";
import "../social/xNitterFetcher";
+// A forum thread (platform "xenforo"): claims only thread URLs.
+import "../social/xenforoFetcher";
import type { XCookieSource } from "../social/xCookieSource";
import {
resolveXCookieSourceFor,
@@ -82,6 +84,9 @@ export type FetchPostsOptions = {
// emptyAccountOlderProblem). Only with `older`.
force?: boolean;
limit?: number;
+ // Cap on pages read this run, for a fetcher that walks pages (a forum
+ // thread's "latest N pages").
+ pages?: number;
onLog?: (line: string) => void;
signal?: AbortSignal;
// The job's Drain: the fetch stops at its next resume point and keeps it,
@@ -299,6 +304,10 @@ export async function fetchPosts(
cookieSource: xLogin?.source,
browserCookies: xLogin?.browserSpec,
limit: opts.limit,
+ ...(opts.pages ? { pages: opts.pages } : {}),
+ ...(config.postPagePauseSeconds
+ ? { pagePauseMs: config.postPagePauseSeconds * 1000 }
+ : {}),
stopAtKnown: !opts.full,
onCheckpoint,
signal: effectiveSignal,
diff --git a/common/controller/importForumPages.test.ts b/common/controller/importForumPages.test.ts
@@ -0,0 +1,116 @@
+// Importing browser-saved forum thread pages into a forum-thread channel, end
+// to end over a temp corpus: the pages are parsed, posts merged (new ones
+// appended, an edited one updated), and pages that are not this thread's are
+// skipped by name. SYNTHETIC pages only.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/importForumPages.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "forum-import-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+
+const { getPaths } = await import("../lib/paths");
+const { importForumPages } = await import("./importForumPages");
+const { readAllPosts } = await import("../lib/posts-server");
+const { CHALLENGE_PAGE, THREAD_URL, threadPage } = await import("../social/__fixtures__/xenforoPages");
+
+const T0 = 1_760_000_000;
+const post = (id: number, position: number, body: string, editedTs?: number) => ({
+ id,
+ author: `Member${id % 3}`,
+ userId: 10 + (id % 3),
+ ts: T0 + position * 60,
+ position,
+ body,
+ ...(editedTs ? { editedTs } : {}),
+});
+
+async function channel(slug: string, config: Record<string, unknown>) {
+ const dir = path.join(getPaths().channelsDir, slug);
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, "config.json"), JSON.stringify(config));
+}
+
+await channel("teapots", {
+ handling: "youtube",
+ sourceKind: "social",
+ platform: "xenforo",
+ postFetcher: "xenforo-thread",
+ url: THREAD_URL,
+ socialHandle: "the-teapot-collectors-thread.4242",
+});
+await channel("someone-on-bluesky", {
+ handling: "youtube",
+ sourceKind: "social",
+ platform: "bluesky",
+ url: "https://bsky.app/profile/someone.example",
+});
+
+const SAVES = path.join(ROOT, "saves");
+await mkdir(path.join(SAVES, "The Teapot Collectors Thread_files"), { recursive: true });
+await writeFile(
+ path.join(SAVES, "The Teapot Collectors Thread _ Page 2.html"),
+ threadPage({ page: 2, last: 2, saved: true, posts: [post(21, 4, "Fourth."), post(22, 5, "Fifth.")] }),
+);
+await writeFile(
+ path.join(SAVES, "The Teapot Collectors Thread.html"),
+ threadPage({ page: 1, last: 2, saved: true, posts: [post(11, 1, "First."), post(12, 2, "Second."), post(13, 3, "Third.")] }),
+);
+await writeFile(path.join(SAVES, "check.html"), CHALLENGE_PAGE);
+await writeFile(
+ path.join(SAVES, "other-thread.html"),
+ threadPage({ page: 1, last: 1, posts: [post(99, 1, "Elsewhere.")] }).replace(/4242/g, "5151"),
+);
+await writeFile(path.join(SAVES, "The Teapot Collectors Thread_files", "x.jpg"), "not a page");
+
+test("imports a folder of saved pages: posts merged, strangers skipped by name", async () => {
+ const log: string[] = [];
+ const res = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [SAVES], onLog: (l) => log.push(l) });
+ assert.equal(res.ok, true, res.error);
+ assert.equal(res.parsed, 5);
+ assert.equal(res.written, 5);
+ const byFile = Object.fromEntries(res.pages.map((p) => [p.file, p]));
+ assert.equal(byFile["check.html"].skipped?.startsWith("not a thread page (challenge"), true);
+ assert.match(byFile["other-thread.html"].skipped ?? "", /another thread \(5151, not 4242\)/);
+ assert.equal(byFile["The Teapot Collectors Thread _ Page 2.html"].page, 2);
+ assert.equal(res.pages.length, 4, "the _files folder is not walked");
+
+ const posts = await readAllPosts(path.join(getPaths().channelsDir, "teapots"));
+ assert.deepEqual(posts.map((p) => p.id), ["22", "21", "13", "12", "11"]);
+ assert.equal(posts[0].forum?.page, 2);
+ assert.equal(posts[0].channelSlug, "teapots");
+});
+
+test("a later save of an edited post updates it; the same save again changes nothing", async () => {
+ const file = path.join(ROOT, "page-2-again.html");
+ await writeFile(
+ file,
+ threadPage({ page: 2, last: 2, saved: true, posts: [post(21, 4, "Fourth, edited.", T0 + 5000), post(22, 5, "Fifth.")] }),
+ );
+ const res = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [file] });
+ assert.deepEqual([res.written, res.updated, res.unchanged], [0, 1, 1]);
+ const posts = await readAllPosts(path.join(getPaths().channelsDir, "teapots"));
+ assert.equal(posts.find((p) => p.id === "21")?.text, "Fourth, edited.");
+ assert.equal(posts.length, 5);
+
+ const again = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [file] });
+ assert.deepEqual([again.written, again.updated, again.unchanged], [0, 0, 2]);
+});
+
+test("a dry run writes nothing; refusals name the problem", async () => {
+ const dry = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [SAVES], dryRun: true });
+ assert.equal(dry.ok, true);
+ assert.equal(dry.written, 0);
+
+ const notForum = await importForumPages({ paths: getPaths(), slug: "someone-on-bluesky", inputs: [SAVES] });
+ assert.match(notForum.error ?? "", /not a forum-thread channel/);
+ const missing = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [path.join(ROOT, "nope")] });
+ assert.match(missing.error ?? "", /No such file or directory/);
+ const noChannel = await importForumPages({ paths: getPaths(), slug: "nobody", inputs: [SAVES] });
+ assert.match(noChannel.error ?? "", /No such channel/);
+});
diff --git a/common/controller/importForumPages.ts b/common/controller/importForumPages.ts
@@ -0,0 +1,153 @@
+// Import forum thread pages the operator SAVED from their own browser ("Save
+// page as", complete or HTML only) into a forum-thread channel. The second way
+// in beside the live fetcher, through the same parser (xenforoParse.ts) and
+// the same store (posts-server.ts): new posts are appended, a post already
+// archived is updated when the save holds a newer record of it (an edit), and
+// one that arrives unchanged is left alone.
+//
+// Server-only (node:fs). Nothing here touches the network: a saved page's
+// assets are read from the page as it is, never fetched.
+
+import { readdir, readFile, stat } from "node:fs/promises";
+import path from "node:path";
+import { readChannelConfig } from "./channels";
+import { isSocialChannel } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import type { Post } from "../lib/posts";
+import { upsertPosts } from "../lib/posts-server";
+import {
+ classifyForumPage,
+ parseXenforoThreadPage,
+ parseXenforoThreadUrl,
+} from "../social/xenforoParse";
+
+export type ImportForumPagesOptions = {
+ paths: Paths;
+ slug: string;
+ // Files and directories. A directory contributes the .html/.htm files in
+ // it (not the `<name>_files/` asset folders a browser saves beside them).
+ inputs: ReadonlyArray<string>;
+ // Parse and count, write nothing.
+ dryRun?: boolean;
+ onLog?: (line: string) => void;
+};
+
+export type ImportedPage = {
+ file: string;
+ page?: number;
+ posts: number;
+ // Why the file contributed nothing, when it did not.
+ skipped?: string;
+};
+
+export type ImportForumPagesResult = {
+ ok: boolean;
+ error?: string;
+ pages: ImportedPage[];
+ parsed: number;
+ written: number;
+ updated: number;
+ unchanged: number;
+};
+
+const PAGE_EXT = /\.(html?|xhtml)$/i;
+
+// The page files the inputs name, sorted, each once.
+export async function listSavedPages(inputs: ReadonlyArray<string>): Promise<string[]> {
+ const out = new Set<string>();
+ for (const input of inputs) {
+ const abs = path.resolve(input);
+ const st = await stat(abs).catch(() => null);
+ if (!st) throw new Error(`No such file or directory: ${input}`);
+ if (st.isDirectory()) {
+ for (const e of await readdir(abs, { withFileTypes: true })) {
+ if (e.isFile() && PAGE_EXT.test(e.name)) out.add(path.join(abs, e.name));
+ }
+ } else {
+ out.add(abs);
+ }
+ }
+ return [...out].sort();
+}
+
+export async function importForumPages(opts: ImportForumPagesOptions): Promise<ImportForumPagesResult> {
+ const log = (line: string) => opts.onLog?.(line);
+ const fail = (error: string): ImportForumPagesResult => ({
+ ok: false,
+ error,
+ pages: [],
+ parsed: 0,
+ written: 0,
+ updated: 0,
+ unchanged: 0,
+ });
+ const config = await readChannelConfig(opts.paths, opts.slug);
+ if (!config) return fail(`No such channel: ${opts.slug}`);
+ if (!isSocialChannel(config) || config.platform !== "xenforo") {
+ return fail(`${opts.slug} is not a forum-thread channel (platform "xenforo").`);
+ }
+ const thread = parseXenforoThreadUrl(config.url ?? "");
+ if (!thread) return fail(`${opts.slug}'s URL is not a XenForo thread URL.`);
+
+ let files: string[];
+ try {
+ files = await listSavedPages(opts.inputs);
+ } catch (err) {
+ return fail((err as Error).message);
+ }
+ if (files.length === 0) return fail("No saved pages (.html) found in what was given.");
+
+ const pages: ImportedPage[] = [];
+ const posts: Post[] = [];
+ for (const file of files) {
+ const name = path.basename(file);
+ let html: string;
+ try {
+ html = await readFile(file, "utf8");
+ } catch (err) {
+ pages.push({ file: name, posts: 0, skipped: `unreadable: ${(err as Error).message}` });
+ continue;
+ }
+ const block = classifyForumPage(html);
+ if (block) {
+ pages.push({ file: name, posts: 0, skipped: `not a thread page (${block.kind}: ${block.detail})` });
+ continue;
+ }
+ const parsed = parseXenforoThreadPage(html, { channelSlug: opts.slug, pageUrl: thread.base });
+ if (parsed.threadId && parsed.threadId !== thread.threadId) {
+ pages.push({
+ file: name,
+ page: parsed.page,
+ posts: 0,
+ skipped: `a page of another thread (${parsed.threadId}, not ${thread.threadId})`,
+ });
+ continue;
+ }
+ if (parsed.host && parsed.host !== thread.host) {
+ pages.push({ file: name, page: parsed.page, posts: 0, skipped: `a page of another forum (${parsed.host})` });
+ continue;
+ }
+ pages.push({ file: name, page: parsed.page, posts: parsed.posts.length });
+ posts.push(...parsed.posts);
+ log(`${name}: page ${parsed.page}/${parsed.lastPage}, ${parsed.posts.length} post(s).`);
+ }
+
+ if (opts.dryRun) {
+ log(`Dry run: ${posts.length} post(s) parsed from ${files.length} file(s); nothing written.`);
+ return { ok: true, pages, parsed: posts.length, written: 0, updated: 0, unchanged: 0 };
+ }
+ const channelRoot = path.join(opts.paths.channelsDir, opts.slug);
+ const res = await upsertPosts(channelRoot, posts);
+ log(
+ `Imported ${posts.length} post(s) from ${files.length} file(s): ${res.written} new, ` +
+ `${res.updated} updated, ${res.unchanged} unchanged.`,
+ );
+ return {
+ ok: true,
+ pages,
+ parsed: posts.length,
+ written: res.written,
+ updated: res.updated,
+ unchanged: res.unchanged,
+ };
+}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -523,6 +523,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
replayable: true,
queueKeyStrategy: "platform",
},
+ // Forum thread pages the operator saved from a browser, imported into a
+ // forum-thread channel (controller/importForumPages.ts). Offline — nothing is
+ // fetched — but on the PLATFORM queue so it never writes the channel's posts
+ // beside a fetch writing the same shards. Not drainable (one parse and one
+ // write); not replayable (the files named are the operator's, and may be
+ // gone).
+ "import-forum-pages": {
+ kind: "import-forum-pages",
+ label: "Import saved forum pages",
+ drainable: false,
+ replayable: false,
+ queueKeyStrategy: "platform",
+ },
// A screenshot and the attached media of specific archived posts (and the X
// Article a post links to), into the channel's `posts-media/`
// (controller/capturePosts.ts). On the PLATFORM queue, like fetch-posts:
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -118,6 +118,7 @@ export type ChannelConfig = {
sourceKind?: ChannelSourceKind;
postFetcher?: string;
socialHandle?: string;
+ postPagePauseSeconds?: number;
platform?: Platform;
name?: string;
url?: string;
@@ -157,6 +158,8 @@ export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = {
'Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed.',
socialHandle:
'Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account.',
+ postPagePauseSeconds:
+ "Social channels that are read page by page (a forum thread) only: the pause between two page loads, in seconds; each pause is jittered to 0.85–1.65× of it. Absent = the fetcher's own (12 s, so 10–20 s); floored at 5, capped at 600.",
platform: "The source platform (youtube, rumble, …). An unknown value is dropped.",
name: "Display name.",
url: "The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced.",
@@ -363,6 +366,8 @@ export const CHANNEL_CONFIG_COERCIONS: {
postFetcher: trimmedNonBlank,
socialHandle: (v) =>
typeof v === "string" && v.trim() ? v.trim().replace(/^@/, "") : undefined,
+ postPagePauseSeconds: (v) =>
+ isFiniteNumber(v) && v > 0 ? Math.min(Math.max(Math.floor(v), 5), 600) : undefined,
platform: (v) =>
typeof v === "string" && PLATFORM_VALUES.includes(v as Platform)
? (v as Platform)
diff --git a/common/lib/channelConfigSchema.ts b/common/lib/channelConfigSchema.ts
@@ -47,6 +47,7 @@ export const channelConfigObjectSchema = z.object({
sourceKind: field("sourceKind"),
postFetcher: field("postFetcher"),
socialHandle: field("socialHandle"),
+ postPagePauseSeconds: field("postPagePauseSeconds"),
platform: field("platform"),
name: field("name"),
url: field("url"),
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -94,7 +94,7 @@ const SHARD_SCHEME = {
// timestamp fragment.
const POST_SCHEME = {
description:
- "Social posts (X/Twitter, Bluesky) are archived as a PARALLEL corpus to " +
+ "Social posts (X/Twitter, Bluesky, forum threads) are archived as a PARALLEL corpus to " +
"video transcripts and are served as paginated JSON shards under the same " +
"scheme: (1) GET the channel's posts manifest; (2) look up the post id in " +
"its `slugToPage` map to get a page number N; (3) GET page-<NNNN>.json and " +
@@ -105,7 +105,11 @@ const POST_SCHEME = {
postPage:
"/posts/<slug>/page-<NNNN>.json -> array of { id, slug, channelSlug, author, " +
"authorName, createdAt, uploadDate, text, url, platform, threadId, replyTo, " +
- "quoted, repostOf, isReply, isRepost, links, mediaCount, engagement }",
+ "quoted, repostOf, isReply, isRepost, links, mediaCount, media, engagement, " +
+ "forum }; `forum` (platform \"xenforo\": one channel is one forum thread) " +
+ "carries { host, threadId, threadTitle, page, position, authorId, editedAt, " +
+ "quotes: [{ postId, author }] }, and quoted text in a forum post's `text` " +
+ "is marked with leading \"> \" lines",
ordering:
"newest first, by `createdAt` (ISO-8601, ms precision where the source provides it)",
dateFilter:
diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs
@@ -8,6 +8,16 @@
/** @typedef {import("./platform").Platform} Platform */
+// Forums known to run XenForo, by host. Any other XenForo forum is recognised
+// by its thread URL (THREAD_PATH_RE) instead — a host table is a convenience,
+// not the rule.
+export const XENFORO_HOSTS = ["kiwifarms.st", "kiwifarms.net"];
+
+// A XenForo thread path: /threads/<slug>.<id>/ or /threads/<id>/, optionally
+// under a prefix (/community/threads/…) or behind index.php? (non-friendly
+// URLs).
+export const XENFORO_THREAD_PATH_RE = /(?:^|\/|\?)threads\/(?:[^/?#]*\.)?(\d+)(?:\/|$)/;
+
/**
* @param {string | undefined | null} url
* @returns {Platform | null}
@@ -24,6 +34,11 @@ export function detectPlatform(url) {
if (host === "x.com" || host.endsWith(".x.com")) return "twitter";
if (host.endsWith("twitter.com")) return "twitter";
if (host === "bsky.app" || host.endsWith(".bsky.app")) return "bluesky";
+ if (XENFORO_HOSTS.some((h) => host === h || host.endsWith(`.${h}`))) {
+ return "xenforo";
+ }
+ const u = new URL(url);
+ if (XENFORO_THREAD_PATH_RE.test(u.pathname + u.search)) return "xenforo";
} catch {
/* fall through */
}
diff --git a/common/lib/forumPosts.test.ts b/common/lib/forumPosts.test.ts
@@ -0,0 +1,147 @@
+// Forum posts (platform "xenforo") in the model and the store: detection, the
+// stored-record validator, the conversation a forum post belongs to, and
+// upsertPosts — the import's write, which updates an edited post in place
+// (append-only, last record wins). SYNTHETIC posts only.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/forumPosts.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { detectPlatform, isSocialPlatform } from "./platform";
+import {
+ isPostPlatform,
+ parsePost,
+ postConversation,
+ postPermalink,
+ postPlatformLabel,
+ uploadDateFromCreatedAt,
+ type Post,
+} from "./posts";
+import { readAllPosts, readSeenPostIds, supersedesPost, upsertPosts } from "./posts-server";
+import { handleFromAccountUrl } from "../social/fetchers";
+
+function forumPost(id: string, position: number, extra: Partial<Post> = {}): Post {
+ const createdAt = new Date(Date.UTC(2026, 0, 1, 0, position)).toISOString();
+ return {
+ id,
+ slug: `teapots/${id}`,
+ channelSlug: "teapots",
+ author: `Member${position}`,
+ createdAt,
+ uploadDate: uploadDateFromCreatedAt(createdAt),
+ text: `Words of post ${position}.`,
+ url: `https://forum.example/posts/${id}/`,
+ platform: "xenforo",
+ isReply: position > 1,
+ isRepost: false,
+ links: [],
+ forum: { host: "forum.example", threadId: "4242", position, page: 1 },
+ ...extra,
+ };
+}
+
+test("detection: forum thread URLs are the xenforo platform, a social one", () => {
+ assert.equal(detectPlatform("https://forum.example/threads/a-title.4242/"), "xenforo");
+ assert.equal(detectPlatform("https://forum.example/threads/a-title.4242/page-9"), "xenforo");
+ assert.equal(detectPlatform("https://forum.example/index.php?threads/a-title.4242/"), "xenforo");
+ // Kiwi Farms is known by host (it runs XenForo).
+ assert.equal(detectPlatform("https://kiwifarms.st/members/someone.1/"), "xenforo");
+ assert.equal(detectPlatform("https://forum.example/members/someone.1/"), null);
+ assert.equal(detectPlatform("https://www.youtube.com/@x/threads"), "youtube");
+ assert.equal(isSocialPlatform("xenforo"), true);
+ assert.equal(isPostPlatform("xenforo"), true);
+ assert.equal(postPlatformLabel("xenforo"), "Forum");
+ assert.equal(handleFromAccountUrl("https://forum.example/threads/a-title.4242/page-3"), "a-title.4242");
+ assert.equal(postPermalink("xenforo", "", "77", "https://forum.example/"), "https://forum.example/posts/77/");
+});
+
+test("the validator keeps forum facts and media, and drops what is malformed", () => {
+ const p = forumPost("10", 3, {
+ media: [
+ { kind: "image", url: "https://images.example/a.jpg", name: "a.jpg" },
+ { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK", provider: "youtube" },
+ ],
+ mediaCount: 2,
+ forum: {
+ host: "forum.example",
+ threadId: "4242",
+ threadTitle: "Teapots",
+ position: 3,
+ page: 1,
+ authorId: "9",
+ editedAt: "2026-01-02T00:00:00.000Z",
+ quotes: [{ postId: "8", author: "Member1" }],
+ },
+ });
+ assert.deepEqual(parsePost(JSON.parse(JSON.stringify(p))), p);
+ const bad = parsePost({
+ ...p,
+ media: [{ kind: "sticker", url: "x" }, { kind: "image" }],
+ forum: { host: "forum.example" },
+ });
+ assert.equal(bad?.media, undefined);
+ assert.equal(bad?.forum, undefined);
+ // A forum post with no stored url gets its permalink from the forum host.
+ const { url: _url, ...noUrl } = p;
+ assert.equal(parsePost(noUrl)?.url, "https://forum.example/posts/10/");
+});
+
+test("a forum post's conversation is its quote graph, in thread order", () => {
+ const a = forumPost("1", 1);
+ const b = forumPost("2", 2, { forum: { host: "forum.example", threadId: "4242", position: 2, quotes: [{ postId: "1" }] } });
+ const c = forumPost("3", 3, { forum: { host: "forum.example", threadId: "4242", position: 3, quotes: [{ postId: "2" }] } });
+ const d = forumPost("4", 4, { forum: { host: "forum.example", threadId: "4242", position: 4, quotes: [{ postId: "3" }] } });
+ const unrelated = forumPost("5", 5);
+ const all = [unrelated, d, c, b, a];
+ assert.deepEqual(postConversation(c, all).map((p) => p.id), ["1", "2", "3", "4"]);
+ assert.deepEqual(postConversation(unrelated, all).map((p) => p.id), ["5"]);
+ assert.deepEqual(postConversation(c, all, 1).map((p) => p.id), ["2", "3", "4"]);
+});
+
+test("a non-forum post's conversation is still its reply thread", () => {
+ const base = { ...forumPost("x", 1), platform: "bluesky" as const, forum: undefined };
+ const root = { ...base, id: "r", threadId: "r", createdAt: "2026-01-01T00:00:00.000Z" };
+ const reply = { ...base, id: "s", threadId: "r", createdAt: "2026-01-01T00:01:00.000Z" };
+ const other = { ...base, id: "t", threadId: "t" };
+ assert.deepEqual(postConversation(reply, [other, reply, root]).map((p) => p.id), ["r", "s"]);
+});
+
+test("supersedesPost: a later edit wins, an older one never does", () => {
+ const stored = forumPost("1", 1, { forum: { host: "h", threadId: "1", editedAt: "2026-02-01T00:00:00.000Z" } });
+ const newer = { ...stored, text: "changed", forum: { ...stored.forum!, editedAt: "2026-03-01T00:00:00.000Z" } };
+ const older = { ...stored, text: "older words", forum: { ...stored.forum!, editedAt: "2026-01-15T00:00:00.000Z" } };
+ const unedited = { ...stored, text: "never edited", forum: { host: "h", threadId: "1" } };
+ assert.equal(supersedesPost(stored, { ...stored }), false);
+ assert.equal(supersedesPost(stored, newer), true);
+ assert.equal(supersedesPost(stored, older), false);
+ assert.equal(supersedesPost(stored, unedited), false);
+ const plain = forumPost("2", 2);
+ assert.equal(supersedesPost(plain, { ...plain, text: "re-read" }), true);
+});
+
+test("upsertPosts: new posts appended, an edit updates the post, a repeat changes nothing", async () => {
+ const root = await mkdtemp(path.join(tmpdir(), "forum-upsert-"));
+ const first = await upsertPosts(root, [forumPost("1", 1), forumPost("2", 2)]);
+ assert.deepEqual([first.written, first.updated, first.unchanged], [2, 0, 0]);
+
+ const edited = forumPost("2", 2, {
+ text: "Words of post 2, edited.",
+ forum: { host: "forum.example", threadId: "4242", position: 2, page: 1, editedAt: "2026-01-05T00:00:00.000Z" },
+ });
+ const second = await upsertPosts(root, [forumPost("1", 1), edited, forumPost("3", 3)]);
+ assert.deepEqual([second.written, second.updated, second.unchanged], [1, 1, 1]);
+
+ const all = await readAllPosts(root);
+ assert.deepEqual(all.map((p) => p.id), ["3", "2", "1"]);
+ assert.equal(all.find((p) => p.id === "2")?.text, "Words of post 2, edited.");
+ assert.equal((await readSeenPostIds(root)).size, 3);
+ // The archive lists each id once: an update is not a new post.
+ const archive = await readFile(path.join(root, "posts-archive"), "utf8");
+ assert.equal(archive.trim().split("\n").length, 3);
+
+ const third = await upsertPosts(root, [edited]);
+ assert.deepEqual([third.written, third.updated, third.unchanged], [0, 0, 1]);
+});
diff --git a/common/lib/platform.ts b/common/lib/platform.ts
@@ -9,7 +9,8 @@ export type Platform =
| "twitch"
| "kick"
| "twitter"
- | "bluesky";
+ | "bluesky"
+ | "xenforo";
export const PLATFORM_VALUES: ReadonlyArray<Platform> = [
"youtube",
@@ -19,6 +20,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [
"kick",
"twitter",
"bluesky",
+ "xenforo",
];
// The subset that carries social posts rather than videos. A channel on one of
@@ -27,10 +29,11 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [
export const SOCIAL_PLATFORM_VALUES: ReadonlyArray<Platform> = [
"twitter",
"bluesky",
+ "xenforo",
];
export function isSocialPlatform(platform: Platform | null | undefined): boolean {
- return platform === "twitter" || platform === "bluesky";
+ return platform === "twitter" || platform === "bluesky" || platform === "xenforo";
}
// Host → platform. Lives in `detectPlatform.mjs` (plain JS so umtool's `.mjs`
@@ -54,6 +57,10 @@ export function defaultWebpageUrl(platform: Platform, id: string): string {
// every archived post carries its own canonical `url` (see postPermalink).
if (platform === "twitter") return `https://x.com/i/status/${id}`;
if (platform === "bluesky") return `https://bsky.app/profile/${id}`;
+ // A forum post's URL needs its forum's host, which an id alone does not
+ // carry; every archived forum post has its own `url`. A thread URL passed as
+ // the id is returned as it is.
+ if (platform === "xenforo") return /^https?:\/\//.test(id) ? id : "";
return `https://www.youtube.com/watch?v=${id}`;
}
diff --git a/common/lib/posts-server.ts b/common/lib/posts-server.ts
@@ -129,6 +129,96 @@ export async function writePosts(
};
}
+export type UpsertPostsResult = WritePostsResult & {
+ // Archived posts whose stored record was superseded by a newer one.
+ updated: number;
+ // Archived posts that arrived again unchanged (or older than what is
+ // stored).
+ unchanged: number;
+};
+
+// What makes two records of one post differ, for an update.
+function postContentKey(p: Post): string {
+ return JSON.stringify([
+ p.text,
+ p.author,
+ p.links,
+ p.media ?? null,
+ p.forum?.editedAt ?? null,
+ p.forum?.position ?? null,
+ p.forum?.threadTitle ?? null,
+ ]);
+}
+
+// Is `incoming` a newer record of the same post than `stored`? A later edit
+// time wins; an older one never replaces a newer one; with no edit times to
+// compare (or equal ones), different content is taken as the later read.
+export function supersedesPost(stored: Post, incoming: Post): boolean {
+ if (postContentKey(stored) === postContentKey(incoming)) return false;
+ const a = stored.forum?.editedAt;
+ const b = incoming.forum?.editedAt;
+ if (a && b && a !== b) return b > a;
+ if (a && !b) return false;
+ return true;
+}
+
+// writePosts, plus UPDATES: a post already archived whose incoming record is
+// newer (supersedesPost — an edit made since it was archived) is appended
+// again to its month shard. The JSONL stays append-only; every reader takes the
+// LAST record of an id (readAllPosts, the index build), so the newer one wins.
+// For sources that re-read posts they already hold (a forum page saved twice).
+export async function upsertPosts(
+ channelRoot: string,
+ posts: ReadonlyArray<Post>,
+): Promise<UpsertPostsResult> {
+ // One record per id from the batch: the newest of them.
+ const batch = new Map<string, Post>();
+ for (const p of posts) {
+ const prev = batch.get(p.id);
+ if (!prev || supersedesPost(prev, p)) batch.set(p.id, p);
+ }
+ const seen = await readSeenPostIds(channelRoot);
+ const fresh = [...batch.values()].filter((p) => !seen.has(p.id));
+ const known = [...batch.values()].filter((p) => seen.has(p.id));
+ const written = await writePosts(channelRoot, fresh);
+ let updated = 0;
+ let unchanged = 0;
+ const shards = new Set(written.shards);
+ if (known.length > 0) {
+ const stored = new Map<string, Post>();
+ for (const p of await readAllPosts(channelRoot)) stored.set(p.id, p);
+ const byShard = new Map<string, Post[]>();
+ for (const p of known) {
+ const old = stored.get(p.id);
+ if (old && !supersedesPost(old, p)) {
+ unchanged++;
+ continue;
+ }
+ // Keep the shard the post already lives in: its createdAt does not
+ // change with an edit.
+ const shard = monthShardFromCreatedAt(old?.createdAt ?? p.createdAt);
+ const bucket = byShard.get(shard);
+ const rec = old ? { ...p, createdAt: old.createdAt, uploadDate: old.uploadDate } : p;
+ if (bucket) bucket.push(rec);
+ else byShard.set(shard, [rec]);
+ updated++;
+ }
+ if (byShard.size > 0) await mkdir(channelPostsDir(channelRoot), { recursive: true });
+ for (const [shard, shardPosts] of byShard) {
+ const body = shardPosts.map((p) => JSON.stringify(p)).join("\n") + "\n";
+ await appendFile(shardPath(channelRoot, shard), body, "utf8");
+ shards.add(shard);
+ }
+ }
+ return {
+ written: written.written,
+ skipped: written.skipped,
+ shards: [...shards].sort(),
+ updated,
+ unchanged,
+ };
+}
+
// List the channel's month shards, oldest first.
export async function listPostShards(
channelRoot: string,
diff --git a/common/lib/posts.ts b/common/lib/posts.ts
@@ -10,17 +10,71 @@
import { pageFileName } from "./manifest";
-export type PostPlatform = "twitter" | "bluesky";
+// "xenforo" is a forum THREAD read as a posts source: the channel is one
+// thread, each forum post a Post. It is named for the forum software, not a
+// host — which forum a thread is on is its URL (and `Post.forum.host`).
+export type PostPlatform = "twitter" | "bluesky" | "xenforo";
export const POST_PLATFORM_VALUES: ReadonlyArray<PostPlatform> = [
"twitter",
"bluesky",
+ "xenforo",
];
export function isPostPlatform(v: unknown): v is PostPlatform {
- return v === "twitter" || v === "bluesky";
+ return v === "twitter" || v === "bluesky" || v === "xenforo";
}
+// The platform's name as a reader sees it.
+export function postPlatformLabel(platform: PostPlatform | string | undefined): string {
+ if (platform === "twitter") return "X";
+ if (platform === "bluesky") return "Bluesky";
+ if (platform === "xenforo") return "Forum";
+ return "Post";
+}
+
+// Media a post carries, as LINKS (the archive keeps the URLs, not the bytes —
+// a post capture downloads them). Recorded by the forum parser; the X and
+// Bluesky normalizers count media instead (`mediaCount`).
+export type PostMediaKind = "image" | "video" | "attachment" | "embed" | "link-card";
+
+export const POST_MEDIA_KINDS: ReadonlyArray<PostMediaKind> = [
+ "image",
+ "video",
+ "attachment",
+ "embed",
+ "link-card",
+];
+
+export type PostMedia = {
+ kind: PostMediaKind;
+ url: string;
+ // A file name (an attachment's), or an image's alt text.
+ name?: string;
+ // An embed's provider ("youtube", "twitter", …), as the forum named it.
+ provider?: string;
+};
+
+// What a forum post carries beyond the common shape (platform "xenforo").
+export type ForumPostInfo = {
+ host: string;
+ // The thread's numeric id, and its title and canonical URL when read.
+ threadId: string;
+ threadTitle?: string;
+ threadUrl?: string;
+ // The thread page the post was read on, and its position in the thread
+ // (the "#N" the forum shows).
+ page?: number;
+ position?: number;
+ // The author's numeric member id.
+ authorId?: string;
+ // When the forum says the post was last edited (ISO-8601).
+ editedAt?: string;
+ // The posts this one quotes, in order: the quoted post's id and author
+ // when the quote names them.
+ quotes?: { postId?: string; author?: string; authorId?: string }[];
+};
+
// A pointer to another post, which may or may not itself be archived. Used for
// reply/quote/repost edges so thread structure survives even when the
// referenced post is outside the archived account.
@@ -68,6 +122,10 @@ export type Post = {
isRepost: boolean;
links: string[]; // expanded outbound urls
mediaCount?: number; // counted, not archived (v1)
+ // The media's URLs, where the source gives them (forum posts).
+ media?: PostMedia[];
+ // Forum-specific facts (platform "xenforo").
+ forum?: ForumPostInfo;
engagement?: PostEngagement;
// Merged in at index time from the channel's availability sidecar (the same
// shape videos use: the stored record is the source of truth, the flag on
@@ -209,15 +267,21 @@ export function parsePostSlug(
// Canonical permalink for a post. Bluesky needs the handle (its URLs are
// /profile/<handle>/post/<rkey>); X only needs the numeric id but includes the
-// handle for readability.
+// handle for readability. A forum post needs its forum's origin (`base`,
+// "https://forum.example"): XenForo's /posts/<id>/ resolves to the post in its
+// thread wherever the thread has moved.
export function postPermalink(
platform: PostPlatform,
author: string,
id: string,
+ base?: string,
): string {
if (platform === "bluesky") {
return `https://bsky.app/profile/${author}/post/${id}`;
}
+ if (platform === "xenforo") {
+ return `${(base ?? "").replace(/\/+$/, "")}/posts/${id}/`;
+ }
return `https://x.com/${author || "i"}/status/${id}`;
}
@@ -251,7 +315,12 @@ export function parsePost(raw: unknown): Post | null {
url:
typeof r.url === "string" && r.url
? r.url
- : postPermalink(r.platform, author, r.id),
+ : postPermalink(
+ r.platform,
+ author,
+ r.id,
+ r.platform === "xenforo" ? forumBaseOf(r.forum) : undefined,
+ ),
platform: r.platform,
isReply: r.isReply === true,
isRepost: r.isRepost === true,
@@ -279,9 +348,77 @@ export function parsePost(raw: unknown): Post | null {
}
const engagement = parseEngagement(r.engagement);
if (engagement) post.engagement = engagement;
+ const media = parseMediaList(r.media);
+ if (media) post.media = media;
+ const forum = parseForumInfo(r.forum);
+ if (forum) post.forum = forum;
return post;
}
+function forumBaseOf(raw: unknown): string | undefined {
+ const host = (raw as { host?: unknown } | null)?.host;
+ return typeof host === "string" && host ? `https://${host}` : undefined;
+}
+
+function parseMediaList(raw: unknown): PostMedia[] | null {
+ if (!Array.isArray(raw)) return null;
+ const out: PostMedia[] = [];
+ for (const item of raw) {
+ if (!item || typeof item !== "object") continue;
+ const m = item as Record<string, unknown>;
+ if (typeof m.url !== "string" || !m.url) continue;
+ if (!(POST_MEDIA_KINDS as string[]).includes(m.kind as string)) continue;
+ const media: PostMedia = { kind: m.kind as PostMediaKind, url: m.url };
+ if (typeof m.name === "string" && m.name) media.name = m.name;
+ if (typeof m.provider === "string" && m.provider) media.provider = m.provider;
+ out.push(media);
+ }
+ return out.length > 0 ? out : null;
+}
+
+const posInt = (v: unknown): number | undefined =>
+ typeof v === "number" && Number.isFinite(v) && v >= 1 ? Math.floor(v) : undefined;
+const nonBlank = (v: unknown): string | undefined =>
+ typeof v === "string" && v ? v : undefined;
+
+function parseForumInfo(raw: unknown): ForumPostInfo | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ const host = nonBlank(r.host);
+ const threadId = nonBlank(r.threadId);
+ if (!host || !threadId) return null;
+ const out: ForumPostInfo = { host, threadId };
+ const threadTitle = nonBlank(r.threadTitle);
+ if (threadTitle) out.threadTitle = threadTitle;
+ const threadUrl = nonBlank(r.threadUrl);
+ if (threadUrl) out.threadUrl = threadUrl;
+ const page = posInt(r.page);
+ if (page) out.page = page;
+ const position = posInt(r.position);
+ if (position) out.position = position;
+ const authorId = nonBlank(r.authorId);
+ if (authorId) out.authorId = authorId;
+ const editedAt = nonBlank(r.editedAt);
+ if (editedAt) out.editedAt = editedAt;
+ if (Array.isArray(r.quotes)) {
+ const quotes: NonNullable<ForumPostInfo["quotes"]> = [];
+ for (const q of r.quotes) {
+ if (!q || typeof q !== "object") continue;
+ const qq = q as Record<string, unknown>;
+ const one: { postId?: string; author?: string; authorId?: string } = {};
+ const postId = nonBlank(qq.postId);
+ if (postId) one.postId = postId;
+ const author = nonBlank(qq.author);
+ if (author) one.author = author;
+ const authorId = nonBlank(qq.authorId);
+ if (authorId) one.authorId = authorId;
+ quotes.push(one);
+ }
+ if (quotes.length > 0) out.quotes = quotes;
+ }
+ return out;
+}
+
function parsePostRef(raw: unknown): PostRef | null {
if (!raw || typeof raw !== "object") return null;
const r = raw as Record<string, unknown>;
@@ -316,6 +453,56 @@ export function comparePostsNewestFirst(a: Post, b: Post): number {
return a.id < b.id ? 1 : a.id > b.id ? -1 : 0;
}
+// The CONVERSATION around a post, oldest first: the "more context" unit the
+// viewer's thread view and the MCP get_thread tool show.
+//
+// For X and Bluesky that is the reply thread (threadId). A forum thread is one
+// channel of possibly thousands of posts, so a forum post's conversation is
+// the quote graph instead: the posts it quotes (and theirs, to `depth`), and
+// the posts that quote it — read in thread order.
+export function postConversation(
+ post: Post,
+ candidates: ReadonlyArray<Post>,
+ depth = 3,
+): Post[] {
+ if (post.platform !== "xenforo") {
+ const threadId = post.threadId || post.id;
+ const thread = candidates.filter((p) => (p.threadId || p.id) === threadId);
+ if (!thread.some((p) => p.id === post.id)) thread.push(post);
+ return thread.sort((a, b) => -comparePostsNewestFirst(a, b));
+ }
+ const byId = new Map<string, Post>();
+ for (const p of candidates) byId.set(p.id, p);
+ byId.set(post.id, post);
+ const keep = new Map<string, Post>([[post.id, post]]);
+ let frontier: Post[] = [post];
+ for (let d = 0; d < depth && frontier.length > 0; d++) {
+ const next: Post[] = [];
+ for (const p of frontier) {
+ for (const q of p.forum?.quotes ?? []) {
+ const hit = q.postId ? byId.get(q.postId) : undefined;
+ if (hit && !keep.has(hit.id)) {
+ keep.set(hit.id, hit);
+ next.push(hit);
+ }
+ }
+ }
+ frontier = next;
+ }
+ for (const p of candidates) {
+ if (p.forum?.quotes?.some((q) => q.postId === post.id)) keep.set(p.id, p);
+ }
+ return [...keep.values()].sort(compareForumOrder);
+}
+
+// Thread order for forum posts: position when both have one, else time.
+function compareForumOrder(a: Post, b: Post): number {
+ const pa = a.forum?.position;
+ const pb = b.forum?.position;
+ if (pa && pb && pa !== pb) return pa - pb;
+ return -comparePostsNewestFirst(a, b);
+}
+
// Group a flat post list into threads keyed by threadId (falling back to the
// post's own id for a root/standalone post). Used by the viewer's thread
// context and the MCP get_thread tool.
diff --git a/common/lib/report/convert-server.ts b/common/lib/report/convert-server.ts
@@ -112,7 +112,7 @@ export function diskPostOf(channelsDir: string): PostOf {
const post = (await p).get(id);
if (!post) return null;
const rec: PostRecord = {
- platform: post.platform === "bluesky" ? "bluesky" : "x",
+ platform: post.platform === "bluesky" ? "bluesky" : post.platform === "xenforo" ? "forum" : "x",
url: post.url,
createdAt: post.createdAt,
author: post.author,
diff --git a/common/lib/report/convertManifest.ts b/common/lib/report/convertManifest.ts
@@ -355,7 +355,9 @@ export async function manifestToReport(manifest: unknown, opts: ManifestConvertO
// What a post record says that a post citation does not (the CLI reads it off
// the channel's archive).
export type PostRecord = {
- platform: "x" | "bluesky";
+ // "forum": a forum-thread post (platform "xenforo"), which a video manifest
+ // does not carry yet — it is left out with a warning, never relabelled.
+ platform: "x" | "bluesky" | "forum";
url: string;
createdAt?: string;
author?: string;
@@ -489,6 +491,10 @@ export async function reportToManifest(raw: unknown, opts: ManifestOptions = {})
if (postsDone.has(cid)) return;
postsDone.add(cid);
const rec = opts.postOf ? await opts.postOf(c.channel, c.id) : null;
+ if (rec?.platform === "forum") {
+ warnings.push(`${cid}: post ${c.channel}/${c.id} is a forum post, which a video manifest does not carry — left out of posts`);
+ return;
+ }
const platform = rec?.platform ?? (/^\d+$/.test(c.id) ? "x" : null);
const url = rec?.url ?? (platform === "x" ? `https://x.com/i/status/${c.id}` : null);
const date = c.date ?? rec?.createdAt;
diff --git a/common/social/__fixtures__/xenforoPages.ts b/common/social/__fixtures__/xenforoPages.ts
@@ -0,0 +1,222 @@
+// SYNTHETIC XenForo 2 thread pages for the forum-thread tests. The forum, the
+// thread, the members and every word are invented; the markup follows the
+// shape XenForo 2 serves (article.message, .bbWrapper, the attribution header,
+// the page nav), and one variant is shaped the way a browser's "Save page as"
+// rewrites a page.
+
+export const ORIGIN = "https://forum.example";
+export const THREAD_PATH = "/threads/the-teapot-collectors-thread.4242/";
+export const THREAD_URL = `${ORIGIN}${THREAD_PATH}`;
+
+export type FakePost = {
+ id: number;
+ author: string;
+ userId: number;
+ // Epoch seconds.
+ ts: number;
+ position: number;
+ body: string;
+ editedTs?: number;
+ attachments?: { href: string; name: string }[];
+};
+
+export function message(p: FakePost, opts: { savedAssets?: boolean; article?: boolean } = {}): string {
+ const iso = new Date(p.ts * 1000).toISOString().replace(".000Z", "+0000");
+ const avatar = opts.savedAssets
+ ? `./The Teapot Collectors Thread_files/${p.userId}.jpg`
+ : `/data/avatars/m/0/${p.userId}.jpg?1700000000`;
+ const attachments = p.attachments?.length
+ ? `<section class="message-attachments">
+ <h4 class="block-textHeader">Attachments</h4>
+ <ul class="attachmentList">${p.attachments
+ .map(
+ (a) => `<li class="file file--linked">
+ <a class="u-anchorTarget" id="attachment-${p.id}"></a>
+ <a class="file-preview js-lbImage" href="${a.href}" target="_blank">
+ <img src="${opts.savedAssets ? "./The Teapot Collectors Thread_files/thumb.jpg" : "/data/attachments/thumb.jpg"}" alt="${a.name}" width="200" loading="lazy">
+ </a>
+ <div class="file-content"><div class="file-info">
+ <span class="file-name" title="${a.name}">${a.name}</span>
+ <div class="file-meta">50 KB · Views: 3</div>
+ </div></div>
+ </li>`,
+ )
+ .join("")}</ul>
+ </section>`
+ : "";
+ const edited = p.editedTs
+ ? `<div class="message-lastEdit">Last edited: <time class="u-dt" dir="auto" datetime="${new Date(p.editedTs * 1000).toISOString().replace(".000Z", "+0000")}" data-timestamp="${p.editedTs}" data-date-string="x" data-time-string="y" title="x">x</time></div>`
+ : "";
+ return `
+<article class="message ${opts.article ? "message--article" : "message--post"} js-post js-inlineModContainer " data-author="${p.author}" data-content="post-${p.id}" id="js-post-${p.id}" itemscope itemtype="https://schema.org/Comment" itemid="${ORIGIN}/posts/${p.id}/">
+ <span class="u-anchorTarget" id="post-${p.id}"></span>
+ <div class="message-inner">
+ <div class="message-cell message-cell--user">
+ <section class="message-user" itemprop="author" itemscope itemtype="https://schema.org/Person" itemid="${ORIGIN}/members/${p.author.toLowerCase()}.${p.userId}/">
+ <div class="message-avatar"><div class="message-avatar-wrapper">
+ <a href="/members/${p.author.toLowerCase()}.${p.userId}/" class="avatar avatar--m" data-user-id="${p.userId}" data-xf-init="member-tooltip">
+ <img src="${avatar}" alt="${p.author}" class="avatar-u${p.userId}-m" width="96" height="96" loading="lazy" itemprop="image">
+ </a>
+ </div></div>
+ <div class="message-userDetails">
+ <h4 class="message-name"><a href="/members/${p.author.toLowerCase()}.${p.userId}/" class="username " dir="auto" data-user-id="${p.userId}" data-xf-init="member-tooltip"><span itemprop="name">${p.author}</span></a></h4>
+ <h5 class="userTitle message-userTitle" dir="auto" itemprop="jobTitle">Collector</h5>
+ </div>
+ </section>
+ </div>
+ <div class="message-cell message-cell--main">
+ <div class="message-main js-quickEditTarget">
+ <header class="message-attribution message-attribution--split">
+ <ul class="message-attribution-main listInline ">
+ <li class="u-concealed">
+ <a href="${THREAD_PATH}post-${p.id}" rel="nofollow" itemprop="url">
+ <time class="u-dt" dir="auto" datetime="${iso}" data-timestamp="${p.ts}" data-date-string="d" data-time-string="t" title="t" itemprop="datePublished">d</time>
+ </a>
+ </li>
+ </ul>
+ <ul class="message-attribution-opposite message-attribution-opposite--list ">
+ <li><a href="${THREAD_PATH}post-${p.id}" class="message-attribution-gadget" data-xf-init="share-tooltip" rel="nofollow"><i class="fa--xf fal fa-share-alt"><svg><use href="#share-alt"></use></svg></i></a></li>
+ <li><a href="${THREAD_PATH}post-${p.id}" class="message-attribution-gadget bookmarkLink" rel="nofollow"><span class="js-bookmarkText u-srOnly">Add bookmark</span></a></li>
+ ${opts.article ? "" : `<li>
+ <a href="${THREAD_PATH}post-${p.id}" rel="nofollow">
+ #${p.position.toLocaleString("en-US")}
+ </a>
+ </li>`}
+ </ul>
+ </header>
+ <div class="message-content js-messageContent">
+ <div class="message-userContent lbContainer js-lbContainer " data-lb-id="post-${p.id}">
+ <article class="message-body js-selectToQuote">
+ <div itemprop="text">
+ <div class="bbWrapper">${p.body}</div>
+ </div>
+ <div class="js-selectToQuoteEnd"> </div>
+ </article>
+ ${attachments}
+ </div>
+ ${edited}
+ </div>
+ <footer class="message-footer">
+ <div class="message-actionBar actionBar"><div class="actionBar-set actionBar-set--external">
+ <a href="${THREAD_PATH}post-${p.id}/react" class="reaction actionBar-action">Like</a>
+ </div></div>
+ <div class="reactionsBar js-reactionsList is-active"><a class="reactionsBar-link" href="/posts/${p.id}/reactions">Someone and 2 others</a></div>
+ </footer>
+ </div>
+ </div>
+ </div>
+</article>`;
+}
+
+export function pageNav(page: number, last: number): string {
+ if (last <= 1) return "";
+ const items: string[] = [];
+ for (const n of [1, page - 1, page, page + 1, last]) {
+ if (n < 1 || n > last || items.some((i) => i.includes(`>${n}<`))) continue;
+ items.push(
+ `<li class="pageNav-page ${n === page ? "pageNav-page--current " : ""}"><a href="${THREAD_PATH}${n === 1 ? "" : `page-${n}`}">${n}</a></li>`,
+ );
+ }
+ return `<nav class="pageNavWrapper pageNavWrapper--mixed">
+ <div class="pageNav">
+ <ul class="pageNav-main">${items.join("")}</ul>
+ </div>
+ <div class="pageNavSimple">
+ <a class="pageNavSimple-el pageNavSimple-el--current" data-xf-init="tooltip" title="Go to page">${page} of ${last}</a>
+ </div>
+ <div class="js-pageJumpPage-wrap"><input type="number" class="input input--number js-pageJumpPage" value="${page}" min="1" max="${last}" step="1" required="required" data-menu-autofocus="true"></div>
+ </nav>`;
+}
+
+export function threadPage(opts: {
+ page: number;
+ last: number;
+ posts: FakePost[];
+ saved?: boolean;
+ title?: string;
+ // An article thread: this post (its first) shown atop the page.
+ article?: FakePost;
+}): string {
+ const canonical = `${ORIGIN}${THREAD_PATH}${opts.page > 1 ? `page-${opts.page}` : ""}`;
+ const savedComment = opts.saved ? `<!-- saved from url=(0068)${canonical} -->\n` : "";
+ // A browser's save keeps the canonical link; the variant drops it so the
+ // saved-from comment is what names the page.
+ const canonicalLink = opts.saved ? "" : `<link rel="canonical" href="${canonical}" />`;
+ return `<!DOCTYPE html>
+${savedComment}<html id="XF" lang="en-US" dir="LTR" data-xf="2.2" data-app="public" data-template="thread_view" data-container-key="node-7" data-content-key="thread-4242" data-logged-in="false">
+<head>
+ <meta charset="utf-8" />
+ <title>${opts.title ?? "The Teapot Collectors Thread"}${opts.page > 1 ? ` | Page ${opts.page}` : ""} | Example Forum</title>
+ ${canonicalLink}
+ <meta property="og:title" content="${opts.title ?? "The Teapot Collectors Thread"}" />
+ <script>window.XF = {}; if (1 < 2) { var x = "</div>"; }</script>
+ <style>.message { color: red; }</style>
+</head>
+<body data-template="thread_view">
+ <div class="p-body-header">
+ <div class="p-title "><h1 class="p-title-value"><span class="label label--blue" dir="auto">Hobby</span><span class="label-append"> </span>${opts.title ?? "The Teapot Collectors Thread"}</h1></div>
+ </div>
+ ${pageNav(opts.page, opts.last)}
+ <div class="block block--messages" data-xf-init="lightbox select-to-quote">
+ <div class="block-body js-replyNewMessageContainer">
+ ${opts.article ? message(opts.article, { savedAssets: opts.saved, article: true }) : ""}
+ ${opts.posts.map((p) => message(p, { savedAssets: opts.saved })).join("\n")}
+ </div>
+ </div>
+ ${pageNav(opts.page, opts.last)}
+</body>
+</html>`;
+}
+
+// A browser check page of the KiwiFlare kind: no posts, a proof-of-work script
+// and its marker.
+export const CHALLENGE_PAGE = `<!DOCTYPE html>
+<html><head><title>Checking your browser</title>
+<script src="/.sssg/api/challenge.js"></script></head>
+<body><div id="sssg-root">Please wait while your browser solves a proof-of-work challenge.</div>
+<noscript>This check needs JavaScript.</noscript></body></html>`;
+
+export const CAPTCHA_PAGE = `<!DOCTYPE html>
+<html><head><title>One more step</title></head>
+<body><div class="h-captcha" data-sitekey="00000000-0000-0000-0000-000000000000"></div></body></html>`;
+
+export const LOGIN_PAGE = `<!DOCTYPE html>
+<html data-template="login"><head><title>Log in | Example Forum</title></head>
+<body><div class="blockMessage">You must be logged-in to do that.</div></body></html>`;
+
+export const NOT_FOUND_PAGE = `<!DOCTYPE html>
+<html data-template="error"><head><title>Oops! We ran into some problems. | Example Forum</title></head>
+<body><div class="blockMessage">The requested thread could not be found.</div></body></html>`;
+
+// A body exercising everything the body reader handles.
+export const RICH_BODY = `
+<blockquote data-attributes="member: 7" data-quote="Marigold" data-source="post: 1001" class="bbCodeBlock bbCodeBlock--expandable bbCodeBlock--quote js-expandWatch">
+ <div class="bbCodeBlock-title"><a href="/goto/post?id=1001" class="bbCodeBlock-sourceJump" rel="nofollow" data-xf-click="attribution" data-content-selector="#post-1001">Marigold said:</a></div>
+ <div class="bbCodeBlock-content">
+ <div class="bbCodeBlock-expandContent js-expandContent ">
+ The blue one is a reproduction.<br />
+ Look at the glaze.
+ </div>
+ <div class="bbCodeBlock-expandLink js-expandLink"><a role="button" tabindex="0">Click to expand...</a></div>
+ </div>
+</blockquote>
+I disagree, <a href="/members/marigold.7/" class="username" data-xf-init="member-tooltip" data-user-id="7" data-username="@Marigold">@Marigold</a>.<br />
+<br />
+Here is the catalogue: <a href="https://archive.example/AbCd1" target="_blank" class="link link--external" rel="nofollow ugc noopener">https://archive.example/AbCd1</a><br />
+And the <a href="https://museum.example/teapots?id=9" target="_blank" class="link link--external" rel="nofollow ugc noopener">museum page</a> <img src="/styles/default/smilies/smile.png" class="smilie" alt=":)" title="Smile :)" data-shortname=":)" />
+<div class="bbImageWrapper js-lbImage" title="teapot.jpg" data-src="https://images.example/teapot.jpg" data-lb-sidebar-href="" data-lb-caption-extra-html="" data-single-image="1">
+ <img src="https://forum.example/proxy.php?image=https%3A%2F%2Fimages.example%2Fteapot.jpg" data-url="https://images.example/teapot.jpg" class="bbImage" data-zoom-target="1" style="" alt="teapot.jpg" title="" width="" height="" loading="lazy" />
+</div>
+<span data-s9e-mediaembed="youtube" style="display:inline-block;width:100%;max-width:640px"><span style="display:block;overflow:hidden;position:relative;padding-bottom:56.25%"><iframe allowfullscreen="" loading="lazy" scrolling="no" style="border:0;height:100%;left:0;position:absolute;width:100%" src="https://www.youtube.com/embed/AbCdEfGhIjK"></iframe></span></span>
+<iframe data-s9e-mediaembed="twitter" allow="autoplay *" allowfullscreen="" loading="lazy" scrolling="no" src="https://s9e.github.io/iframe/2/twitter.min.html#1234567890123456789" style="background:url(https://abs.twimg.com/favicons/favicon.ico) no-repeat 50% 50%;border:0;height:350px;max-width:550px;width:100%"></iframe>
+<div class="bbCodeSpoiler">
+ <button type="button" class="bbCodeSpoiler-button button--longText button" data-xf-click="toggle" data-xf-init="tooltip" title="Click to reveal or hide spoiler"><span class="button-text"><span>Spoiler: <span class="bbCodeSpoiler-button-title">the ending</span></span></span></button>
+ <div class="bbCodeSpoiler-content"><div class="bbCodeBlock bbCodeBlock--spoiler"><div class="bbCodeBlock-content">The lid was glued on.</div></div></div>
+</div>
+<div class="bbMediaWrapper"><div class="bbMediaWrapper-inner"><video controls="" data-xf-init="video-init"><source src="/data/video/12/12345-abc.mp4" /><div class="bbMediaWrapper-fallback">Your browser is not able to display this video.</div></video></div></div>
+<div class="bbCodeBlock bbCodeBlock--unfurl js-unfurl fauxBlockLink" data-unfurl="true" data-result-id="77" data-url="https://news.example/teapot-auction" data-host="news.example" data-pending="false">
+ <div class="contentRow"><div class="contentRow-main"><h3 class="contentRow-header js-unfurl-title"><a href="https://news.example/teapot-auction" class="link link--external fauxBlockLink-blockLink" target="_blank" rel="nofollow ugc noopener" data-proxy-href="">Teapot sells for a fortune</a></h3><div class="contentRow-snippet js-unfurl-desc">An auction report.</div></div></div>
+</div>
+<ul><li>first point</li><li>second point</li></ul>
+<a href="https://forum.example/attachments/receipt-png.555/" target="_blank"><img src="https://forum.example/data/attachments/0/555-receipt.jpg" class="bbImage" alt="receipt.png" /></a>
+`;
diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts
@@ -54,6 +54,12 @@ export type PostFetchInput = {
// Soft cap on how many posts to return in one run. Undefined = no cap
// beyond the watermark/seen-id stop conditions.
limit?: number;
+ // Cap on how many PAGES one run reads, for a fetcher that walks pages (a
+ // forum thread: "the latest N pages"). Ignored by the others.
+ pages?: number;
+ // The pause between two page loads, for a paced page walker (a forum
+ // thread). Default: the fetcher's own.
+ pagePauseMs?: number;
// Stop once the walk reaches posts that are already archived. False for a
// --full repair, which re-walks everything on purpose. Undefined: the
// fetcher's own default.
@@ -181,8 +187,10 @@ export type PostAvailabilityInput = {
// (postCapture.ts owns the layout).
export type PostCaptureInput = Pick<
PostFetchInput,
- "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog"
+ "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog" | "pagePauseMs"
> & {
+ // The channel's URL (a forum capture needs the forum's origin).
+ accountUrl?: string;
ids: ReadonlyArray<string>;
handle: string;
outDir: string;
@@ -197,7 +205,17 @@ export type PostCaptureInput = Pick<
// What the posts archive holds for each id — its text and expanded links —
// for the links a capture follows (an X Article's). Ids absent are read
// from the page alone.
- archived?: ReadonlyMap<string, { text: string; links?: ReadonlyArray<string> }>;
+ archived?: ReadonlyMap<
+ string,
+ {
+ text: string;
+ links?: ReadonlyArray<string>;
+ // The post's own URL and media, where the archive has them (forum
+ // posts: the capture opens the URL and downloads the media).
+ url?: string;
+ media?: ReadonlyArray<{ kind: string; url: string; name?: string }>;
+ }
+ >;
// A soft stop: no new post is started once it fires, and the one in hand
// finishes (`signal` cancels outright).
drain?: AbortSignal;
@@ -341,6 +359,11 @@ export function handleFromAccountUrl(input: string): string | null {
}
const segments = url.pathname.split("/").filter(Boolean);
if (segments.length === 0) return null;
+ // A forum thread: its "<slug>.<id>" key is the channel's handle.
+ const threadIdx = segments.findIndex((s) => s === "threads");
+ if (threadIdx >= 0 && /(?:^|\.)\d+$/.test(segments[threadIdx + 1] ?? "")) {
+ return decodeURIComponent(segments[threadIdx + 1]);
+ }
// bsky.app/profile/<handle-or-did>
if (segments[0] === "profile" && segments[1]) {
return decodeURIComponent(segments[1]).replace(/^@/, "");
diff --git a/common/social/forumSession.ts b/common/social/forumSession.ts
@@ -0,0 +1,406 @@
+// THE FORUM BROWSER: one persistent Chromium profile per forum host, opened
+// headless by the thread fetcher and the post capture, and HEADED by Connect.
+//
+// Why a persistent profile: a forum behind a browser check (Kiwi Farms'
+// KiwiFlare — a JavaScript proof of work — or a "just a moment" page) sets a
+// clearance cookie once the check passes. A real browser clears it by itself;
+// keeping the profile keeps the cookie, so every later page and run reuses it
+// instead of meeting the check again.
+//
+// Connect is the operator's way through anything the headless browser cannot
+// pass (a captcha, a login a members-only thread needs): a headed window on the
+// same profile, at the thread, which the operator clears by hand and closes.
+// Mirrors the X session broker (xSessionBroker.ts): the same browser choice
+// and launch options (xBrowser.ts — the operator's own Chromium when one is
+// installed, without the automation bar), the same record of which browser
+// wrote the profile, and the same rule that a window opens only on an explicit
+// operator action, never from a fetch.
+//
+// The user agent is the browser's own. Headless Chromium names itself
+// "HeadlessChrome" there; that one word is replaced with "Chrome" so the
+// headless profile presents as the browser it is. Nothing else is disguised,
+// and a captcha is never answered by code.
+//
+// Layout, per host: transcripts/.forum-session/<host>/profile/ (the profile),
+// transcripts/.forum-session/<host>/session.json (the last Connect).
+
+import path from "node:path";
+import { mkdir, rm, stat } from "node:fs/promises";
+import { execa } from "execa";
+import type { Paths } from "../lib/paths";
+import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server";
+import {
+ importPlaywright,
+ type BrowserContextLike,
+ type ChromiumLike,
+ type PageLike,
+} from "./playwrightRuntime";
+import {
+ buildXBrowserLaunchOptions,
+ describeXBrowser,
+ findXBrowser,
+ readXBrowserRecord,
+ recordedXBrowser,
+ writeXBrowserRecord,
+ type XBrowserChoice,
+} from "./xBrowser";
+import { classifyForumPage, type ForumBlock } from "./xenforoParse";
+
+// A host as a directory name: lowercased, nothing but [a-z0-9.-].
+export function forumHostKey(host: string): string {
+ const key = host.trim().toLowerCase().replace(/[^a-z0-9.-]/g, "_");
+ if (!key || key.startsWith(".")) throw new Error(`"${host}" is not a forum host`);
+ return key;
+}
+
+export function forumSessionDir(paths: Pick<Paths, "transcriptsDir">, host: string): string {
+ return path.join(paths.transcriptsDir, ".forum-session", forumHostKey(host));
+}
+
+export function forumProfileDir(paths: Pick<Paths, "transcriptsDir">, host: string): string {
+ return path.join(forumSessionDir(paths, host), "profile");
+}
+
+export type ForumSessionRecord = {
+ host: string;
+ connectedAt: string;
+ // The window's last look at the forum before it closed: did a thread page
+ // show (the check was cleared), and how many cookies the host had set.
+ cleared: boolean;
+ cookies: number;
+ browser?: string;
+};
+
+export type ForumSessionStatus = {
+ host: string;
+ hasProfile: boolean;
+ lastConnect?: ForumSessionRecord;
+};
+
+export async function readForumSessionStatus(
+ paths: Pick<Paths, "transcriptsDir">,
+ host: string,
+): Promise<ForumSessionStatus> {
+ const dir = forumSessionDir(paths, host);
+ const hasProfile = await stat(path.join(dir, "profile"))
+ .then(() => true)
+ .catch(() => false);
+ const read = await readJsonFile(path.join(dir, "session.json"));
+ const rec = read.ok ? (read.value as ForumSessionRecord | null) : null;
+ return {
+ host,
+ hasProfile,
+ ...(rec && typeof rec.connectedAt === "string" ? { lastConnect: rec } : {}),
+ };
+}
+
+export async function clearForumSession(paths: Pick<Paths, "transcriptsDir">, host: string): Promise<void> {
+ await rm(forumSessionDir(paths, host), { recursive: true, force: true });
+}
+
+function firstLine(err: unknown): string {
+ return ((err as Error)?.message ?? String(err)).split("\n")[0];
+}
+
+// --- opening the profile ---------------------------------------------------------
+
+const BUNDLED: XBrowserChoice = { kind: "bundled" };
+
+// The headless user agent with its one tell removed, learned once per process.
+let normalUserAgent: string | undefined;
+
+export function presentableUserAgent(ua: string): string {
+ return ua.replace(/HeadlessChrome/g, "Chrome");
+}
+
+async function launchHeadless(
+ chromium: ChromiumLike,
+ profileDir: string,
+ browser: XBrowserChoice,
+ userAgent?: string,
+): Promise<BrowserContextLike> {
+ return chromium.launchPersistentContext(profileDir, {
+ ...buildXBrowserLaunchOptions({ browser, headless: true }),
+ viewport: { width: 1280, height: 900 },
+ ...(userAgent ? { userAgent } : {}),
+ });
+}
+
+// The host's profile, headless. The bundled build first; the browser the
+// profile records (the one Connect opened) when the bundled build cannot open
+// it — as launchXProfile does for X.
+export async function openForumProfile(
+ profileDir: string,
+ opts: { onLog?: (line: string) => void; chromium?: ChromiumLike } = {},
+): Promise<BrowserContextLike> {
+ const log = opts.onLog ?? (() => {});
+ await mkdir(profileDir, { recursive: true });
+ const chromium = opts.chromium ?? (await importPlaywright()).chromium;
+ let browser: XBrowserChoice = BUNDLED;
+ let context: BrowserContextLike;
+ try {
+ context = await launchHeadless(chromium, profileDir, browser, normalUserAgent);
+ } catch (err) {
+ const record = await readXBrowserRecord(profileDir);
+ const recorded = recordedXBrowser(record);
+ if (!recorded) throw err;
+ log(
+ `Playwright's bundled Chromium could not open the forum profile (${firstLine(err)}); ` +
+ `using ${describeXBrowser(recorded, record?.version)}, headless.`,
+ );
+ browser = recorded;
+ context = await launchHeadless(chromium, profileDir, browser, normalUserAgent);
+ }
+ if (normalUserAgent) return context;
+ // First launch in this process: read the browser's own agent; if it names
+ // itself headless, reopen once with that word replaced.
+ try {
+ const page = context.pages()[0] ?? (await context.newPage());
+ const ua = String(await page.evaluate("navigator.userAgent"));
+ const fixed = presentableUserAgent(ua);
+ normalUserAgent = fixed;
+ if (fixed !== ua) {
+ await context.close().catch(() => {});
+ context = await launchHeadless(chromium, profileDir, browser, fixed);
+ }
+ } catch (err) {
+ log(`Could not read the browser's user agent (${firstLine(err)}); keeping its default.`);
+ }
+ return context;
+}
+
+// --- loading one page through a browser check ---------------------------------------
+
+export type ForumPageLoad = {
+ html: string;
+ // The response status, when the page came straight back (no check waited
+ // out — after a check the first response's status no longer describes the
+ // page).
+ status?: number;
+ url: string;
+ // How long a browser check took to clear, when there was one.
+ challengeMs?: number;
+};
+
+export type ForumPageLoader = {
+ load(url: string, signal: AbortSignal): Promise<ForumPageLoad>;
+ close(): Promise<void>;
+};
+
+export const CHALLENGE_TIMEOUT_MS = 60_000;
+const CHALLENGE_POLL_MS = 2_000;
+const OUTER_HTML = "document.documentElement ? document.documentElement.outerHTML : ''";
+
+type ResponseLike = { status?: () => number } | null | undefined;
+
+// Go to `url` in `page` and come back with what it shows. A browser check
+// (classifyForumPage → "challenge") is WAITED OUT, up to `timeoutMs`: the
+// check's own script runs, sets its cookie and reloads, and the poll reads
+// the page that follows. Nothing is clicked or solved. What the page shows at
+// the end — the thread, or the check still standing — is the caller's to
+// judge.
+export async function loadForumPage(
+ page: PageLike,
+ url: string,
+ opts: {
+ signal: AbortSignal;
+ onLog?: (line: string) => void;
+ timeoutMs?: number;
+ pollMs?: number;
+ now?: () => number;
+ },
+): Promise<ForumPageLoad> {
+ const now = opts.now ?? Date.now;
+ const timeoutMs = opts.timeoutMs ?? CHALLENGE_TIMEOUT_MS;
+ const pollMs = opts.pollMs ?? CHALLENGE_POLL_MS;
+ let status: number | undefined;
+ try {
+ const res = (await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 })) as ResponseLike;
+ status = typeof res?.status === "function" ? res.status() : undefined;
+ } catch (err) {
+ // A navigation the check's own reload interrupted still leaves a page to
+ // read; anything else is a load failure.
+ if (!/interrupted|ERR_ABORTED|frame was detached/i.test(firstLine(err))) {
+ throw new Error(`Could not load ${url}: ${firstLine(err)}`);
+ }
+ }
+ const read = async (): Promise<string | null> => {
+ try {
+ return String(await page.evaluate(OUTER_HTML));
+ } catch {
+ return null; // mid-navigation
+ }
+ };
+ let html = (await read()) ?? "";
+ let block: ForumBlock | null = classifyForumPage(html, status);
+ if (block?.kind !== "challenge") {
+ return { html, ...(status !== undefined ? { status } : {}), url: await currentUrl(page, url) };
+ }
+ const started = now();
+ opts.onLog?.(
+ `The forum served a browser check (${block.detail}); waiting up to ${Math.round(timeoutMs / 1000)} s for it to clear.`,
+ );
+ while (now() - started < timeoutMs) {
+ if (opts.signal.aborted) break;
+ await page.waitForTimeout(pollMs);
+ const next = await read();
+ if (next === null) continue;
+ html = next;
+ block = classifyForumPage(html);
+ if (block?.kind !== "challenge") break;
+ }
+ const challengeMs = now() - started;
+ if (block?.kind === "challenge") {
+ opts.onLog?.(`The browser check had not cleared after ${Math.round(challengeMs / 1000)} s.`);
+ } else {
+ opts.onLog?.(`The browser check cleared in ${Math.round(challengeMs / 1000)} s.`);
+ }
+ return { html, url: await currentUrl(page, url), challengeMs };
+}
+
+async function currentUrl(page: PageLike, fallback: string): Promise<string> {
+ try {
+ const href = String(await page.evaluate("location.href"));
+ return href && href !== "about:blank" ? href : fallback;
+ } catch {
+ return fallback;
+ }
+}
+
+// A loader over the host's headless profile, opened on the first load and
+// kept for the run (one browser, one page, every page in turn).
+export function browserForumLoader(
+ profileDir: string,
+ opts: { onLog?: (line: string) => void; timeoutMs?: number; chromium?: ChromiumLike } = {},
+): ForumPageLoader & { page(): Promise<PageLike> } {
+ let context: BrowserContextLike | undefined;
+ let page: PageLike | undefined;
+ const ensure = async (): Promise<PageLike> => {
+ if (!context) {
+ context = await openForumProfile(profileDir, { onLog: opts.onLog, chromium: opts.chromium });
+ opts.onLog?.("Opened the forum browser profile (headless).");
+ }
+ page ??= context.pages()[0] ?? (await context.newPage());
+ return page;
+ };
+ return {
+ page: ensure,
+ async load(url, signal) {
+ const p = await ensure();
+ return loadForumPage(p, url, { signal, onLog: opts.onLog, timeoutMs: opts.timeoutMs });
+ },
+ async close() {
+ await context?.close().catch(() => {});
+ context = undefined;
+ page = undefined;
+ },
+ };
+}
+
+// --- Connect: the headed window ----------------------------------------------------------
+
+async function browserVersion(b: XBrowserChoice): Promise<string | undefined> {
+ if (b.kind !== "system") return undefined;
+ try {
+ const res = await execa(b.executablePath, ["--version"], { reject: false, timeout: 5_000 });
+ const line = `${res.stdout ?? ""}`.trim().split("\n")[0];
+ return res.exitCode === 0 && line ? line : undefined;
+ } catch {
+ return undefined;
+ }
+}
+
+type CookieLike = { domain?: string };
+
+function cookiesForHost(cookies: ReadonlyArray<CookieLike>, host: string): number {
+ const h = host.toLowerCase();
+ return cookies.filter((c) => {
+ const d = (c.domain ?? "").replace(/^\./, "").toLowerCase();
+ return d === h || h.endsWith(`.${d}`) || d.endsWith(`.${h}`);
+ }).length;
+}
+
+// Open a HEADED browser on the host's profile at `url` (the thread), for the
+// operator to clear the check, answer a captcha or log in. Resolves when the
+// window is closed, with the window's last look at the forum. The window opens
+// on the display of the machine running the editor (as Connect X does).
+export async function connectForumSession(
+ paths: Pick<Paths, "transcriptsDir">,
+ url: string,
+ opts: { onLog?: (line: string) => void; timeoutMs?: number; env?: Record<string, string | undefined> } = {},
+): Promise<ForumSessionRecord> {
+ const log = opts.onLog ?? (() => {});
+ const host = new URL(url).hostname.toLowerCase();
+ const profileDir = forumProfileDir(paths, host);
+ await mkdir(profileDir, { recursive: true });
+ const browser = findXBrowser({ env: opts.env });
+ const version = await browserVersion(browser);
+ const label = describeXBrowser(browser, version);
+ const { chromium } = await importPlaywright();
+ log(`Opening ${label} at ${host}. Clear the check (or log in), wait for the thread to show, then close the window.`);
+ let context: BrowserContextLike;
+ try {
+ context = await chromium.launchPersistentContext(
+ profileDir,
+ buildXBrowserLaunchOptions({ browser, headless: false, sandbox: true }),
+ );
+ } catch (err) {
+ log(`The sandboxed launch failed (${firstLine(err)}); opening without the sandbox.`);
+ context = await chromium.launchPersistentContext(
+ profileDir,
+ buildXBrowserLaunchOptions({ browser, headless: false, sandbox: false }),
+ );
+ }
+ await writeXBrowserRecord(profileDir, browser, version).catch((err) => {
+ log(`Could not record the browser in the profile: ${firstLine(err)}`);
+ });
+
+ let cleared = false;
+ let cookies = 0;
+ const look = async () => {
+ try {
+ const page = context.pages()[0];
+ if (page) {
+ const html = String(await page.evaluate(OUTER_HTML));
+ cleared = classifyForumPage(html) === null;
+ }
+ cookies = cookiesForHost((await context.cookies()) as CookieLike[], host);
+ } catch {
+ /* closing, or mid-navigation */
+ }
+ };
+ try {
+ const page = context.pages()[0] ?? (await context.newPage());
+ await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 }).catch((err: unknown) => {
+ log(`The first load did not finish (${firstLine(err)}); the window stays open.`);
+ });
+ await new Promise<void>((resolve) => {
+ let settled = false;
+ const timer = setInterval(() => void look(), 2_000);
+ const finish = () => {
+ if (settled) return;
+ settled = true;
+ clearInterval(timer);
+ resolve();
+ };
+ context.on("close", finish);
+ if (opts.timeoutMs) setTimeout(finish, opts.timeoutMs);
+ });
+ } finally {
+ await context.close().catch(() => {});
+ }
+ const record: ForumSessionRecord = {
+ host,
+ connectedAt: new Date().toISOString(),
+ cleared,
+ cookies,
+ browser: label,
+ };
+ await writeJsonAtomic(path.join(forumSessionDir(paths, host), "session.json"), record, { mkdir: true });
+ log(
+ cleared
+ ? `The thread showed before the window closed; ${cookies} cookie(s) for ${host} kept in the profile.`
+ : `The window closed without a thread page showing; ${cookies} cookie(s) for ${host} in the profile.`,
+ );
+ return record;
+}
diff --git a/common/social/htmlReader.ts b/common/social/htmlReader.ts
@@ -0,0 +1,121 @@
+// A small HTML reader, pure: no DOM, no dependency. Shared by the X Article
+// reader (xArticle.ts) and the XenForo thread parser (xenforoParse.ts).
+//
+// For HTML a browser serialised (outerHTML): attributes are double-quoted, void
+// elements are unclosed, text escapes only & < > and nbsp. It tolerates more
+// (single quotes, bare values, stray end tags), but it is not a general parser.
+
+export type HtmlElement = {
+ tag: string;
+ attrs: Record<string, string>;
+ children: HtmlNode[];
+};
+export type HtmlNode = HtmlElement | string;
+
+const VOID_TAGS = new Set([
+ "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta",
+ "param", "source", "track", "wbr",
+]);
+const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]);
+
+const NAMED_ENTITIES: Record<string, string> = {
+ amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0",
+ hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“",
+ rdquo: "”", copy: "©", reg: "®", trade: "™",
+};
+
+export function decodeEntities(s: string): string {
+ return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => {
+ if (name[0] === "#") {
+ const code =
+ name[1] === "x" || name[1] === "X"
+ ? parseInt(name.slice(2), 16)
+ : parseInt(name.slice(1), 10);
+ return Number.isFinite(code) && code > 0 && code <= 0x10ffff
+ ? String.fromCodePoint(code)
+ : whole;
+ }
+ return NAMED_ENTITIES[name.toLowerCase()] ?? whole;
+ });
+}
+
+const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y;
+const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y;
+
+// The document's top-level nodes, under a synthetic root element.
+export function parseHtml(html: string): HtmlElement {
+ const root: HtmlElement = { tag: "#root", attrs: {}, children: [] };
+ // Lowered once: a raw-text element's end is searched for in it (a page holds
+ // dozens of scripts, and lowering the whole page per script is quadratic).
+ let lower: string | undefined;
+ const stack: HtmlElement[] = [root];
+ const top = () => stack[stack.length - 1];
+ let i = 0;
+ while (i < html.length) {
+ const lt = html.indexOf("<", i);
+ if (lt < 0) {
+ top().children.push(decodeEntities(html.slice(i)));
+ break;
+ }
+ if (lt > i) top().children.push(decodeEntities(html.slice(i, lt)));
+ i = lt;
+ if (html.startsWith("<!--", i)) {
+ const end = html.indexOf("-->", i + 4);
+ i = end < 0 ? html.length : end + 3;
+ continue;
+ }
+ if (html[i + 1] === "!" || html[i + 1] === "?") {
+ const end = html.indexOf(">", i);
+ i = end < 0 ? html.length : end + 1;
+ continue;
+ }
+ if (html[i + 1] === "/") {
+ const end = html.indexOf(">", i);
+ const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase();
+ i = end < 0 ? html.length : end + 1;
+ // Close up to the matching open element; a stray end tag is ignored.
+ for (let k = stack.length - 1; k > 0; k--) {
+ if (stack[k].tag === name) {
+ stack.length = k;
+ break;
+ }
+ }
+ continue;
+ }
+ TAG_NAME_RE.lastIndex = i + 1;
+ const nameMatch = TAG_NAME_RE.exec(html);
+ if (!nameMatch) {
+ // A "<" that opens no tag is text.
+ top().children.push("<");
+ i++;
+ continue;
+ }
+ const tag = nameMatch[0].toLowerCase();
+ let j = TAG_NAME_RE.lastIndex;
+ const attrs: Record<string, string> = {};
+ for (;;) {
+ ATTR_RE.lastIndex = j;
+ const a = ATTR_RE.exec(html);
+ if (!a || a[0].length === 0) break;
+ attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? "");
+ j = ATTR_RE.lastIndex;
+ }
+ const close = html.indexOf(">", j);
+ const selfClosing = close > 0 && html[close - 1] === "/";
+ i = close < 0 ? html.length : close + 1;
+ const el: HtmlElement = { tag, attrs, children: [] };
+ top().children.push(el);
+ if (RAW_TEXT_TAGS.has(tag)) {
+ lower ??= html.toLowerCase();
+ const end = lower.indexOf(`</${tag}`, i);
+ const stop = end < 0 ? html.length : end;
+ if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop)));
+ const gt = end < 0 ? -1 : html.indexOf(">", end);
+ i = gt < 0 ? html.length : gt + 1;
+ continue;
+ }
+ if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el);
+ }
+ return root;
+}
+
diff --git a/common/social/xArticle.ts b/common/social/xArticle.ts
@@ -19,6 +19,13 @@
// and an article whose body marker is missing is read by the fallback — the
// root's text split at block elements — and says so (`extraction`).
+import {
+ decodeEntities,
+ parseHtml,
+ type HtmlElement,
+ type HtmlNode,
+} from "./htmlReader";
+
// --- the link ------------------------------------------------------------------
export type XArticleLink = {
@@ -60,121 +67,13 @@ export function xArticleLinkFromArchive(
return findXArticleLink([archived.text, ...(archived.links ?? [])]);
}
-// --- a small HTML reader -------------------------------------------------------
+// --- the HTML reader -----------------------------------------------------------
//
-// For HTML a browser serialised (outerHTML): attributes are double-quoted, void
-// elements are unclosed, text escapes only & < > and nbsp. It tolerates more
-// (single quotes, bare values, stray end tags), but it is not a general parser.
-
-export type HtmlElement = {
- tag: string;
- attrs: Record<string, string>;
- children: HtmlNode[];
-};
-export type HtmlNode = HtmlElement | string;
-
-const VOID_TAGS = new Set([
- "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta",
- "param", "source", "track", "wbr",
-]);
-const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]);
-
-const NAMED_ENTITIES: Record<string, string> = {
- amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0",
- hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“",
- rdquo: "”", copy: "©", reg: "®", trade: "™",
-};
+// Shared with the forum-thread parser (xenforoParse.ts): it lives in
+// htmlReader.ts and is re-exported here for the callers that import it from
+// this module.
-export function decodeEntities(s: string): string {
- return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => {
- if (name[0] === "#") {
- const code =
- name[1] === "x" || name[1] === "X"
- ? parseInt(name.slice(2), 16)
- : parseInt(name.slice(1), 10);
- return Number.isFinite(code) && code > 0 && code <= 0x10ffff
- ? String.fromCodePoint(code)
- : whole;
- }
- return NAMED_ENTITIES[name.toLowerCase()] ?? whole;
- });
-}
-
-const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y;
-const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y;
-
-// The document's top-level nodes, under a synthetic root element.
-export function parseHtml(html: string): HtmlElement {
- const root: HtmlElement = { tag: "#root", attrs: {}, children: [] };
- const stack: HtmlElement[] = [root];
- const top = () => stack[stack.length - 1];
- let i = 0;
- while (i < html.length) {
- const lt = html.indexOf("<", i);
- if (lt < 0) {
- top().children.push(decodeEntities(html.slice(i)));
- break;
- }
- if (lt > i) top().children.push(decodeEntities(html.slice(i, lt)));
- i = lt;
- if (html.startsWith("<!--", i)) {
- const end = html.indexOf("-->", i + 4);
- i = end < 0 ? html.length : end + 3;
- continue;
- }
- if (html[i + 1] === "!" || html[i + 1] === "?") {
- const end = html.indexOf(">", i);
- i = end < 0 ? html.length : end + 1;
- continue;
- }
- if (html[i + 1] === "/") {
- const end = html.indexOf(">", i);
- const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase();
- i = end < 0 ? html.length : end + 1;
- // Close up to the matching open element; a stray end tag is ignored.
- for (let k = stack.length - 1; k > 0; k--) {
- if (stack[k].tag === name) {
- stack.length = k;
- break;
- }
- }
- continue;
- }
- TAG_NAME_RE.lastIndex = i + 1;
- const nameMatch = TAG_NAME_RE.exec(html);
- if (!nameMatch) {
- // A "<" that opens no tag is text.
- top().children.push("<");
- i++;
- continue;
- }
- const tag = nameMatch[0].toLowerCase();
- let j = TAG_NAME_RE.lastIndex;
- const attrs: Record<string, string> = {};
- for (;;) {
- ATTR_RE.lastIndex = j;
- const a = ATTR_RE.exec(html);
- if (!a || a[0].length === 0) break;
- attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? "");
- j = ATTR_RE.lastIndex;
- }
- const close = html.indexOf(">", j);
- const selfClosing = close > 0 && html[close - 1] === "/";
- i = close < 0 ? html.length : close + 1;
- const el: HtmlElement = { tag, attrs, children: [] };
- top().children.push(el);
- if (RAW_TEXT_TAGS.has(tag)) {
- const end = html.toLowerCase().indexOf(`</${tag}`, i);
- const stop = end < 0 ? html.length : end;
- if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop)));
- const gt = end < 0 ? -1 : html.indexOf(">", end);
- i = gt < 0 ? html.length : gt + 1;
- continue;
- }
- if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el);
- }
- return root;
-}
+export { decodeEntities, parseHtml, type HtmlElement, type HtmlNode };
// --- reading the article ------------------------------------------------------
diff --git a/common/social/xenforoFetcher.test.ts b/common/social/xenforoFetcher.test.ts
@@ -0,0 +1,523 @@
+// The forum-thread walk, capture and page loader over fakes: no browser, no
+// network. SYNTHETIC pages (__fixtures__/xenforoPages.ts).
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xenforoFetcher.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, readdir } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import type { Post } from "../lib/posts";
+import type { PostFetchInput } from "./fetchers";
+import { loadForumPage, presentableUserAgent, type ForumPageLoader } from "./forumSession";
+import type { PageLike } from "./playwrightRuntime";
+import {
+ captureForumPosts,
+ DEFAULT_PAGE_PAUSE_MS,
+ downloadableMedia,
+ forumBlockFailure,
+ jitteredPause,
+ LAST_PAGE_PROBE,
+ MIN_PAGE_PAUSE_MS,
+ walkXenforoThread,
+ xenforoFetcher,
+} from "./xenforoFetcher";
+import {
+ CHALLENGE_PAGE,
+ CAPTCHA_PAGE,
+ NOT_FOUND_PAGE,
+ ORIGIN,
+ THREAD_URL,
+ threadPage,
+ type FakePost,
+} from "./__fixtures__/xenforoPages";
+
+const T0 = 1_760_000_000;
+const LAST = 5;
+const PER_PAGE = 3;
+
+// Post k (1-based position) of the thread, on page ceil(k / 3).
+function fakePost(position: number): FakePost {
+ return {
+ id: 9000 + position,
+ author: `Member${position % 4}`,
+ userId: 50 + (position % 4),
+ ts: T0 + position * 600,
+ position,
+ body: `Words of post ${position}.`,
+ };
+}
+
+function pageHtml(page: number): string {
+ const posts: FakePost[] = [];
+ for (let k = (page - 1) * PER_PAGE + 1; k <= page * PER_PAGE; k++) posts.push(fakePost(k));
+ return threadPage({ page, last: LAST, posts });
+}
+
+const pageOf = (url: string): number => {
+ const m = /page-(\d+)$/.exec(url);
+ if (!m) return 1;
+ return Math.min(Number(m[1]), LAST); // past the end → the last page
+};
+
+type FakeLoader = ForumPageLoader & { urls: string[]; closed: number };
+
+function fakeLoader(
+ override: (url: string, page: number) => { html: string; status?: number } | undefined = () => undefined,
+): FakeLoader {
+ const loader: FakeLoader = {
+ urls: [],
+ closed: 0,
+ async load(url) {
+ loader.urls.push(url);
+ const page = pageOf(url);
+ const o = override(url, page);
+ return {
+ html: o?.html ?? pageHtml(page),
+ ...(o?.status ? { status: o.status } : { status: 200 }),
+ url: `${THREAD_URL}${page > 1 ? `page-${page}` : ""}`,
+ };
+ },
+ async close() {
+ loader.closed++;
+ },
+ };
+ return loader;
+}
+
+function input(over: Partial<PostFetchInput> = {}): PostFetchInput {
+ return {
+ accountUrl: THREAD_URL,
+ handle: "the-teapot-collectors-thread.4242",
+ channelSlug: "teapots",
+ seenIds: new Set(),
+ signal: new AbortController().signal,
+ ...over,
+ };
+}
+
+function deps(loader: ForumPageLoader) {
+ const sleeps: number[] = [];
+ return {
+ sleeps,
+ deps: {
+ loader,
+ pauseMs: () => 11_000,
+ sleep: async (ms: number) => {
+ sleeps.push(ms);
+ },
+ },
+ };
+}
+
+const ids = (posts: Post[]) => posts.map((p) => Number(p.id) - 9000);
+
+test("first run: starts at the last page, walks newest first to page 1, paced between loads", async () => {
+ const loader = fakeLoader();
+ const { deps: d, sleeps } = deps(loader);
+ const log: string[] = [];
+ const res = await walkXenforoThread(input({ onLog: (l) => log.push(l) }), d);
+ assert.equal(res.complete, true);
+ assert.equal(res.cursor, undefined);
+ assert.equal(loader.urls[0], `${THREAD_URL}page-${LAST_PAGE_PROBE}`);
+ assert.deepEqual(loader.urls.slice(1), [4, 3, 2].map((n) => `${THREAD_URL}page-${n}`).concat([THREAD_URL]));
+ // Every page's posts, each page in thread order, newest page first.
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12, 7, 8, 9, 4, 5, 6, 1, 2, 3]);
+ // One pause before every load but the first, none in parallel.
+ assert.deepEqual(sleeps, [11_000, 11_000, 11_000, 11_000]);
+ assert.equal(loader.closed, 1);
+ assert.match(log[0], /^Page 5\/5: 3 post\(s\) parsed, 3 new\.$/);
+});
+
+test("a pages cap reads the latest N pages and leaves the cursor at the next one", async () => {
+ const loader = fakeLoader();
+ const res = await walkXenforoThread(input({ pages: 2 }), deps(loader).deps);
+ assert.equal(res.complete, false);
+ assert.equal(res.cursor, "3");
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12]);
+ assert.equal(loader.urls.length, 2);
+});
+
+test("resume: a stored cursor continues backwards, and an archived post there does not end it", async () => {
+ const loader = fakeLoader();
+ // Post 9 (page 3) shifted in from a page already read: archived.
+ const res = await walkXenforoThread(input({ cursor: "3", seenIds: new Set(["9009", "9013"]) }), deps(loader).deps);
+ assert.equal(res.complete, true);
+ assert.deepEqual(loader.urls, [`${THREAD_URL}page-3`, `${THREAD_URL}page-2`, THREAD_URL]);
+ assert.deepEqual(ids(res.posts), [7, 8, 4, 5, 6, 1, 2, 3]);
+});
+
+test("incremental: stops on the page holding an already-archived post, keeping its new ones", async () => {
+ const loader = fakeLoader();
+ const seen = new Set(["9001", "9002", "9003", "9004", "9005", "9006", "9007", "9008", "9009", "9010"]);
+ const res = await walkXenforoThread(input({ seenIds: seen }), deps(loader).deps);
+ assert.equal(res.complete, true);
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]);
+ assert.equal(loader.urls.length, 2);
+});
+
+test("a full re-walk (stopAtKnown false) does not stop at archived posts", async () => {
+ const loader = fakeLoader();
+ const res = await walkXenforoThread(
+ input({ seenIds: new Set(["9013", "9010"]), stopAtKnown: false }),
+ deps(loader).deps,
+ );
+ assert.equal(res.complete, true);
+ assert.equal(loader.urls.length, 5);
+ assert.equal(res.posts.length, 13);
+});
+
+test("the watermark stops the walk", async () => {
+ const since = new Date((T0 + 10 * 600) * 1000).toISOString(); // post 10's time
+ const res = await walkXenforoThread(input({ since }), deps(fakeLoader()).deps);
+ assert.equal(res.complete, true);
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]);
+});
+
+test("a post limit is checked after a whole page", async () => {
+ const res = await walkXenforoThread(input({ limit: 4 }), deps(fakeLoader()).deps);
+ assert.equal(res.complete, false);
+ assert.equal(res.cursor, "3");
+ assert.equal(res.posts.length, 6);
+});
+
+test("checkpoints: each page's posts are saved with the next page as the resume point", async () => {
+ const saved: { posts: number[]; cursor: string }[] = [];
+ const res = await walkXenforoThread(
+ input({
+ pages: 3,
+ onCheckpoint: async (cp) => {
+ saved.push({ posts: ids(cp.posts), cursor: cp.cursor });
+ },
+ }),
+ deps(fakeLoader()).deps,
+ );
+ assert.deepEqual(saved, [
+ { posts: [13, 14, 15], cursor: "4" },
+ { posts: [10, 11, 12], cursor: "3" },
+ { posts: [7, 8, 9], cursor: "2" },
+ ]);
+ assert.equal(res.posts.length, 0, "checkpointed posts are not handed back again");
+ assert.equal(res.cursor, "2");
+});
+
+test("drain stops between pages, keeping the next page; cancel stops too", async () => {
+ const drain = new AbortController();
+ const loader = fakeLoader((_url, page) => {
+ if (page === 4) drain.abort();
+ return undefined;
+ });
+ const res = await walkXenforoThread(input({ drain: drain.signal }), deps(loader).deps);
+ assert.equal(res.drained, true);
+ assert.equal(res.complete, false);
+ assert.equal(res.cursor, "3");
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12]);
+
+ const cancel = new AbortController();
+ const loader2 = fakeLoader((_url, page) => {
+ if (page === 5) cancel.abort();
+ return undefined;
+ });
+ const res2 = await walkXenforoThread(input({ signal: cancel.signal }), deps(loader2).deps);
+ assert.equal(res2.complete, false);
+ assert.equal(loader2.urls.length, 1);
+});
+
+test("a browser check that will not clear stops the run, typed, with the cursor kept and Connect named", async () => {
+ const loader = fakeLoader((_url, page) => (page === 4 ? { html: CHALLENGE_PAGE } : undefined));
+ const res = await walkXenforoThread(input(), deps(loader).deps);
+ assert.equal(res.complete, false);
+ assert.equal(res.needsCookies, true);
+ assert.equal(res.cursor, "4");
+ assert.match(res.error ?? "", /browser check/);
+ assert.match(res.error ?? "", /Connect forum session/);
+ assert.deepEqual(ids(res.posts), [13, 14, 15], "what was read before the check is kept");
+ assert.equal(loader.urls.length, 2, "never retried");
+});
+
+test("a captcha on the very first load keeps no cursor; a 429 is not a session problem", async () => {
+ const captcha = await walkXenforoThread(
+ input(),
+ deps(fakeLoader(() => ({ html: CAPTCHA_PAGE }))).deps,
+ );
+ assert.equal(captcha.needsCookies, true);
+ assert.equal(captcha.cursor, undefined);
+ assert.match(captcha.error ?? "", /captcha/);
+
+ const limited = await walkXenforoThread(
+ input({ cursor: "2" }),
+ deps(fakeLoader(() => ({ html: "<html><body>slow down</body></html>", status: 429 }))).deps,
+ );
+ assert.equal(limited.needsCookies, undefined);
+ assert.equal(limited.cursor, "2");
+ assert.match(limited.error ?? "", /HTTP 429/);
+});
+
+test("an article thread: its first post atop every page never stops the walk, and is taken once", async () => {
+ const article = fakePost(1);
+ const withArticle = (page: number) => {
+ const posts: FakePost[] = [];
+ for (let k = (page - 1) * PER_PAGE + 1; k <= page * PER_PAGE; k++) if (k > 1) posts.push(fakePost(k));
+ return { html: threadPage({ page, last: LAST, posts, article }) };
+ };
+ // Incremental: the starter is archived, and old — neither ends the walk on
+ // the last page; the archived post 10 on page 4 does.
+ const seen = new Set(["9001", "9010"]);
+ const since = new Date((T0 + 10 * 600) * 1000).toISOString();
+ const res = await walkXenforoThread(
+ input({ seenIds: seen, since }),
+ deps(fakeLoader((_url, page) => withArticle(page))).deps,
+ );
+ assert.equal(res.complete, true);
+ assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]);
+ // A first run takes the starter once, not once per page.
+ const all = await walkXenforoThread(input(), deps(fakeLoader((_url, page) => withArticle(page))).deps);
+ assert.equal(all.posts.filter((p) => p.id === "9001").length, 1);
+ assert.equal(all.posts.length, 15);
+});
+
+test("a page that parses to no posts stops rather than stepping past it", async () => {
+ const empty = threadPage({ page: 4, last: LAST, posts: [] });
+ const res = await walkXenforoThread(
+ input(),
+ deps(fakeLoader((_url, page) => (page === 4 ? { html: empty } : undefined))).deps,
+ );
+ assert.equal(res.complete, false);
+ assert.equal(res.cursor, "4");
+ assert.match(res.error ?? "", /no posts/);
+});
+
+test("a bad cursor and a URL that is not a thread are refused before any load", async () => {
+ const loader = fakeLoader();
+ assert.match((await walkXenforoThread(input({ cursor: "x" }), deps(loader).deps)).error ?? "", /not a page number/);
+ assert.match(
+ (await walkXenforoThread(input({ accountUrl: "https://forum.example/members/a.1/" }), deps(loader).deps)).error ?? "",
+ /not a XenForo thread URL/,
+ );
+ assert.equal(loader.urls.length, 0);
+});
+
+test("pacing: jittered around the base, never below the floor", () => {
+ assert.equal(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0), Math.round(DEFAULT_PAGE_PAUSE_MS * 0.85));
+ assert.ok(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0.999) <= 20_000);
+ assert.ok(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0) >= 10_000);
+ assert.equal(jitteredPause(100, () => 0), Math.round(MIN_PAGE_PAUSE_MS * 0.85));
+});
+
+test("block failures: which ones ask for Connect", () => {
+ assert.equal(forumBlockFailure({ kind: "challenge", detail: "x" }, "forum.example").needsCookies, true);
+ assert.equal(forumBlockFailure({ kind: "login", detail: "x" }, "forum.example").needsCookies, true);
+ assert.equal(forumBlockFailure({ kind: "blocked", detail: "HTTP 403" }, "forum.example").needsCookies, false);
+ assert.equal(forumBlockFailure({ kind: "not-found", detail: "x" }, "forum.example").needsCookies, false);
+});
+
+test("the fetcher claims thread URLs only, and probes offline", async () => {
+ assert.equal(xenforoFetcher.detect(THREAD_URL), true);
+ assert.equal(xenforoFetcher.detect("https://x.com/someone"), false);
+ assert.equal(xenforoFetcher.detect("https://forum.example/members/a.1/"), false);
+ const probe = await xenforoFetcher.probe(`${THREAD_URL}page-3`);
+ assert.deepEqual(probe, {
+ ok: true,
+ name: "The Teapot Collectors Thread",
+ handle: "the-teapot-collectors-thread.4242",
+ url: THREAD_URL,
+ });
+});
+
+// --- the page loader ---------------------------------------------------------------
+
+function fakePage(script: { status?: number; htmls: string[]; href?: string }) {
+ let clock = 0;
+ let reads = 0;
+ const page = {
+ waits: 0,
+ async goto() {
+ return { status: () => script.status ?? 200 };
+ },
+ async waitForSelector() {},
+ async waitForTimeout(ms: number) {
+ page.waits++;
+ clock += ms;
+ },
+ async evaluate(fn: string) {
+ if (fn === "location.href") return script.href ?? THREAD_URL;
+ const html = script.htmls[Math.min(reads, script.htmls.length - 1)];
+ reads++;
+ return html;
+ },
+ on() {},
+ async screenshot() {
+ return new Uint8Array();
+ },
+ };
+ return { page: page as unknown as PageLike & { waits: number }, now: () => clock };
+}
+
+test("loader: a browser check is waited out and the thread that follows is returned", async () => {
+ const thread = pageHtml(2);
+ const { page, now } = fakePage({ status: 403, htmls: [CHALLENGE_PAGE, CHALLENGE_PAGE, thread] });
+ const log: string[] = [];
+ const res = await loadForumPage(page, `${THREAD_URL}page-2`, {
+ signal: new AbortController().signal,
+ onLog: (l) => log.push(l),
+ now,
+ });
+ assert.equal(res.html, thread);
+ assert.equal(res.status, undefined, "the check's 403 does not describe the page that followed");
+ assert.equal(res.challengeMs, 4_000);
+ assert.match(log.join("\n"), /cleared in 4 s/);
+});
+
+test("loader: a check still up at the timeout is handed back for the walk to judge", async () => {
+ const { page, now } = fakePage({ status: 403, htmls: [CHALLENGE_PAGE] });
+ const res = await loadForumPage(page, THREAD_URL, {
+ signal: new AbortController().signal,
+ timeoutMs: 10_000,
+ now,
+ });
+ assert.equal(res.html, CHALLENGE_PAGE);
+ assert.ok((res.challengeMs ?? 0) >= 10_000);
+});
+
+test("loader: a plain page comes straight back with its status", async () => {
+ const { page, now } = fakePage({ status: 200, htmls: [pageHtml(1)] });
+ const res = await loadForumPage(page, THREAD_URL, { signal: new AbortController().signal, now });
+ assert.equal(res.status, 200);
+ assert.equal(res.challengeMs, undefined);
+ assert.equal(page.waits, 0);
+});
+
+test("the headless agent presents as the browser it is", () => {
+ assert.equal(
+ presentableUserAgent("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) HeadlessChrome/147.0.0.0 Safari/537.36"),
+ "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.0.0 Safari/537.36",
+ );
+});
+
+// --- capture -------------------------------------------------------------------------
+
+function captureLoader(pages: Record<string, string>, rect: { x: number; y: number; width: number; height: number } | null) {
+ const urls: string[] = [];
+ const got: string[] = [];
+ const page = {
+ async goto() {
+ return null;
+ },
+ async waitForSelector() {},
+ async waitForTimeout() {},
+ async evaluate(fn: string) {
+ if (fn.includes("getBoundingClientRect")) return { rect };
+ return 0;
+ },
+ on() {},
+ async screenshot() {
+ return new Uint8Array([137, 80, 78, 71]);
+ },
+ request: {
+ async get(url: string) {
+ got.push(url);
+ return {
+ ok: () => !url.includes("missing"),
+ status: () => (url.includes("missing") ? 404 : 200),
+ headers: () => ({ "content-type": "image/jpeg" }),
+ body: async () => new Uint8Array([1, 2, 3]),
+ };
+ },
+ },
+ } as unknown as PageLike;
+ return {
+ urls,
+ got,
+ loader: {
+ async page() {
+ return page;
+ },
+ async load(url: string) {
+ urls.push(url);
+ return { html: pages[url] ?? pageHtml(1), status: 200, url };
+ },
+ async close() {},
+ },
+ };
+}
+
+test("capture: shoots the post's article and downloads its file media through the profile", async () => {
+ const outDir = await mkdtemp(path.join(os.tmpdir(), "forum-capture-"));
+ const { loader, urls, got } = captureLoader({}, { x: 10, y: 400, width: 700, height: 300 });
+ const sleeps: number[] = [];
+ const res = await captureForumPosts(
+ {
+ ids: ["9001", "9002"],
+ handle: "t",
+ accountUrl: THREAD_URL,
+ outDir,
+ archived: new Map([
+ [
+ "9001",
+ {
+ text: "x",
+ url: `${ORIGIN}/posts/9001/`,
+ media: [
+ { kind: "image", url: "https://images.example/a.jpg" },
+ { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK" },
+ { kind: "attachment", url: `${ORIGIN}/attachments/scan-jpg.900/` },
+ ],
+ },
+ ],
+ ]),
+ signal: new AbortController().signal,
+ },
+ { loader, pauseMs: () => 7_000, sleep: async (ms) => void sleeps.push(ms), now: () => new Date(0) },
+ );
+ assert.deepEqual(res.outcomes.map((o) => [o.id, o.state, o.files]), [
+ ["9001", "captured", 3],
+ ["9002", "captured", 1],
+ ]);
+ assert.deepEqual(urls, [`${ORIGIN}/posts/9001/`, `${ORIGIN}/posts/9002/`]);
+ assert.deepEqual(got, ["https://images.example/a.jpg", `${ORIGIN}/attachments/scan-jpg.900/`], "embeds are pages, not files");
+ // Every contact after the first is paced: 2 downloads + the second shot.
+ assert.equal(sleeps.length, 3);
+ assert.deepEqual((await readdir(path.join(outDir, "9001"))).sort(), ["capture.json", "media-1.jpg", "media-2.jpg", "shot.png"]);
+ const rec = JSON.parse(await readFile(path.join(outDir, "9001", "capture.json"), "utf8"));
+ assert.equal(rec.state, "captured");
+ assert.equal(rec.mediaState, "ok");
+ assert.equal(rec.url, `${ORIGIN}/posts/9001/`);
+ const rec2 = JSON.parse(await readFile(path.join(outDir, "9002", "capture.json"), "utf8"));
+ assert.equal(rec2.mediaState, "none");
+});
+
+test("capture: a missing post is deleted; a check that will not clear stops the run for Connect", async () => {
+ const outDir = await mkdtemp(path.join(os.tmpdir(), "forum-capture-"));
+ const gone = captureLoader({ [`${ORIGIN}/posts/1/`]: NOT_FOUND_PAGE }, null);
+ const res = await captureForumPosts(
+ { ids: ["1"], handle: "t", accountUrl: THREAD_URL, outDir, signal: new AbortController().signal, media: false },
+ { loader: gone.loader, pauseMs: () => 0, sleep: async () => {} },
+ );
+ assert.equal(res.outcomes[0].state, "deleted");
+ assert.equal(res.outcomes[0].availability, "deleted");
+
+ const walled = captureLoader({ [`${ORIGIN}/posts/2/`]: CHALLENGE_PAGE }, null);
+ const res2 = await captureForumPosts(
+ { ids: ["2", "3"], handle: "t", accountUrl: THREAD_URL, outDir, signal: new AbortController().signal },
+ { loader: walled.loader, pauseMs: () => 0, sleep: async () => {} },
+ );
+ assert.equal(res2.needsCookies, true);
+ assert.match(res2.stoppedEarly ?? "", /Connect forum session/);
+ assert.equal(res2.outcomes.length, 1, "the rest are left for a later run");
+});
+
+test("downloadable media: files only", () => {
+ assert.deepEqual(
+ downloadableMedia([
+ { kind: "image", url: "https://a.example/1.png" },
+ { kind: "link-card", url: "https://a.example/page" },
+ { kind: "video", url: "./Saved_files/v.mp4" },
+ { kind: "video", url: "https://a.example/v.mp4" },
+ ]),
+ [
+ { kind: "image", url: "https://a.example/1.png" },
+ { kind: "video", url: "https://a.example/v.mp4" },
+ ],
+ );
+});
diff --git a/common/social/xenforoFetcher.ts b/common/social/xenforoFetcher.ts
@@ -0,0 +1,626 @@
+// A XenForo forum THREAD as a posts source (platform "xenforo"): the channel is
+// one thread, each forum post a Post. Kiwi Farms is the first host; any
+// XenForo 2 forum whose thread pages a browser can open works the same way.
+//
+// THE WALK, NEWEST FIRST. A thread grows at its end, so a run starts on the
+// last page (asked for with a page number past the end, which XenForo
+// redirects to the last page) and steps back a page at a time. It stops at the
+// first page holding an already-archived post (the incremental case), at the
+// watermark, at page 1, or at a cap — `pages` ("the latest N pages") or
+// `limit` (posts), checked after a whole page so a page is never half-read.
+// The cursor is the next page to read, so a capped or drained run resumes
+// there; a resumed run walks back to page 1 (pages shift as posts are
+// deleted, so meeting an archived post there does not end it).
+//
+// PACED, SERIAL, ONE BROWSER. One page at a time with a jittered pause between
+// loads (about 10–20 s by default; `pagePauseMs` from the channel), in one
+// headless Chromium on the host's persistent profile (forumSession.ts). Never
+// parallel, never faster on a slow run.
+//
+// A BROWSER CHECK (KiwiFlare's proof of work and the like) is waited out by the
+// loader, up to a minute; the profile keeps the clearance for later pages and
+// runs. A check that does not clear, a captcha, a login wall, a ban or a
+// 403/429 STOPS the run with a typed failure and keeps the cursor — never a
+// retry loop, never an attempt at a captcha. `needsCookies` is set where
+// Connect (a headed window on the same profile) is the way through.
+//
+// One parser for everything: xenforoParse.ts reads the live page here and the
+// operator's saved pages in the import (controller/importForumPages.ts).
+
+import { mkdir, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { getPaths } from "../lib/paths";
+import type { Post } from "../lib/posts";
+import {
+ registerSocialFetcher,
+ type PostCaptureInput,
+ type PostCaptureOutcome,
+ type PostCaptureResult,
+ type PostFetchInput,
+ type PostFetchResult,
+ type SocialFetcher,
+ type SocialFetcherProbe,
+} from "./fetchers";
+import {
+ browserForumLoader,
+ forumProfileDir,
+ type ForumPageLoader,
+} from "./forumSession";
+import {
+ captureAvailability,
+ captureWork,
+ describeCapturedFile,
+ postCaptureDir,
+ readPostCapture,
+ SHOT_FILENAME,
+ writePostCapture,
+ type CapturedFile,
+ type CaptureMediaState,
+ type PostCaptureRecord,
+ type PostCaptureState,
+} from "./postCapture";
+import type { PageLike } from "./playwrightRuntime";
+import {
+ classifyForumPage,
+ parseXenforoThreadPage,
+ parseXenforoThreadUrl,
+ xenforoPageUrl,
+ xenforoThreadHandle,
+ type ForumBlock,
+ type XenforoThreadUrl,
+} from "./xenforoParse";
+
+// A page number past any real thread's end: XenForo redirects it to the last
+// page, which saves reading page 1 just to learn where the end is.
+export const LAST_PAGE_PROBE = 1_000_000;
+
+export const DEFAULT_PAGE_PAUSE_MS = 12_000;
+// The floor a configured pause is held to.
+export const MIN_PAGE_PAUSE_MS = 5_000;
+
+// The pause before a load: `base` jittered to [0.85, 1.65) of itself — 12 s
+// gives 10.2–19.8 s.
+export function jitteredPause(base: number, random: () => number): number {
+ const b = Math.max(MIN_PAGE_PAUSE_MS, base);
+ return Math.round(b * (0.85 + random() * 0.8));
+}
+
+export function sleepUnlessAborted(ms: number, signal: AbortSignal): Promise<void> {
+ return new Promise((resolve) => {
+ if (signal.aborted) return resolve();
+ const timer = setTimeout(done, ms);
+ function done() {
+ clearTimeout(timer);
+ signal.removeEventListener("abort", done);
+ resolve();
+ }
+ signal.addEventListener("abort", done, { once: true });
+ });
+}
+
+// What a stop on a forum block says, and whether Connect is the way through.
+export function forumBlockFailure(
+ block: ForumBlock,
+ host: string,
+): { error: string; needsCookies: boolean } {
+ const connect =
+ `Use “Connect forum session” on the channel page: it opens a browser window on this host's profile ` +
+ `at the thread — clear it there, wait for the thread to show, close the window, and fetch again.`;
+ const shown = block.title ? ` (page title “${block.title}”)` : "";
+ switch (block.kind) {
+ case "challenge":
+ return {
+ error: `${host} kept its browser check (${block.detail}) up past the wait${shown}; the headless browser could not clear it. ${connect}`,
+ needsCookies: true,
+ };
+ case "captcha":
+ return {
+ error: `${host} asked for a captcha (${block.detail})${shown}, which only a person answers. ${connect}`,
+ needsCookies: true,
+ };
+ case "login":
+ return {
+ error: `${host} shows this thread only to a logged-in member${shown}. ${connect} (log in in that window).`,
+ needsCookies: true,
+ };
+ case "blocked":
+ return {
+ error: `${host} refused the page (${block.detail})${shown}. Stopped; the next run resumes here — do not run it again at once.`,
+ needsCookies: false,
+ };
+ case "not-found":
+ return { error: `${host} says the thread could not be found (${block.detail}).`, needsCookies: false };
+ default:
+ return {
+ error: `${host} answered with a page that is not a thread page (${block.detail})${shown}. Stopped.`,
+ needsCookies: false,
+ };
+ }
+}
+
+export type ThreadWalkDeps = {
+ loader: ForumPageLoader;
+ // The gap before each load after the run's first.
+ pauseMs: () => number;
+ sleep: (ms: number, signal: AbortSignal) => Promise<void>;
+};
+
+// THE WALK. Pure over its deps: the loader hands back HTML, the parser reads it.
+export async function walkXenforoThread(
+ input: PostFetchInput,
+ deps: ThreadWalkDeps,
+): Promise<PostFetchResult> {
+ const { channelSlug, seenIds, signal, onLog } = input;
+ const thread = parseXenforoThreadUrl(input.accountUrl);
+ if (!thread) {
+ return { posts: [], complete: false, error: `${input.accountUrl} is not a XenForo thread URL (…/threads/<title>.<id>/).` };
+ }
+ const resuming = Boolean(input.cursor);
+ const startPage = resuming ? Number(input.cursor) : undefined;
+ if (resuming && (!Number.isInteger(startPage) || startPage! < 1)) {
+ return { posts: [], complete: false, error: `The stored resume point "${input.cursor}" is not a page number.` };
+ }
+ const watermark = resuming ? undefined : input.since;
+ const stopAtKnown = (input.stopAtKnown ?? true) && !resuming;
+ const posts: Post[] = [];
+ const taken = new Set<string>();
+ let returnedCount = 0;
+ let loads = 0;
+ let pagesRead = 0;
+
+ const fetchPage = async (url: string) => {
+ if (loads++ > 0) await deps.sleep(deps.pauseMs(), signal);
+ return deps.loader.load(url, signal);
+ };
+
+ let page = startPage ?? LAST_PAGE_PROBE;
+ if (resuming) onLog?.(`Resuming the thread walk at page ${page}.`);
+ let lastPage: number | undefined;
+ const visited = new Set<number>();
+
+ try {
+ for (;;) {
+ if (signal.aborted) {
+ return { posts, complete: false, cursor: String(page) };
+ }
+ const url = xenforoPageUrl(thread, page);
+ let load;
+ try {
+ load = await fetchPage(url);
+ } catch (err) {
+ return {
+ posts,
+ complete: false,
+ cursor: page === LAST_PAGE_PROBE ? undefined : String(page),
+ error: (err as Error).message,
+ };
+ }
+ if (signal.aborted) return { posts, complete: false, cursor: page === LAST_PAGE_PROBE ? undefined : String(page) };
+ const block = classifyForumPage(load.html, load.status);
+ if (block) {
+ const why = forumBlockFailure(block, thread.host);
+ onLog?.(`Page ${page === LAST_PAGE_PROBE ? "(last)" : page}: ${block.kind} — ${block.detail}.`);
+ return {
+ posts,
+ complete: false,
+ ...(page === LAST_PAGE_PROBE ? {} : { cursor: String(page) }),
+ error: why.error,
+ ...(why.needsCookies ? { needsCookies: true } : {}),
+ };
+ }
+ const parsed = parseXenforoThreadPage(load.html, { channelSlug, pageUrl: load.url });
+ if (visited.has(parsed.page)) {
+ return {
+ posts,
+ complete: false,
+ cursor: String(page),
+ error: `Asked for page ${page} and was shown page ${parsed.page} again; stopped rather than loop.`,
+ };
+ }
+ visited.add(parsed.page);
+ page = parsed.page;
+ lastPage = Math.max(lastPage ?? 0, parsed.lastPage);
+ pagesRead++;
+
+ let sawKnown = false;
+ let sawOld = false;
+ const fresh: Post[] = [];
+ // Newest first within the page too, so the stops read in time order.
+ for (const post of [...parsed.posts].reverse()) {
+ // A post that belongs to another page — an article thread's first
+ // post, shown atop every page — says nothing about where this page
+ // stands: it never stops the walk, and is taken once if new.
+ const repeated = post.forum?.page !== undefined && post.forum.page !== parsed.page;
+ if (repeated) {
+ if (!seenIds.has(post.id) && !taken.has(post.id)) {
+ taken.add(post.id);
+ fresh.push(post);
+ }
+ continue;
+ }
+ if (seenIds.has(post.id)) {
+ if (stopAtKnown) sawKnown = true;
+ continue;
+ }
+ if (watermark && post.createdAt <= watermark) {
+ sawOld = true;
+ continue;
+ }
+ if (taken.has(post.id)) continue;
+ taken.add(post.id);
+ fresh.push(post);
+ }
+ fresh.reverse();
+ onLog?.(
+ `Page ${page}/${lastPage}: ${parsed.posts.length} post(s) parsed, ${fresh.length} new` +
+ (load.challengeMs !== undefined ? ` (browser check cleared in ${Math.round(load.challengeMs / 1000)} s)` : "") +
+ (sawKnown ? ", reached already-archived posts" : "") +
+ (sawOld ? ", reached the watermark" : "") +
+ ".",
+ );
+ if (parsed.posts.length === 0 && page > 1) {
+ // A thread page with no posts in it is not a page to step past
+ // silently: it is most likely markup this parser does not know.
+ return {
+ posts,
+ complete: false,
+ cursor: String(page),
+ error: `Page ${page} of the thread parsed to no posts; stopped rather than walking on.`,
+ };
+ }
+
+ const next = page - 1;
+ returnedCount += fresh.length;
+ if (input.onCheckpoint && fresh.length > 0) {
+ await input.onCheckpoint({ posts: fresh, cursor: String(Math.max(1, next)) });
+ } else {
+ posts.push(...fresh);
+ }
+
+ if (sawKnown || sawOld || next < 1) {
+ return { posts, complete: true };
+ }
+ if (input.pages && pagesRead >= input.pages) {
+ onLog?.(`Read ${pagesRead} page(s), the cap for this run; the next run continues at page ${next}.`);
+ return { posts, complete: false, cursor: String(next) };
+ }
+ if (input.limit && returnedCount >= input.limit) {
+ return { posts, complete: false, cursor: String(next) };
+ }
+ if (input.drain?.aborted) {
+ onLog?.(`Drained; the next run continues at page ${next}.`);
+ return { posts, complete: false, cursor: String(next), drained: true };
+ }
+ page = next;
+ }
+ } finally {
+ await deps.loader.close().catch(() => {});
+ }
+}
+
+// --- capture -------------------------------------------------------------------------
+
+// What the page shows for one post: the post's own article (its box, in
+// document coordinates), and whatever the page says instead.
+type ForumPostSnapshot = {
+ rect: { x: number; y: number; width: number; height: number } | null;
+};
+
+const SNAPSHOT_SCRIPT = (id: string) => `(() => {
+ const el = document.querySelector('article.message[data-content="post-${id}"]') ||
+ document.getElementById('js-post-${id}');
+ if (!el) return { rect: null };
+ el.scrollIntoView({ block: "center" });
+ const r = el.getBoundingClientRect();
+ return { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height } };
+})()`;
+
+// Let the post's images load (each capped), and open its spoilers, so the shot
+// shows what the post holds.
+const PREPARE_SCRIPT = (id: string) => `(() => {
+ const el = document.querySelector('article.message[data-content="post-${id}"]') ||
+ document.getElementById('js-post-${id}');
+ if (!el) return 0;
+ for (const s of el.querySelectorAll('.bbCodeSpoiler')) s.classList.add('is-active');
+ for (const c of el.querySelectorAll('.bbCodeBlock--expandable')) c.classList.add('is-expanded');
+ const imgs = Array.from(el.querySelectorAll('img')).filter((i) => !i.complete);
+ return Promise.all(imgs.map((i) => new Promise((r) => {
+ i.addEventListener('load', r, { once: true });
+ i.addEventListener('error', r, { once: true });
+ setTimeout(r, 5000);
+ }))).then(() => imgs.length);
+})()`;
+
+export type ForumShotResult = {
+ state: PostCaptureState;
+ shot?: CapturedFile;
+ error?: string;
+ stop?: string;
+};
+
+export async function shootForumPost(
+ load: (url: string) => Promise<{ html: string; status?: number }>,
+ page: PageLike,
+ postUrl: string,
+ id: string,
+ dir: string,
+ host: string,
+): Promise<ForumShotResult> {
+ let got;
+ try {
+ got = await load(postUrl);
+ } catch (err) {
+ return { state: "error", error: (err as Error).message };
+ }
+ const block = classifyForumPage(got.html, got.status);
+ if (block) {
+ if (block.kind === "not-found") return { state: "deleted" };
+ const why = forumBlockFailure(block, host);
+ return {
+ state: block.kind === "challenge" || block.kind === "captcha" || block.kind === "login" ? "login-wall" : "error",
+ error: why.error,
+ stop: why.error,
+ };
+ }
+ await page.waitForTimeout(1_000);
+ await page.evaluate(PREPARE_SCRIPT(id)).catch(() => {});
+ const snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as ForumPostSnapshot;
+ if (!snap.rect) {
+ // The thread rendered without this post: XenForo shows a deleted post to
+ // nobody but moderators, and a moved post redirects elsewhere.
+ return { state: "deleted" };
+ }
+ const r = snap.rect;
+ if (r.width < 1 || r.height < 1) return { state: "error", error: "The post rendered with no size to shoot." };
+ let png: Uint8Array;
+ try {
+ png = await page.screenshot({
+ type: "png",
+ fullPage: true,
+ clip: { x: Math.max(0, Math.floor(r.x)), y: Math.max(0, Math.floor(r.y)), width: Math.ceil(r.width), height: Math.ceil(r.height) },
+ });
+ } catch (err) {
+ return { state: "error", error: `The screenshot failed: ${((err as Error).message ?? "").split("\n")[0]}` };
+ }
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, SHOT_FILENAME), png);
+ return { state: "captured", shot: await describeCapturedFile(dir, SHOT_FILENAME, postUrl) };
+}
+
+const EXT_BY_TYPE: Record<string, string> = {
+ "image/jpeg": "jpg",
+ "image/png": "png",
+ "image/gif": "gif",
+ "image/webp": "webp",
+ "image/avif": "avif",
+ "video/mp4": "mp4",
+ "video/webm": "webm",
+ "application/pdf": "pdf",
+};
+
+function extFor(url: string, contentType: string | undefined): string {
+ const ct = (contentType ?? "").split(";")[0].trim().toLowerCase();
+ if (EXT_BY_TYPE[ct]) return EXT_BY_TYPE[ct];
+ const m = /\.([a-z0-9]{2,5})(?:[?#/]|$)/i.exec(new URL(url).pathname.replace(/-([a-z0-9]{2,5})\.\d+\/?$/i, ".$1"));
+ return m ? m[1].toLowerCase() : "bin";
+}
+
+// The media a capture downloads: images, attachments and videos the post
+// carries as files. Embeds and link cards are pages, not files.
+export function downloadableMedia(
+ media: ReadonlyArray<{ kind: string; url: string }> | undefined,
+): { kind: string; url: string }[] {
+ return (media ?? []).filter(
+ (m) => (m.kind === "image" || m.kind === "attachment" || m.kind === "video") && /^https?:\/\//.test(m.url),
+ );
+}
+
+export type ForumCaptureDeps = {
+ loader: ForumPageLoader & { page(): Promise<PageLike> };
+ pauseMs: () => number;
+ sleep: (ms: number, signal: AbortSignal) => Promise<void>;
+ now?: () => Date;
+};
+
+const STOP_AFTER_ERRORS = 3;
+
+// The capture loop for forum posts: per id, the post's page (its permalink,
+// which XenForo resolves to the post in its thread), the shot of its article,
+// then its media fetched through the same browser profile (so a clearance
+// cookie covers the downloads too). Every contact after the first is paced.
+export async function captureForumPosts(
+ input: PostCaptureInput,
+ deps: ForumCaptureDeps,
+): Promise<PostCaptureResult> {
+ const { signal, onLog } = input;
+ const thread = parseXenforoThreadUrl(input.accountUrl ?? "");
+ if (!thread) {
+ return { outcomes: [], stoppedEarly: `${input.accountUrl ?? "(no URL)"} is not a XenForo thread URL.` };
+ }
+ const wanted = { shots: input.shots ?? true, media: input.media ?? true, force: input.force ?? false };
+ const now = deps.now ?? (() => new Date());
+ const outcomes: PostCaptureOutcome[] = [];
+ let contacts = 0;
+ let errorsInARow = 0;
+ const contact = async () => {
+ if (contacts++ > 0) await deps.sleep(deps.pauseMs(), signal);
+ };
+ const stopped = (why: string, extra: Partial<PostCaptureResult> = {}): PostCaptureResult => {
+ onLog?.(why);
+ return { outcomes, stoppedEarly: why, ...extra };
+ };
+
+ try {
+ for (const [i, id] of input.ids.entries()) {
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ if (input.drain?.aborted) return stopped("Drained; the rest are left for a later run.");
+ const dir = postCaptureDir(input.outDir, id);
+ const existing = await readPostCapture(dir);
+ const work = captureWork(existing, { ...wanted, articles: false });
+ if (!work.shot && !work.media) {
+ onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`);
+ continue;
+ }
+ onLog?.(`[${i + 1}/${input.ids.length}] ${id}`);
+ const archived = input.archived?.get(id);
+ const postUrl =
+ archived?.url && /^https?:\/\//.test(archived.url) ? archived.url : `${thread.origin}/posts/${id}/`;
+
+ let state: PostCaptureState | undefined = work.shot ? undefined : existing?.state;
+ let shot = work.shot ? undefined : existing?.shot;
+ let error: string | undefined;
+ let stop: string | undefined;
+
+ if (work.shot) {
+ await contact();
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ const page = await deps.loader.page();
+ const res = await shootForumPost(
+ (url) => deps.loader.load(url, signal),
+ page,
+ postUrl,
+ id,
+ dir,
+ thread.host,
+ );
+ state = res.state;
+ shot = res.shot;
+ error = res.error;
+ stop = res.stop;
+ }
+
+ let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped");
+ let media: CapturedFile[] = work.media ? [] : (existing?.media ?? []);
+ const postIsThere = state === undefined || state === "captured";
+ if (work.media && postIsThere && !stop) {
+ const files = downloadableMedia(archived?.media);
+ if (files.length === 0) {
+ mediaState = "none";
+ state ??= "captured";
+ } else {
+ const page = await deps.loader.page();
+ let failed = 0;
+ for (const [n, m] of files.entries()) {
+ await contact();
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ try {
+ const res = await page.request!.get(m.url, { timeout: 60_000, failOnStatusCode: false });
+ if (!res.ok()) {
+ failed++;
+ onLog?.(`${id}: media ${n + 1} answered HTTP ${res.status()}.`);
+ continue;
+ }
+ const name = `media-${n + 1}.${extFor(m.url, res.headers()["content-type"])}`;
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, name), await res.body());
+ media.push(await describeCapturedFile(dir, name, m.url));
+ } catch (err) {
+ failed++;
+ onLog?.(`${id}: media ${n + 1} failed: ${((err as Error).message ?? "").split("\n")[0]}`);
+ }
+ }
+ mediaState = failed > 0 ? "error" : "ok";
+ if (failed > 0) error = error ? `${error}; ${failed} media file(s) failed` : `${failed} media file(s) failed`;
+ state ??= "captured";
+ }
+ }
+
+ const finalState: PostCaptureState = state ?? "error";
+ const record: PostCaptureRecord = {
+ version: 1,
+ id,
+ url: postUrl,
+ capturedAt: now().toISOString(),
+ state: finalState,
+ ...(shot ? { shot } : {}),
+ mediaState,
+ media,
+ ...(error ? { error } : {}),
+ };
+ await writePostCapture(dir, record);
+ outcomes.push({
+ id,
+ state: finalState,
+ ...(work.shot && captureAvailability(finalState) ? { availability: captureAvailability(finalState) } : {}),
+ files: (shot ? 1 : 0) + media.length,
+ ...(error ? { error } : {}),
+ });
+ onLog?.(
+ `${id}: ${finalState}` +
+ (shot ? ", shot" : "") +
+ (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") +
+ (error ? ` — ${error}` : ""),
+ );
+ if (stop) {
+ return stopped(`${stop} Stopped at ${id}; the rest are left for a later run.`, {
+ needsCookies: finalState === "login-wall",
+ });
+ }
+ errorsInARow = finalState === "error" ? errorsInARow + 1 : 0;
+ if (errorsInARow >= STOP_AFTER_ERRORS) {
+ return stopped(`${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`);
+ }
+ }
+ return { outcomes };
+ } finally {
+ await deps.loader.close().catch(() => {});
+ }
+}
+
+// --- the fetcher ----------------------------------------------------------------------
+
+function threadOrThrow(url: string): XenforoThreadUrl {
+ const t = parseXenforoThreadUrl(url);
+ if (!t) throw new Error(`${url} is not a XenForo thread URL (…/threads/<title>.<id>/).`);
+ return t;
+}
+
+const livePause = (base?: number) => () => jitteredPause(base ?? DEFAULT_PAGE_PAUSE_MS, Math.random);
+
+export const xenforoFetcher: SocialFetcher = {
+ id: "xenforo-thread",
+ label: "Forum thread (XenForo, headless browser)",
+ platform: "xenforo",
+ fields: { limit: true },
+
+ detect(url: string): boolean {
+ return parseXenforoThreadUrl(url) !== null;
+ },
+
+ // Offline: a probe that loaded the thread would be a paced page load (and a
+ // browser check) just to fill a form. The URL names the thread.
+ async probe(url: string): Promise<SocialFetcherProbe> {
+ const t = parseXenforoThreadUrl(url);
+ if (!t) return { ok: false, error: "Not a XenForo thread URL (…/threads/<title>.<id>/)." };
+ const handle = xenforoThreadHandle(url) ?? t.threadId;
+ const words = handle.replace(/\.\d+$/, "").replace(/[-_]+/g, " ").trim();
+ return {
+ ok: true,
+ name: words ? words.replace(/\b\w/g, (c) => c.toUpperCase()) : `Thread ${t.threadId}`,
+ handle,
+ url: t.base,
+ };
+ },
+
+ async fetch(input: PostFetchInput): Promise<PostFetchResult> {
+ const thread = threadOrThrow(input.accountUrl);
+ const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog });
+ return walkXenforoThread(input, {
+ loader,
+ pauseMs: livePause(input.pagePauseMs),
+ sleep: sleepUnlessAborted,
+ });
+ },
+
+ async captureByIds(input: PostCaptureInput): Promise<PostCaptureResult> {
+ const thread = threadOrThrow(input.accountUrl ?? "");
+ const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog });
+ return captureForumPosts(input, {
+ loader,
+ pauseMs: livePause(input.pagePauseMs),
+ sleep: sleepUnlessAborted,
+ });
+ },
+};
+
+registerSocialFetcher(xenforoFetcher);
diff --git a/common/social/xenforoParse.test.ts b/common/social/xenforoParse.test.ts
@@ -0,0 +1,245 @@
+// The XenForo thread parser over SYNTHETIC pages (__fixtures__/xenforoPages.ts):
+// posts, quotes, links, media, edits, the page nav, a browser-saved page, and
+// the pages that stand in a thread's way.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xenforoParse.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ classifyForumPage,
+ embedTarget,
+ parseXenforoThreadPage,
+ parseXenforoThreadUrl,
+ resolveForumUrl,
+ xenforoPageUrl,
+ xenforoThreadHandle,
+} from "./xenforoParse";
+import { parsePost } from "../lib/posts";
+import {
+ CAPTCHA_PAGE,
+ CHALLENGE_PAGE,
+ LOGIN_PAGE,
+ NOT_FOUND_PAGE,
+ ORIGIN,
+ RICH_BODY,
+ THREAD_URL,
+ threadPage,
+ type FakePost,
+} from "./__fixtures__/xenforoPages";
+
+const T0 = 1_760_000_000; // epoch seconds
+
+function post(id: number, position: number, body = `Post number ${position}.`, extra: Partial<FakePost> = {}): FakePost {
+ return { id, author: `Member${id % 5}`, userId: 100 + (id % 5), ts: T0 + position * 60, position, body, ...extra };
+}
+
+test("thread URLs: friendly, paged, prefixed and non-friendly forms", () => {
+ const t = parseXenforoThreadUrl(`${THREAD_URL}page-7`);
+ assert.deepEqual(t, {
+ origin: ORIGIN,
+ host: "forum.example",
+ base: THREAD_URL,
+ threadId: "4242",
+ page: 7,
+ });
+ assert.equal(parseXenforoThreadUrl(`${THREAD_URL}post-99`)?.base, THREAD_URL);
+ assert.equal(parseXenforoThreadUrl("https://forum.example/community/threads/x.12/")?.base, "https://forum.example/community/threads/x.12/");
+ assert.equal(parseXenforoThreadUrl("https://forum.example/index.php?threads/x.12/")?.threadId, "12");
+ assert.equal(parseXenforoThreadUrl("https://forum.example/threads/12/")?.threadId, "12");
+ assert.equal(parseXenforoThreadUrl("https://forum.example/members/someone.3/"), null);
+ assert.equal(parseXenforoThreadUrl("not a url"), null);
+ assert.equal(xenforoPageUrl({ base: THREAD_URL }, 1), THREAD_URL);
+ assert.equal(xenforoPageUrl({ base: THREAD_URL }, 3), `${THREAD_URL}page-3`);
+ assert.equal(xenforoThreadHandle(`${THREAD_URL}page-2`), "the-teapot-collectors-thread.4242");
+});
+
+test("a page: thread metadata, page numbers and every post in page order", () => {
+ const html = threadPage({ page: 3, last: 9, posts: [post(301, 41), post(302, 42), post(303, 43)] });
+ const page = parseXenforoThreadPage(html, { channelSlug: "teapots" });
+ assert.equal(page.origin, ORIGIN);
+ assert.equal(page.host, "forum.example");
+ assert.equal(page.threadId, "4242");
+ assert.equal(page.threadTitle, "The Teapot Collectors Thread");
+ assert.equal(page.threadUrl, THREAD_URL);
+ assert.equal(page.page, 3);
+ assert.equal(page.lastPage, 9);
+ assert.deepEqual(page.posts.map((p) => p.id), ["301", "302", "303"]);
+ const p = page.posts[1];
+ assert.equal(p.slug, "teapots/302");
+ assert.equal(p.platform, "xenforo");
+ assert.equal(p.author, "Member2");
+ assert.equal(p.createdAt, new Date((T0 + 42 * 60) * 1000).toISOString());
+ assert.equal(p.uploadDate.length, 8);
+ assert.equal(p.url, `${ORIGIN}/posts/302/`);
+ assert.equal(p.text, "Post number 42.");
+ assert.equal(p.isReply, true);
+ assert.equal(p.isRepost, false);
+ assert.deepEqual(p.forum, {
+ host: "forum.example",
+ threadId: "4242",
+ page: 3,
+ threadTitle: "The Teapot Collectors Thread",
+ threadUrl: THREAD_URL,
+ position: 42,
+ authorId: "102",
+ });
+ // Round-trips through the stored-record validator unchanged.
+ assert.deepEqual(parsePost(JSON.parse(JSON.stringify(p))), p);
+});
+
+test("positions with thousands separators, the first post, and a single-page thread", () => {
+ const html = threadPage({ page: 1, last: 1, posts: [post(1, 1, "Opening post."), post(2, 2101)] });
+ const page = parseXenforoThreadPage(html, { channelSlug: "teapots" });
+ assert.equal(page.page, 1);
+ assert.equal(page.lastPage, 1);
+ assert.equal(page.posts[0].forum?.position, 1);
+ assert.equal(page.posts[0].isReply, false);
+ assert.equal(page.posts[1].forum?.position, 2101);
+});
+
+test("the body: quotes, mentions, links, smilies, images, embeds, spoilers, video, unfurls, lists, attachments", () => {
+ const html = threadPage({ page: 2, last: 2, posts: [post(500, 30, RICH_BODY)] });
+ const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts;
+ const lines = p.text.split("\n");
+ // The quote: a header naming the quoted member and post, then its lines.
+ assert.deepEqual(lines.slice(0, 3), [
+ "> Quoting Marigold (post 1001):",
+ "> The blue one is a reproduction.",
+ "> Look at the glaze.",
+ ]);
+ assert.ok(!p.text.includes("Click to expand"), "the expand link is chrome");
+ assert.ok(!p.text.includes("Marigold said:"), "the quote title is replaced by the header");
+ assert.ok(p.text.includes("I disagree, @Marigold."), "a mention is its text");
+ // A blank line survives <br><br>.
+ assert.ok(/@Marigold\.\n\nHere is the catalogue/.test(p.text));
+ assert.ok(p.text.includes("Here is the catalogue: https://archive.example/AbCd1"));
+ assert.ok(p.text.includes("museum page (https://museum.example/teapots?id=9) :)"));
+ assert.ok(p.text.includes("[image]"));
+ assert.ok(p.text.includes("[embed: https://www.youtube.com/watch?v=AbCdEfGhIjK]"));
+ assert.ok(p.text.includes("[embed: https://x.com/i/status/1234567890123456789]"));
+ assert.ok(p.text.includes("[Spoiler: the ending]\nThe lid was glued on.\n[/Spoiler]"));
+ assert.ok(p.text.includes("[video]"));
+ assert.ok(p.text.includes("https://news.example/teapot-auction"));
+ assert.ok(!p.text.includes("An auction report"), "an unfurl card is its URL, not its blurb");
+ assert.ok(p.text.includes("- first point\n- second point"));
+ assert.ok(!/ /.test(p.text), "no runs of spaces");
+
+ assert.deepEqual(p.forum?.quotes, [{ postId: "1001", author: "Marigold", authorId: "7" }]);
+ assert.deepEqual(p.quoted, {
+ platform: "xenforo",
+ id: "1001",
+ url: `${ORIGIN}/posts/1001/`,
+ author: "Marigold",
+ });
+ assert.deepEqual(p.replyTo, p.quoted);
+ assert.deepEqual(p.links, [
+ "https://archive.example/AbCd1",
+ "https://museum.example/teapots?id=9",
+ "https://www.youtube.com/watch?v=AbCdEfGhIjK",
+ "https://x.com/i/status/1234567890123456789",
+ "https://news.example/teapot-auction",
+ ]);
+ assert.deepEqual(p.media, [
+ { kind: "image", url: "https://images.example/teapot.jpg", name: "teapot.jpg" },
+ { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK", provider: "youtube" },
+ { kind: "embed", url: "https://x.com/i/status/1234567890123456789", provider: "twitter" },
+ { kind: "video", url: `${ORIGIN}/data/video/12/12345-abc.mp4` },
+ { kind: "link-card", url: "https://news.example/teapot-auction" },
+ { kind: "attachment", url: `${ORIGIN}/attachments/receipt-png.555/` },
+ { kind: "image", url: `${ORIGIN}/data/attachments/0/555-receipt.jpg`, name: "receipt.png" },
+ ]);
+ assert.equal(p.mediaCount, p.media?.length);
+});
+
+test("an edited post records when, and listed attachments are media", () => {
+ const html = threadPage({
+ page: 1,
+ last: 1,
+ posts: [
+ post(600, 5, "Edited words.", {
+ editedTs: T0 + 9999,
+ attachments: [{ href: "/attachments/scan-jpg.900/", name: "scan.jpg" }],
+ }),
+ ],
+ });
+ const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts;
+ assert.equal(p.forum?.editedAt, new Date((T0 + 9999) * 1000).toISOString());
+ // The edit time is not the post time.
+ assert.equal(p.createdAt, new Date((T0 + 5 * 60) * 1000).toISOString());
+ assert.deepEqual(p.media, [{ kind: "attachment", url: `${ORIGIN}/attachments/scan-jpg.900/`, name: "scan.jpg" }]);
+});
+
+test("a browser-saved page: no canonical link, rewritten assets, the saved-from comment names it", () => {
+ const html = threadPage({ page: 4, last: 6, saved: true, posts: [post(700, 61), post(701, 62)] });
+ assert.equal(classifyForumPage(html), null);
+ const page = parseXenforoThreadPage(html, { channelSlug: "teapots" });
+ assert.equal(page.origin, ORIGIN);
+ assert.equal(page.threadId, "4242");
+ assert.equal(page.page, 4);
+ assert.equal(page.lastPage, 6);
+ assert.equal(page.posts.length, 2);
+ assert.equal(page.posts[0].url, `${ORIGIN}/posts/700/`);
+ // With no saved-from comment either, the caller's URL is the fallback, and
+ // the content key still names the thread.
+ const bare = html.replace(/<!-- saved from[^>]*-->/, "");
+ const fromFallback = parseXenforoThreadPage(bare, { channelSlug: "teapots", pageUrl: THREAD_URL });
+ assert.equal(fromFallback.origin, ORIGIN);
+ const noUrl = parseXenforoThreadPage(bare, { channelSlug: "teapots" });
+ assert.equal(noUrl.threadId, "4242");
+ assert.equal(noUrl.posts.length, 2);
+});
+
+test("URL resolution keeps saved-asset paths and absolutises forum paths", () => {
+ assert.equal(resolveForumUrl("/attachments/a.1/", ORIGIN), `${ORIGIN}/attachments/a.1/`);
+ assert.equal(resolveForumUrl("./Thread_files/a.jpg", ORIGIN), "./Thread_files/a.jpg");
+ assert.equal(resolveForumUrl("Thread_files/a.jpg", ORIGIN), "Thread_files/a.jpg");
+ assert.equal(resolveForumUrl("//cdn.example/x.png", ORIGIN), "https://cdn.example/x.png");
+ assert.equal(resolveForumUrl("data:image/png;base64,AAAA", ORIGIN), undefined);
+ assert.equal(resolveForumUrl("#post-3", ORIGIN), undefined);
+ assert.deepEqual(embedTarget("https://rumble.com/embed/v1abc/?pub=4"), {
+ url: "https://rumble.com/embed/v1abc/",
+ provider: "rumble",
+ });
+});
+
+test("pages in the way: a browser check, a captcha, a login, a missing thread, a refusal", () => {
+ const thread = threadPage({ page: 1, last: 1, posts: [post(1, 1)] });
+ assert.equal(classifyForumPage(thread), null);
+ assert.equal(classifyForumPage(thread, 200), null);
+
+ const challenge = classifyForumPage(CHALLENGE_PAGE, 403);
+ assert.equal(challenge?.kind, "challenge");
+ assert.equal(challenge?.title, "Checking your browser");
+ assert.equal(classifyForumPage(CAPTCHA_PAGE)?.kind, "captcha");
+ assert.equal(classifyForumPage(LOGIN_PAGE)?.kind, "login");
+ assert.equal(classifyForumPage(NOT_FOUND_PAGE)?.kind, "not-found");
+ assert.equal(classifyForumPage("<html><body>nothing</body></html>", 429)?.kind, "blocked");
+ assert.equal(classifyForumPage("<html><body>nothing</body></html>", 404)?.kind, "not-found");
+ assert.equal(classifyForumPage("<html><body>You have been banned.</body></html>")?.kind, "blocked");
+ assert.equal(classifyForumPage("<html><body>nothing</body></html>")?.kind, "unknown");
+ // A check page that names a captcha in passing is still a check (waited
+ // out), not a captcha (stopped at once).
+ assert.equal(
+ classifyForumPage(CHALLENGE_PAGE.replace("</body>", "<p>No captcha needed.</p></body>"))?.kind,
+ "challenge",
+ );
+});
+
+test("an article thread's first post, atop a later page with no #N, is post 1 of page 1", () => {
+ const html = threadPage({ page: 3, last: 4, article: post(1, 1, "The article."), posts: [post(41, 41), post(42, 42)] });
+ const page = parseXenforoThreadPage(html, { channelSlug: "teapots" });
+ assert.deepEqual(page.posts.map((p) => [p.id, p.forum?.position, p.forum?.page]), [
+ ["1", 1, 1],
+ ["41", 41, 3],
+ ["42", 42, 3],
+ ]);
+ assert.equal(page.posts[0].isReply, false);
+});
+
+test("a message with no id or no date is not a post", () => {
+ const html = threadPage({ page: 1, last: 1, posts: [post(1, 1)] })
+ .replace('data-timestamp="', 'data-x="')
+ .replace(/datetime="[^"]*"/, "");
+ assert.equal(parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts.length, 0);
+});
diff --git a/common/social/xenforoParse.ts b/common/social/xenforoParse.ts
@@ -0,0 +1,840 @@
+// XenForo 2 thread pages → `Post`s. The ONE parser: the live fetcher hands it
+// the page a headless browser loaded (xenforoFetcher.ts), the import hands it
+// a page the operator saved from their own browser (controller/importForumPages.ts), and
+// both get the same posts.
+//
+// Pure: no browser, no network, no fs. The HTML is read with the small shared
+// reader (htmlReader.ts), so every marker below runs in tests against
+// synthetic pages.
+//
+// THE MARKUP (XenForo 2, as served):
+// article.message[data-author][data-content="post-N"]#js-post-N
+// (.message--article: an article thread's first post, repeated atop
+// every page with no "#N")
+// .message-user … a.username[data-user-id] the author
+// header.message-attribution
+// .message-attribution-main time.u-dt[data-timestamp] posted
+// .message-attribution-opposite a "#N" position
+// .bbWrapper the body
+// blockquote.bbCodeBlock--quote[data-quote][data-source="post: N"]
+// .bbCodeSpoiler, .bbCodeBlock--code, .bbImageWrapper, iframe, video…
+// .message-attachments li.file a[href] attachments
+// .message-lastEdit time.u-dt last edited
+// h1.p-title-value thread title
+// .pageNav-page(--current) page / last page
+//
+// A page SAVED from a browser has rewritten asset URLs (`./Thread_files/…`);
+// the original absolute URL is kept wherever the markup still carries it
+// (data-url, data-src, an anchor's href), else what is there is kept.
+
+import {
+ postPermalink,
+ uploadDateFromCreatedAt,
+ type ForumPostInfo,
+ type Post,
+ type PostMedia,
+ type PostRef,
+} from "../lib/posts";
+import { XENFORO_THREAD_PATH_RE } from "../lib/detectPlatform.mjs";
+import { parseHtml, type HtmlElement, type HtmlNode } from "./htmlReader";
+
+// --- thread URLs ---------------------------------------------------------------
+
+export type XenforoThreadUrl = {
+ origin: string; // "https://forum.example"
+ host: string;
+ // The thread's URL with no page: "https://forum.example/threads/a-title.123/"
+ base: string;
+ threadId: string;
+ // The page the URL names, when it names one (/page-N).
+ page?: number;
+};
+
+// A XenForo thread URL, read; null when the URL is not one. Accepts the
+// friendly form (/threads/<slug>.<id>/, under any path prefix) with an
+// optional /page-N and /post-N, and the non-friendly index.php?threads/… form.
+export function parseXenforoThreadUrl(url: string): XenforoThreadUrl | null {
+ let u: URL;
+ try {
+ u = new URL(url.trim());
+ } catch {
+ return null;
+ }
+ if (u.protocol !== "http:" && u.protocol !== "https:") return null;
+ const where = u.pathname + u.search;
+ const m = XENFORO_THREAD_PATH_RE.exec(where);
+ if (!m) return null;
+ const threadId = m[1];
+ // Everything up to and including the thread segment (and its slash).
+ const end = m.index + m[0].length;
+ let basePath = where.slice(0, end);
+ if (!basePath.endsWith("/")) basePath += "/";
+ const pageM = /(?:^|\/)page-(\d+)(?:\/|$|[?#])/.exec(where.slice(end));
+ const out: XenforoThreadUrl = {
+ origin: u.origin,
+ host: u.hostname.toLowerCase(),
+ base: `${u.origin}${basePath}`,
+ threadId,
+ };
+ if (pageM) out.page = Number(pageM[1]);
+ return out;
+}
+
+// The URL of page `page` of a thread (page 1 is the bare thread URL).
+export function xenforoPageUrl(t: Pick<XenforoThreadUrl, "base">, page: number): string {
+ return page <= 1 ? t.base : `${t.base}page-${page}`;
+}
+
+// The thread key a channel uses as its handle: "<slug>.<id>" (or the id).
+export function xenforoThreadHandle(url: string): string | null {
+ const t = parseXenforoThreadUrl(url);
+ if (!t) return null;
+ const seg = /threads\/([^/?#]+)\/?$/.exec(t.base);
+ return seg ? decodeURIComponent(seg[1]) : t.threadId;
+}
+
+// --- the page's state: a thread, or something in its way -------------------------
+
+export type ForumBlockKind =
+ // A JavaScript check (KiwiFlare's proof of work, a "just a moment" page)
+ // that a real browser normally clears by itself.
+ | "challenge"
+ // A captcha: a person has to answer it.
+ | "captcha"
+ // Refused outright: 403 / 429 / a ban or access-denied page.
+ | "blocked"
+ // The thread is only shown to a logged-in member.
+ | "login"
+ // The forum says the thread (or post) does not exist.
+ | "not-found"
+ // Not a thread page, and nothing above recognised.
+ | "unknown";
+
+export type ForumBlock = {
+ kind: ForumBlockKind;
+ // What was recognised, for the log ("KiwiFlare marker", "HTTP 429").
+ detail: string;
+ title?: string;
+};
+
+const POST_MARKER_RE = /data-content="post-\d+"|id="js-post-\d+"/;
+const THREAD_TEMPLATE_RE = /data-template="thread_view"/;
+
+const CAPTCHA_MARKERS: [RegExp, string][] = [
+ [/h-captcha|hcaptcha\.com/i, "hCaptcha"],
+ [/g-recaptcha|google\.com\/recaptcha|recaptcha\/api/i, "reCAPTCHA"],
+ [/cf-turnstile|challenges\.cloudflare\.com\/turnstile/i, "Turnstile"],
+];
+// The bare word is weaker than a widget: a browser check's own page may name
+// it, so it is read only after the check markers.
+const CAPTCHA_WORD = /\bcaptcha\b/i;
+
+const CHALLENGE_MARKERS: [RegExp, string][] = [
+ [/kiwiflare/i, "KiwiFlare"],
+ [/\/\.sssg\/|\bsssg[_-]/i, "KiwiFlare (sssg)"],
+ [/proof[- ]of[- ]work/i, "a proof-of-work check"],
+ [/checking your browser/i, "a browser check"],
+ [/just a moment\.\.\.|cf-chl|challenge-platform|cf_chl_/i, "Cloudflare's browser check"],
+ [/ddos-guard/i, "DDoS-Guard"],
+ [/please wait while (your request|we) .{0,40}verif/i, "a verification page"],
+];
+
+const BLOCK_TEXT: [RegExp, string][] = [
+ [/you have been banned|your (ip|access) (has been|is) (banned|blocked)/i, "a ban page"],
+ [/access denied|error 1020|403 forbidden/i, "an access-denied page"],
+ [/too many requests|rate limit/i, "a rate-limit page"],
+];
+
+const LOGIN_TEXT =
+ /you must be logged[- ]in to do that|you do not have permission to view this page|data-template="login"/i;
+const NOT_FOUND_TEXT =
+ /the requested (thread|post|page) could not be found|data-template="error"[^>]*>[\s\S]{0,4000}could not be found/i;
+
+function titleOf(html: string): string | undefined {
+ const m = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html);
+ const t = m?.[1].replace(/\s+/g, " ").trim();
+ return t || undefined;
+}
+
+// What stands between this page and the thread, or null when it IS a thread
+// page (it carries posts, or XenForo's thread template). `status` is the HTTP
+// status the page came with, when known.
+export function classifyForumPage(html: string, status?: number): ForumBlock | null {
+ if (POST_MARKER_RE.test(html) || THREAD_TEMPLATE_RE.test(html)) return null;
+ const title = titleOf(html);
+ const withTitle = (b: Omit<ForumBlock, "title">): ForumBlock => (title ? { ...b, title } : b);
+ for (const [re, what] of CAPTCHA_MARKERS) {
+ if (re.test(html)) return withTitle({ kind: "captcha", detail: what });
+ }
+ for (const [re, what] of CHALLENGE_MARKERS) {
+ if (re.test(html)) return withTitle({ kind: "challenge", detail: what });
+ }
+ if (CAPTCHA_WORD.test(html)) return withTitle({ kind: "captcha", detail: "a captcha" });
+ if (LOGIN_TEXT.test(html)) return withTitle({ kind: "login", detail: "the forum asks for a login" });
+ if (NOT_FOUND_TEXT.test(html) || status === 404) {
+ return withTitle({ kind: "not-found", detail: status === 404 ? "HTTP 404" : "the forum says it could not be found" });
+ }
+ if (status === 401 || status === 403 || status === 429 || status === 503) {
+ return withTitle({ kind: "blocked", detail: `HTTP ${status}` });
+ }
+ for (const [re, what] of BLOCK_TEXT) {
+ if (re.test(html)) return withTitle({ kind: "blocked", detail: what });
+ }
+ return withTitle({ kind: "unknown", detail: "no forum posts on the page" });
+}
+
+// --- tree helpers ------------------------------------------------------------------
+
+const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string";
+const classList = (el: HtmlElement): string[] => (el.attrs.class ?? "").split(/\s+/).filter(Boolean);
+const hasClass = (el: HtmlElement, c: string) => classList(el).includes(c);
+const hasClassPrefix = (el: HtmlElement, p: string) => classList(el).some((c) => c.startsWith(p));
+
+function findAll(
+ el: HtmlElement,
+ pred: (e: HtmlElement) => boolean,
+ out: HtmlElement[] = [],
+ skip?: (e: HtmlElement) => boolean,
+): HtmlElement[] {
+ for (const c of el.children) {
+ if (!isEl(c)) continue;
+ if (pred(c)) out.push(c);
+ if (skip?.(c)) continue;
+ findAll(c, pred, out, skip);
+ }
+ return out;
+}
+
+function findFirst(
+ el: HtmlElement,
+ pred: (e: HtmlElement) => boolean,
+ skip?: (e: HtmlElement) => boolean,
+): HtmlElement | null {
+ for (const c of el.children) {
+ if (!isEl(c)) continue;
+ if (pred(c)) return c;
+ if (skip?.(c)) continue;
+ const hit = findFirst(c, pred, skip);
+ if (hit) return hit;
+ }
+ return null;
+}
+
+// Plain text of an element: whitespace collapsed, scripts dropped.
+function plainText(node: HtmlNode): string {
+ if (!isEl(node)) return node;
+ if (node.tag === "script" || node.tag === "style" || node.tag === "template") return "";
+ return node.children.map(plainText).join("");
+}
+const squash = (s: string) => s.replace(/[\s\u00a0]+/g, " ").trim();
+
+// --- URLs ------------------------------------------------------------------------
+
+// A URL from the markup, absolute against the forum's origin. A path a browser
+// wrote when it SAVED the page (`./Thread_files/x.jpg`, a `file:` URL) is kept
+// as it is: it names nothing on the forum.
+export function resolveForumUrl(raw: string | undefined, origin: string | undefined): string | undefined {
+ const v = raw?.trim();
+ if (!v || v.startsWith("data:") || v.startsWith("javascript:") || v.startsWith("#")) return undefined;
+ if (/^[a-z][a-z0-9+.-]*:/i.test(v)) return v;
+ if (v.startsWith("//")) return `https:${v}`;
+ if (v.startsWith("./") || v.startsWith("../") || /_files\//.test(v)) return v;
+ if (!origin) return v;
+ try {
+ return new URL(v, `${origin}/`).href;
+ } catch {
+ return v;
+ }
+}
+
+// The canonical page URL an embed's iframe stands for, and its provider.
+export function embedTarget(src: string, provider?: string): { url: string; provider?: string } {
+ const yt = /youtube(?:-nocookie)?\.com\/embed\/([A-Za-z0-9_-]{6,})/.exec(src);
+ if (yt) return { url: `https://www.youtube.com/watch?v=${yt[1]}`, provider: "youtube" };
+ // s9e's media embeds: an iframe page with the item id in the fragment.
+ const s9e = /s9e\.github\.io\/iframe\/\d+\/([a-z0-9]+)(?:\.min)?\.html#([^&?]+)/i.exec(src);
+ if (s9e) {
+ const kind = s9e[1].toLowerCase();
+ const id = decodeURIComponent(s9e[2]);
+ if (kind === "twitter" && /^\d+$/.test(id)) {
+ return { url: `https://x.com/i/status/${id}`, provider: "twitter" };
+ }
+ if (kind === "youtube") return { url: `https://www.youtube.com/watch?v=${id}`, provider: "youtube" };
+ return { url: src, provider: provider ?? kind };
+ }
+ const tw = /platform\.twitter\.com\/embed\/.*[?&]id=(\d+)/.exec(src);
+ if (tw) return { url: `https://x.com/i/status/${tw[1]}`, provider: "twitter" };
+ const rumble = /rumble\.com\/embed\/([A-Za-z0-9]+)/.exec(src);
+ if (rumble) return { url: `https://rumble.com/embed/${rumble[1]}/`, provider: "rumble" };
+ return provider ? { url: src, provider } : { url: src };
+}
+
+// --- the body → text ------------------------------------------------------------------
+
+type BodyCtx = {
+ origin?: string;
+ links: string[];
+ media: PostMedia[];
+ quotes: NonNullable<ForumPostInfo["quotes"]>;
+};
+
+const BLOCK_TAGS = new Set([
+ "address", "article", "aside", "blockquote", "dd", "details", "div", "dl",
+ "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5",
+ "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section",
+ "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul",
+]);
+const SILENT_TAGS = new Set(["script", "style", "noscript", "template", "svg", "button", "input", "select", "canvas"]);
+
+function addLink(ctx: BodyCtx, url: string | undefined) {
+ if (url && /^https?:\/\//i.test(url) && !ctx.links.includes(url)) ctx.links.push(url);
+}
+function addMedia(ctx: BodyCtx, m: PostMedia) {
+ if (!ctx.media.some((x) => x.url === m.url && x.kind === m.kind)) ctx.media.push(m);
+}
+
+const isQuote = (el: HtmlElement) =>
+ el.tag === "blockquote" && (hasClass(el, "bbCodeBlock--quote") || "data-quote" in el.attrs);
+const isSpoiler = (el: HtmlElement) => hasClass(el, "bbCodeSpoiler") || hasClass(el, "bbCodeInlineSpoiler");
+const isUnfurl = (el: HtmlElement) =>
+ hasClass(el, "bbCodeBlock--unfurl") || (el.attrs["data-unfurl"] === "true" && !!el.attrs["data-url"]);
+const isImageWrapper = (el: HtmlElement) => hasClass(el, "bbImageWrapper");
+const isSmilie = (el: HtmlElement) =>
+ el.tag === "img" && (hasClass(el, "smilie") || hasClassPrefix(el, "smilie--") || "data-shortname" in el.attrs);
+
+function imageUrl(el: HtmlElement, ctx: BodyCtx, wrapper?: HtmlElement): string | undefined {
+ return resolveForumUrl(
+ el.attrs["data-url"] || wrapper?.attrs["data-src"] || el.attrs["data-src"] || el.attrs.src,
+ ctx.origin,
+ );
+}
+
+// A quote's source post id: data-source="post: 123", else the jump link.
+function quoteSource(el: HtmlElement): { postId?: string; author?: string; authorId?: string } {
+ const out: { postId?: string; author?: string; authorId?: string } = {};
+ const src = /post:\s*(\d+)/.exec(el.attrs["data-source"] ?? "");
+ if (src) out.postId = src[1];
+ if (!out.postId) {
+ const jump = findFirst(el, (e) => e.tag === "a" && hasClass(e, "bbCodeBlock-sourceJump"));
+ const m =
+ /[?&]id=(\d+)/.exec(jump?.attrs.href ?? "") ??
+ /post-(\d+)/.exec(jump?.attrs["data-content-selector"] ?? "") ??
+ /\/posts\/(\d+)/.exec(jump?.attrs.href ?? "");
+ if (m) out.postId = m[1];
+ }
+ const author = squash(el.attrs["data-quote"] ?? "");
+ if (author) out.author = author;
+ const member = /member:\s*(\d+)/.exec(el.attrs["data-attributes"] ?? "");
+ if (member) out.authorId = member[1];
+ return out;
+}
+
+// A block boundary: a line break that merges with any break beside it, where
+// "\n" (a <br>) is a break of its own — so <br><br> is a blank line and a
+// list item is not.
+const SOFT = "\u0000";
+
+// Render one body subtree. Text is emitted with "\n" for <br> and SOFT around
+// blocks; `finish` turns the run into lines.
+function render(node: HtmlNode, ctx: BodyCtx, out: string[]): void {
+ if (!isEl(node)) {
+ out.push(node.replace(/[\s\u00a0]+/g, " "));
+ return;
+ }
+ const el = node;
+ if (SILENT_TAGS.has(el.tag) && !(el.tag === "button" && isSpoilerButton(el))) return;
+ if (hasClass(el, "bbCodeBlock-expandLink") || hasClass(el, "bbCodeBlock-title")) return;
+ if (el.tag === "br") {
+ out.push("\n");
+ return;
+ }
+
+ if (isQuote(el)) {
+ const src = quoteSource(el);
+ ctx.quotes.push(src);
+ const inner: string[] = [];
+ const content = findFirst(el, (e) => hasClass(e, "bbCodeBlock-content")) ?? el;
+ for (const c of content.children) render(c, ctx, inner);
+ const body = finish(inner);
+ const head =
+ `Quoting ${src.author ?? "an earlier post"}` + (src.postId ? ` (post ${src.postId})` : "") + ":";
+ const lines = [head, ...(body ? body.split("\n") : [])];
+ out.push(SOFT, lines.map((l) => (l ? `> ${l}` : ">")).join("\n"), SOFT);
+ return;
+ }
+
+ if (isSpoiler(el)) {
+ const titleEl = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-button-title"));
+ const title = titleEl ? squash(plainText(titleEl)) : "";
+ const content = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-content") || hasClass(e, "bbCodeBlock-content")) ?? el;
+ const inner: string[] = [];
+ for (const c of content.children) render(c, ctx, inner);
+ const inline = hasClass(el, "bbCodeInlineSpoiler");
+ if (inline) {
+ out.push(`[spoiler: ${finish(inner).replace(/\n+/g, " ")}]`);
+ return;
+ }
+ out.push(SOFT, `[Spoiler${title && !/^spoiler$/i.test(title) ? `: ${title}` : ""}]`, "\n", finish(inner), "\n", "[/Spoiler]", SOFT);
+ return;
+ }
+
+ if (isUnfurl(el)) {
+ const url = resolveForumUrl(el.attrs["data-url"], ctx.origin);
+ if (url) {
+ addLink(ctx, url);
+ addMedia(ctx, { kind: "link-card", url });
+ out.push(SOFT, url, SOFT);
+ }
+ return;
+ }
+
+ if (isImageWrapper(el)) {
+ const img = findFirst(el, (e) => e.tag === "img");
+ const url = img ? imageUrl(img, ctx, el) : resolveForumUrl(el.attrs["data-src"], ctx.origin);
+ if (url) {
+ const name = el.attrs.title || img?.attrs.alt;
+ addMedia(ctx, { kind: "image", url, ...(name ? { name } : {}) });
+ }
+ out.push("[image]");
+ return;
+ }
+
+ if (el.tag === "img") {
+ if (isSmilie(el)) {
+ out.push(el.attrs.alt ?? "");
+ return;
+ }
+ const url = imageUrl(el, ctx);
+ if (url) addMedia(ctx, { kind: "image", url, ...(el.attrs.alt ? { name: el.attrs.alt } : {}) });
+ out.push("[image]");
+ return;
+ }
+
+ if (el.tag === "video" || el.tag === "audio") {
+ const src =
+ el.attrs.src ??
+ findFirst(el, (e) => e.tag === "source" && !!e.attrs.src)?.attrs.src;
+ const url = resolveForumUrl(src, ctx.origin);
+ if (url) addMedia(ctx, { kind: "video", url });
+ out.push(el.tag === "video" ? "[video]" : "[audio]");
+ return;
+ }
+
+ if (el.tag === "iframe" || "data-s9e-mediaembed" in el.attrs) {
+ const frame = el.tag === "iframe" ? el : findFirst(el, (e) => e.tag === "iframe");
+ const provider = el.attrs["data-s9e-mediaembed"] || frame?.attrs["data-s9e-mediaembed"] || undefined;
+ const raw =
+ frame?.attrs["data-s9e-mediaembed-src"] || frame?.attrs.src || frame?.attrs["data-src"] ||
+ el.attrs["data-s9e-mediaembed-src"];
+ const src = resolveForumUrl(raw, ctx.origin);
+ if (src && /^https?:/i.test(src)) {
+ const target = embedTarget(src, provider);
+ addMedia(ctx, { kind: "embed", url: target.url, ...(target.provider ? { provider: target.provider } : {}) });
+ addLink(ctx, target.url);
+ out.push(SOFT, `[embed: ${target.url}]`, SOFT);
+ } else if (frame || provider) {
+ out.push("[embed]");
+ }
+ return;
+ }
+
+ if (el.tag === "blockquote" && hasClass(el, "twitter-tweet")) {
+ const link = findAll(el, (e) => e.tag === "a" && /\/status\/\d+/.test(e.attrs.href ?? "")).pop();
+ const url = link?.attrs.href;
+ if (url) {
+ addMedia(ctx, { kind: "embed", url, provider: "twitter" });
+ addLink(ctx, url);
+ out.push(SOFT, `[embed: ${url}]`, SOFT);
+ }
+ return;
+ }
+
+ if (el.tag === "a") {
+ renderLink(el, ctx, out);
+ return;
+ }
+
+ if (el.tag === "pre" || el.tag === "code") {
+ if (el.tag === "pre" || hasClass(el, "bbCodeCode")) {
+ out.push(SOFT, plainText(el).replace(/\u00a0/g, " ").replace(/^\n+|\n+$/g, ""), SOFT);
+ return;
+ }
+ }
+
+ const block = BLOCK_TAGS.has(el.tag);
+ if (block) out.push(SOFT);
+ if (el.tag === "li") out.push("- ");
+ for (const c of el.children) render(c, ctx, out);
+ if (block) out.push(SOFT);
+}
+
+function isSpoilerButton(el: HtmlElement): boolean {
+ return hasClass(el, "bbCodeSpoiler-button");
+}
+
+function renderLink(el: HtmlElement, ctx: BodyCtx, out: string[]): void {
+ const href = el.attrs.href;
+ // A member mention or an in-forum jump: its text is the content.
+ const mention = hasClass(el, "username") || "data-user-id" in el.attrs;
+ const internalJump = !href || href.startsWith("#") || /\/goto\/post/.test(href);
+ const hasImg = !!findFirst(el, (e) => e.tag === "img" && !isSmilie(e));
+ if (mention || internalJump) {
+ for (const c of el.children) render(c, ctx, out);
+ return;
+ }
+ const url = resolveForumUrl(el.attrs["data-url"] || href, ctx.origin);
+ if (hasImg) {
+ // An image that links somewhere: an attachment's full-size page, or a
+ // clickable image. The attachment link is the medium; the image inside is
+ // rendered (and recorded) as usual.
+ if (url && /\/attachments\//.test(url)) addMedia(ctx, { kind: "attachment", url });
+ else addLink(ctx, url);
+ for (const c of el.children) render(c, ctx, out);
+ return;
+ }
+ const inner: string[] = [];
+ for (const c of el.children) render(c, ctx, inner);
+ const text = squash(inner.join(""));
+ if (!url) {
+ out.push(text);
+ return;
+ }
+ if (/\/attachments\//.test(url)) addMedia(ctx, { kind: "attachment", url, ...(text ? { name: text } : {}) });
+ else addLink(ctx, url);
+ const bare = text.replace(/(…|\.\.\.)$/, "");
+ if (!text || url === text || (bare.length > 8 && url.includes(bare)) || text === href) {
+ out.push(url);
+ } else {
+ out.push(`${text} (${url})`);
+ }
+}
+
+// Emitted pieces → lines: each line trimmed, at most one blank line in a row,
+// none at the ends.
+function finish(pieces: string[]): string {
+ // Every run of breaks (soft or hard, with the spaces between them) becomes
+ // one line break, or a blank line when it holds two or more hard ones.
+ const joined = pieces
+ .join("")
+ .replace(/[ \t]*[\u0000\n][\u0000\n \t]*/g, (run) =>
+ (run.match(/\n/g)?.length ?? 0) >= 2 ? "\n\n" : "\n",
+ );
+ const lines = joined.split("\n").map((l) => l.replace(/ {2,}/g, " ").trim());
+ const out: string[] = [];
+ for (const l of lines) {
+ if (!l && (out.length === 0 || out[out.length - 1] === "")) continue;
+ out.push(l);
+ }
+ while (out.length && out[out.length - 1] === "") out.pop();
+ return out.join("\n");
+}
+
+// A post body element → text, links, media and quotes.
+export function forumBodyToText(
+ body: HtmlElement,
+ origin?: string,
+): { text: string; links: string[]; media: PostMedia[]; quotes: NonNullable<ForumPostInfo["quotes"]> } {
+ const ctx: BodyCtx = { origin, links: [], media: [], quotes: [] };
+ const pieces: string[] = [];
+ for (const c of body.children) render(c, ctx, pieces);
+ return { text: finish(pieces), links: ctx.links, media: ctx.media, quotes: ctx.quotes };
+}
+
+// --- times --------------------------------------------------------------------------
+
+// A XenForo <time>: data-timestamp (epoch seconds) is exact; datetime
+// ("2024-01-02T03:04:05-0500") is the fallback.
+export function forumTimeIso(el: HtmlElement | null | undefined): string | undefined {
+ if (!el) return undefined;
+ const ts = Number(el.attrs["data-timestamp"] ?? el.attrs["data-time"]);
+ if (Number.isFinite(ts) && ts > 0) return new Date(ts * 1000).toISOString();
+ const dt = el.attrs.datetime;
+ if (dt) {
+ // "-0500" → "-05:00", which every Date parser takes.
+ const norm = dt.replace(/([+-]\d{2})(\d{2})$/, "$1:$2");
+ const ms = Date.parse(norm);
+ if (Number.isFinite(ms)) return new Date(ms).toISOString();
+ }
+ return undefined;
+}
+
+const isTime = (e: HtmlElement) => e.tag === "time" && (hasClass(e, "u-dt") || !!e.attrs["data-timestamp"] || !!e.attrs.datetime);
+
+// --- a whole page ---------------------------------------------------------------------
+
+export type XenforoThreadPage = {
+ origin?: string;
+ host?: string;
+ threadId?: string;
+ threadTitle?: string;
+ // The thread's URL without a page.
+ threadUrl?: string;
+ page: number;
+ lastPage: number;
+ // In page order (oldest first).
+ posts: Post[];
+};
+
+// A post: message--post, or an article thread's starter (message--article).
+const isMessage = (e: HtmlElement) =>
+ e.tag === "article" &&
+ hasClass(e, "message") &&
+ (/^post-\d+$/.test(e.attrs["data-content"] ?? "") || /^js-post-\d+$/.test(e.attrs.id ?? ""));
+
+function postIdOf(el: HtmlElement): string | undefined {
+ return (
+ /^post-(\d+)$/.exec(el.attrs["data-content"] ?? "")?.[1] ??
+ /^js-post-(\d+)$/.exec(el.attrs.id ?? "")?.[1] ??
+ /\/posts\/(\d+)/.exec(el.attrs.itemid ?? "")?.[1]
+ );
+}
+
+function metaContent(root: HtmlElement, pred: (e: HtmlElement) => boolean): string | undefined {
+ return findFirst(root, (e) => e.tag === "meta" && pred(e))?.attrs.content;
+}
+
+// Where the page came from: its canonical link, og:url, the comment a browser
+// writes into a saved page, else the caller's URL.
+function pageUrlOf(root: HtmlElement, html: string, fallback?: string): string | undefined {
+ const canonical = findFirst(root, (e) => e.tag === "link" && (e.attrs.rel ?? "").split(/\s+/).includes("canonical"))?.attrs.href;
+ if (canonical && /^https?:\/\//.test(canonical)) return canonical;
+ const og = metaContent(root, (e) => e.attrs.property === "og:url");
+ if (og && /^https?:\/\//.test(og)) return og;
+ const saved = /<!--\s*saved from url=\(\d+\)(https?:\/\/[^\s>]+)\s*-->/i.exec(html.slice(0, 4096));
+ if (saved) return saved[1];
+ return fallback;
+}
+
+function threadTitleOf(root: HtmlElement): string | undefined {
+ const h1 = findFirst(root, (e) => e.tag === "h1" && hasClass(e, "p-title-value"));
+ if (h1) {
+ const text = squash(
+ h1.children
+ .map((c) => (isEl(c) && (hasClass(c, "label") || hasClass(c, "label-append") || hasClass(c, "labelLink")) ? "" : plainText(c)))
+ .join(""),
+ );
+ if (text) return text;
+ }
+ const og = metaContent(root, (e) => e.attrs.property === "og:title");
+ if (og) return squash(og);
+ const title = findFirst(root, (e) => e.tag === "title");
+ const t = title ? squash(plainText(title)) : "";
+ return t ? t.replace(/\s+\|\s+[^|]+$/, "") : undefined;
+}
+
+function pageNumbers(root: HtmlElement): { current?: number; last?: number } {
+ const nav = findAll(root, (e) => hasClass(e, "pageNav-page"));
+ let current: number | undefined;
+ let last: number | undefined;
+ for (const li of nav) {
+ const n = Number(squash(plainText(li)).replace(/[^\d]/g, ""));
+ if (!Number.isFinite(n) || n < 1) continue;
+ if (hasClass(li, "pageNav-page--current")) current ??= n;
+ last = Math.max(last ?? 0, n);
+ }
+ // The page-jump box carries the last page as its max.
+ for (const input of findAll(root, (e) => e.tag === "input" && hasClass(e, "js-pageJumpPage"))) {
+ const max = Number(input.attrs.max);
+ if (Number.isFinite(max) && max >= 1) last = Math.max(last ?? 0, max);
+ }
+ // The compact nav ("3 of 120").
+ if (last === undefined) {
+ const simple = findFirst(root, (e) => hasClass(e, "pageNavSimple-el--current"));
+ const m = simple ? /(\d+)\s+of\s+(\d+)/i.exec(squash(plainText(simple))) : null;
+ if (m) {
+ current ??= Number(m[1]);
+ last = Number(m[2]);
+ }
+ }
+ return { current, last };
+}
+
+export type ParseXenforoOptions = {
+ channelSlug: string;
+ // The URL the page was loaded from (the live fetcher) — a fallback when the
+ // page names none itself.
+ pageUrl?: string;
+};
+
+export function parseXenforoThreadPage(html: string, opts: ParseXenforoOptions): XenforoThreadPage {
+ const root = parseHtml(html);
+ const pageUrl = pageUrlOf(root, html, opts.pageUrl);
+ const thread = pageUrl ? parseXenforoThreadUrl(pageUrl) : null;
+ const fallbackThread = opts.pageUrl ? parseXenforoThreadUrl(opts.pageUrl) : null;
+ const t = thread ?? fallbackThread;
+ let origin = t?.origin;
+ if (!origin && pageUrl) {
+ try {
+ origin = new URL(pageUrl).origin;
+ } catch {
+ /* no origin */
+ }
+ }
+ const host = origin ? new URL(origin).hostname.toLowerCase() : undefined;
+ const htmlEl = findFirst(root, (e) => e.tag === "html");
+ const contentKey = /^thread-(\d+)$/.exec(htmlEl?.attrs["data-content-key"] ?? "")?.[1];
+ const threadId = t?.threadId ?? contentKey;
+ const threadTitle = threadTitleOf(root);
+ const nums = pageNumbers(root);
+ const page = nums.current ?? t?.page ?? 1;
+ const lastPage = Math.max(page, nums.last ?? page);
+
+ const posts: Post[] = [];
+ const seen = new Set<string>();
+ for (const msg of findAll(root, isMessage, [], isMessage)) {
+ const post = messageToPost(msg, {
+ channelSlug: opts.channelSlug,
+ origin,
+ host,
+ threadId,
+ threadTitle,
+ threadUrl: t?.base,
+ page,
+ });
+ if (post && !seen.has(post.id)) {
+ seen.add(post.id);
+ posts.push(post);
+ }
+ }
+ return {
+ ...(origin ? { origin } : {}),
+ ...(host ? { host } : {}),
+ ...(threadId ? { threadId } : {}),
+ ...(threadTitle ? { threadTitle } : {}),
+ ...(t?.base ? { threadUrl: t.base } : {}),
+ page,
+ lastPage,
+ posts,
+ };
+}
+
+type MessageCtx = {
+ channelSlug: string;
+ origin?: string;
+ host?: string;
+ threadId?: string;
+ threadTitle?: string;
+ threadUrl?: string;
+ page: number;
+};
+
+// One article.message → a Post, or null when it carries no id or no date (a
+// deleted-post placeholder, an ad slot dressed as a message).
+export function messageToPost(msg: HtmlElement, ctx: MessageCtx): Post | null {
+ const id = postIdOf(msg);
+ if (!id) return null;
+ const inBody = (e: HtmlElement) => hasClass(e, "bbWrapper") || hasClass(e, "message-body");
+ const isLastEdit = (e: HtmlElement) => hasClass(e, "message-lastEdit");
+ const attribution =
+ findFirst(msg, (e) => hasClass(e, "message-attribution-main")) ??
+ findFirst(msg, (e) => hasClass(e, "message-attribution"));
+ const time =
+ (attribution ? findFirst(attribution, isTime) : null) ??
+ findFirst(msg, isTime, (e) => inBody(e) || isLastEdit(e));
+ const createdAt = forumTimeIso(time);
+ if (!createdAt) return null;
+
+ const userLink =
+ findFirst(msg, (e) => hasClass(e, "username") && !!e.attrs["data-user-id"], inBody) ??
+ findFirst(msg, (e) => !!e.attrs["data-user-id"], inBody);
+ const nameEl = findFirst(msg, (e) => hasClass(e, "message-name"), inBody);
+ const author = squash(msg.attrs["data-author"] ?? "") || (nameEl ? squash(plainText(nameEl)) : "") || (userLink ? squash(plainText(userLink)) : "");
+ const authorId = userLink?.attrs["data-user-id"];
+
+ let position: number | undefined;
+ const opposite = findFirst(msg, (e) => hasClass(e, "message-attribution-opposite"));
+ for (const a of findAll(opposite ?? msg, (e) => e.tag === "a", [], inBody)) {
+ const m = /^#\s*([\d,]+)$/.exec(squash(plainText(a)));
+ if (m) {
+ position = Number(m[1].replace(/,/g, ""));
+ break;
+ }
+ }
+
+ // An ARTICLE thread shows its first post (message--article) at the top of
+ // every page, with no "#N": it is post 1, of page 1, wherever it is read.
+ const articleStarter = hasClass(msg, "message--article");
+ if (articleStarter && position === undefined) position = 1;
+
+ const bodyEl = findFirst(msg, (e) => hasClass(e, "bbWrapper"));
+ const body = bodyEl
+ ? forumBodyToText(bodyEl, ctx.origin)
+ : { text: "", links: [], media: [], quotes: [] };
+
+ // Attachments listed under the post (not inline in the body).
+ const attachments = findFirst(msg, (e) => hasClass(e, "message-attachments"));
+ if (attachments) {
+ for (const li of findAll(attachments, (e) => e.tag === "li" && hasClass(e, "file"))) {
+ const a =
+ findFirst(li, (e) => e.tag === "a" && /\/attachments\//.test(e.attrs.href ?? "")) ??
+ findFirst(li, (e) => e.tag === "a" && !!e.attrs.href && !hasClass(e, "u-anchorTarget"));
+ const url = resolveForumUrl(a?.attrs.href, ctx.origin);
+ if (!url) continue;
+ const nameEl2 = findFirst(li, (e) => hasClass(e, "file-name"));
+ const name = nameEl2 ? nameEl2.attrs.title || squash(plainText(nameEl2)) : undefined;
+ if (!body.media.some((m) => m.url === url)) {
+ body.media.push({ kind: "attachment", url, ...(name ? { name } : {}) });
+ }
+ }
+ }
+
+ const editedAt = forumTimeIso(findFirst(findFirst(msg, isLastEdit) ?? { tag: "#", attrs: {}, children: [] }, isTime));
+
+ const itemid = msg.attrs.itemid;
+ const url =
+ itemid && /^https?:\/\//.test(itemid)
+ ? itemid
+ : ctx.origin
+ ? postPermalink("xenforo", author, id, ctx.origin)
+ : `/posts/${id}/`;
+
+ const forum: ForumPostInfo = {
+ host: ctx.host ?? "",
+ threadId: ctx.threadId ?? "",
+ page: articleStarter && position === 1 ? 1 : ctx.page,
+ };
+ if (ctx.threadTitle) forum.threadTitle = ctx.threadTitle;
+ if (ctx.threadUrl) forum.threadUrl = ctx.threadUrl;
+ if (position) forum.position = position;
+ if (authorId && authorId !== "0") forum.authorId = authorId;
+ if (editedAt) forum.editedAt = editedAt;
+ if (body.quotes.length > 0) forum.quotes = body.quotes;
+
+ const firstQuoted = body.quotes.find((q) => q.postId);
+ const quotedRef: PostRef | undefined = firstQuoted?.postId
+ ? {
+ platform: "xenforo",
+ id: firstQuoted.postId,
+ ...(ctx.origin ? { url: postPermalink("xenforo", "", firstQuoted.postId, ctx.origin) } : {}),
+ ...(firstQuoted.author ? { author: firstQuoted.author } : {}),
+ }
+ : undefined;
+
+ const post: Post = {
+ id,
+ slug: `${ctx.channelSlug}/${id}`,
+ channelSlug: ctx.channelSlug,
+ author,
+ createdAt,
+ uploadDate: uploadDateFromCreatedAt(createdAt),
+ text: body.text,
+ url,
+ platform: "xenforo",
+ // Every post after the thread's first answers the thread.
+ isReply: position !== 1,
+ isRepost: false,
+ links: body.links,
+ };
+ if (quotedRef) {
+ post.quoted = quotedRef;
+ post.replyTo = quotedRef;
+ }
+ if (body.media.length > 0) {
+ post.media = body.media;
+ post.mediaCount = body.media.length;
+ }
+ // `forum` is meaningful only with a host and a thread; a page that names
+ // neither (a bare fragment) still yields the post.
+ if (forum.host && forum.threadId) post.forum = forum;
+ return post;
+}