Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 18582324729636b2e40cef5a77a2c7e713e0b12a
parent a38ef94914b2d5ecfc92a5799231fcc96dd28a52
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 14:55:41 -0400

posts: forum threads (XenForo) as a posts source — parser, paced walk, import, capture

A XenForo thread is a posts channel (platform "xenforo"), each forum post a
Post. One parser (social/xenforoParse.ts) reads live pages and browser-saved
pages; the headless walk (social/xenforoFetcher.ts) goes newest first from the
last page, paced and resumable, waits out a browser check on a persistent
per-host profile (social/forumSession.ts) and stops typed when one will not
clear. `archilyzer posts import-html` merges saved pages (upsertPosts: an
edited post is updated, append-only). capture-posts shoots a forum post's
article and downloads its media through the same profile. Post gains optional
`media` and `forum`; the HTML reader moves to social/htmlReader.ts.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/bin/archilyzer.ts | 21++++++++++++++++++++-
Mcommon/bin/fetch-posts.ts | 15+++++++++++++--
Acommon/bin/posts-import-html.ts | 35+++++++++++++++++++++++++++++++++++
Mcommon/components/PostModal.tsx | 44+++++++++++++++++++++++++++++++++++++++++---
Mcommon/components/SearchResults.tsx | 3++-
Mcommon/components/postsCache.ts | 23+++++++++++++----------
Mcommon/controller/capturePosts.ts | 31++++++++++++++++++++++++++-----
Mcommon/controller/checkPostAvailability.ts | 1+
Mcommon/controller/fetchPosts.ts | 9+++++++++
Acommon/controller/importForumPages.test.ts | 116+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/importForumPages.ts | 153+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.ts | 13+++++++++++++
Mcommon/lib/channelConfig.ts | 5+++++
Mcommon/lib/channelConfigSchema.ts | 1+
Mcommon/lib/corpus.ts | 8++++++--
Mcommon/lib/detectPlatform.mjs | 15+++++++++++++++
Acommon/lib/forumPosts.test.ts | 147+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/platform.ts | 11+++++++++--
Mcommon/lib/posts-server.ts | 90+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/posts.ts | 195+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/lib/report/convert-server.ts | 2+-
Mcommon/lib/report/convertManifest.ts | 8+++++++-
Acommon/social/__fixtures__/xenforoPages.ts | 222+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/fetchers.ts | 27+++++++++++++++++++++++++--
Acommon/social/forumSession.ts | 406+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/htmlReader.ts | 121+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xArticle.ts | 125++++++++-----------------------------------------------------------------------
Acommon/social/xenforoFetcher.test.ts | 523+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xenforoFetcher.ts | 626+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xenforoParse.test.ts | 245+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xenforoParse.ts | 840+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
31 files changed, 3934 insertions(+), 147 deletions(-)

diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -333,7 +333,26 @@ export const COMMANDS: Command[] = [ script(["duplicates"], "duplicate-shorts.ts", "[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192), script(["posts", "fetch"], "fetch-posts.ts", - "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post)"), + "--slug <channel> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post; --pages: a forum thread's latest N pages)"), + { + path: ["posts", "import-html"], + usage: + "<slug> <file-or-dir>… [--dry-run] import forum thread pages saved from a browser (\"Save page as\", .html) into a forum-thread channel: new posts appended, edited ones updated, nothing fetched", + flags: { "dry-run": "boolean" }, + maxPositionals: 1000, + run: async ({ positionals, flags }) => { + const [slug, ...inputs] = positionals; + if (!slug || inputs.length === 0) { + console.error("posts import-html: pass the channel slug, then one or more saved pages or directories."); + return 2; + } + return (await import("./posts-import-html")).main({ + slug, + inputs, + dryRun: flags["dry-run"] === true, + }); + }, + }, script(["posts", "check"], "check-post-availability.ts", "--slug <channel> [--mode stale|unchecked|all] [--limit N] which archived posts were deleted at the source"), script(["diarize", "backfill"], "diarize-backfill.ts", diff --git a/common/bin/fetch-posts.ts b/common/bin/fetch-posts.ts @@ -2,7 +2,11 @@ // Fetch social posts for one channel into its on-disk posts corpus. // // pnpm --filter yt-dlp-transcript-common exec tsx bin/fetch-posts.ts \ -// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] +// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N] +// +// --pages caps how many pages one run reads, for a source read page by page (a +// forum thread: its latest N pages, newest first; the next run continues where +// this one stopped). // // --full re-walks the account's whole history instead of stopping at the // stored watermark. New posts are still deduped against the posts-archive, so @@ -24,7 +28,7 @@ const flags = parseFlags(process.argv.slice(2)); const slug = flags.slug; if (!slug) { console.error( - "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N]", + "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD] [--force]] [--limit N] [--pages N]", ); process.exit(2); } @@ -35,6 +39,12 @@ const limit = ? Math.floor(limitRaw) : undefined; +const pagesRaw = flags.pages ? Number(flags.pages) : undefined; +const pages = + typeof pagesRaw === "number" && Number.isFinite(pagesRaw) && pagesRaw > 0 + ? Math.floor(pagesRaw) + : undefined; + fetchPosts({ paths: getPaths(), slug, @@ -44,6 +54,7 @@ fetchPosts({ floor: flags.floor, force: flags.force === "true", limit, + pages, onLog: (line) => console.log(line), }) .then((result) => { diff --git a/common/bin/posts-import-html.ts b/common/bin/posts-import-html.ts @@ -0,0 +1,35 @@ +// `archilyzer posts import-html <slug> <file-or-dir>… [--dry-run]` — forum +// thread pages the operator saved from a browser, imported into a forum-thread +// channel (controller/importForumPages.ts). Offline: nothing is fetched. New +// posts are appended, a post saved again after an edit is updated, and an +// unchanged one is left alone. The index picks them up on its next build. + +import { isValidChannelSlug } from "../controller/channels"; +import { importForumPages } from "../controller/importForumPages"; +import { getPaths } from "../lib/paths"; + +export async function main(opts: { + slug: string; + inputs: string[]; + dryRun: boolean; +}): Promise<number> { + if (!isValidChannelSlug(opts.slug)) { + console.error(`posts import-html: "${opts.slug}" is not a channel slug`); + return 2; + } + const result = await importForumPages({ + paths: getPaths(), + slug: opts.slug, + inputs: opts.inputs, + dryRun: opts.dryRun, + onLog: (line) => console.log(line), + }); + if (!result.ok) { + console.error(`posts import-html: ${result.error}`); + return 1; + } + for (const p of result.pages) { + if (p.skipped) console.log(`skipped ${p.file}: ${p.skipped}`); + } + return 0; +} diff --git a/common/components/PostModal.tsx b/common/components/PostModal.tsx @@ -14,7 +14,7 @@ import { XIcon } from "lucide-react"; import { Button } from "./ui/button"; import { useUrlParams, writeUrlParams } from "./urlState"; import { fetchPost, fetchThread } from "./postsCache"; -import type { Post } from "../lib/posts"; +import { postPlatformLabel, type Post } from "../lib/posts"; function formatWhen(iso: string): string { const ms = Date.parse(iso); @@ -30,7 +30,7 @@ function formatWhen(iso: string): string { } function platformLabel(platform: Post["platform"]): string { - return platform === "twitter" ? "X" : "Bluesky"; + return postPlatformLabel(platform); } // A post's capture (a screenshot and its attached media, taken by the @@ -185,7 +185,15 @@ function PostBody({ <span>·</span> <time dateTime={post.createdAt}>{formatWhen(post.createdAt)}</time> {post.isRepost && <Badge>repost</Badge>} - {post.isReply && <Badge>reply</Badge>} + {/* Every forum post after the first answers the thread; the badge + would be on all of them. Its place in the thread says more. */} + {post.isReply && !post.forum && <Badge>reply</Badge>} + {post.forum?.position && <Badge>#{post.forum.position}</Badge>} + {post.forum?.editedAt && ( + <span title={`Last edited ${formatWhen(post.forum.editedAt)}`}> + <Badge>edited</Badge> + </span> + )} {/* The whole point of archiving: this post no longer exists upstream. */} {post.isDeleted && ( <span @@ -207,6 +215,36 @@ function PostBody({ {capture && <PostCapturePanel capture={capture} />} + {post.forum?.threadTitle && !compact && ( + <p className="mt-1 text-xs text-muted-foreground"> + In thread: {post.forum.threadTitle} + {post.forum.page ? ` · page ${post.forum.page}` : ""} + </p> + )} + + {post.media && post.media.length > 0 && !capturedMedia && ( + <ul data-post-media="" className="mt-2 flex flex-col gap-1"> + {post.media.map((m) => ( + <li key={`${m.kind}:${m.url}`} className="text-xs text-muted-foreground"> + {m.kind} + {m.provider ? ` (${m.provider})` : ""}:{" "} + {/^https?:\/\//.test(m.url) ? ( + <a + href={m.url} + target="_blank" + rel="noopener noreferrer" + className="text-primary underline break-all" + > + {m.name || m.url} + </a> + ) : ( + <span className="break-all">{m.name || m.url}</span> + )} + </li> + ))} + </ul> + )} + {links.length > 0 && ( <ul className="mt-2 flex flex-col gap-1"> {links.map((href) => ( diff --git a/common/components/SearchResults.tsx b/common/components/SearchResults.tsx @@ -27,6 +27,7 @@ import { type DuplicateLookup, } from "./duplicatesCache"; import type { DuplicateVideoRef } from "../lib/duplicates"; +import { postPlatformLabel } from "../lib/posts"; import { splitId } from "./originId"; import { usePlayer } from "./PlayerProvider"; import { formatTimestamp } from "../lib/vtt"; @@ -962,7 +963,7 @@ function sectionScope( function PostBadge({ platform }: { platform: string }) { return ( <span className="shrink-0 text-[10px] uppercase tracking-wide font-medium px-1.5 py-0.5 rounded bg-muted text-muted-foreground self-center"> - {platform === "twitter" ? "X" : "Bluesky"} + {postPlatformLabel(platform)} </span> ); } diff --git a/common/components/postsCache.ts b/common/components/postsCache.ts @@ -5,7 +5,12 @@ // federating hub can hold several origins' posts at once without collisions. import { useQueries, useQuery } from "@tanstack/react-query"; -import type { ChannelPostsManifest, Post, PostsManifest } from "../lib/posts"; +import { + postConversation, + type ChannelPostsManifest, + type Post, + type PostsManifest, +} from "../lib/posts"; import { PromiseMap } from "../lib/archive/reader"; import { channelRef, readerFor } from "../lib/archive/readers"; import { makeId, splitId } from "./originId"; @@ -102,22 +107,20 @@ export async function fetchThread(id: string): Promise<Post[]> { const channelSlug = slug.slice(0, slashIdx); const post = await fetchPost(id); const threadId = post.threadId || post.id; + const forum = post.platform === "xenforo"; // A thread can straddle pages, so scan every page of the channel. Pages are - // cached, and a channel's page count is small (byte-capped shards). + // cached, and a channel's page count is small (byte-capped shards). A forum + // post's thread is its conversation (postConversation): the channel IS the + // forum thread, so every page is a candidate. const manifest = await fetchChannelPostsManifest(channelSlug, origin); - const thread: Post[] = []; + const candidates: Post[] = []; for (let i = 0; i < manifest.pageCount; i++) { for (const entry of await fetchPostsPage(channelSlug, i, origin)) { - if ((entry.threadId || entry.id) === threadId) thread.push(entry); + if (forum || (entry.threadId || entry.id) === threadId) candidates.push(entry); } } - thread.sort((a, b) => - a.createdAt === b.createdAt - ? a.id.localeCompare(b.id) - : a.createdAt.localeCompare(b.createdAt), - ); - return thread.length > 0 ? thread : [post]; + return postConversation(post, candidates); } export function usePostsManifest(origin = "") { diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts @@ -42,6 +42,7 @@ import "../social/blueskyFetcher"; import "../social/xGalleryDlFetcher"; import "../social/xPlaywrightFetcher"; import "../social/xNitterFetcher"; +import "../social/xenforoFetcher"; import { resolveXCookieSourceFor, type XLoginSettings, @@ -136,13 +137,27 @@ export async function capturePosts( const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot)); if (stray) return fail(strayIdsRefusal(slug, stray)); - // The archived text of each id, for the article links in it (X only). - let archived: Map<string, { text: string; links: string[] }> | undefined; - if (opts.articles !== false && fetcher.platform === "twitter") { + // The archived record of each id: its text and links for the article links + // in it (X), its URL and media for a forum post (the capture opens the one + // and downloads the other). + type Archived = { + text: string; + links: string[]; + url?: string; + media?: { kind: string; url: string; name?: string }[]; + }; + let archived: Map<string, Archived> | undefined; + const forum = fetcher.platform === "xenforo"; + if ((opts.articles !== false && fetcher.platform === "twitter") || forum) { const wanted = new Set(ids); archived = new Map(); for (const p of await readAllPosts(channelRoot)) { - if (wanted.has(p.id)) archived.set(p.id, { text: p.text, links: p.links ?? [] }); + if (!wanted.has(p.id)) continue; + archived.set(p.id, { + text: p.text, + links: p.links ?? [], + ...(forum ? { url: p.url, ...(p.media ? { media: p.media } : {}) } : {}), + }); } } @@ -169,6 +184,10 @@ export async function capturePosts( result = await fetcher.captureByIds({ ids, handle, + accountUrl, + ...(config.postPagePauseSeconds + ? { pagePauseMs: config.postPagePauseSeconds * 1000 } + : {}), outDir: postsMediaDir(channelRoot), shots: opts.shots, media: opts.media, @@ -222,7 +241,9 @@ export async function capturePosts( ok: false, outcomes: result.outcomes, needsCookies: true, - error: result.stoppedEarly ?? "The capture needs an X login.", + error: + result.stoppedEarly ?? + (forum ? "The capture needs the forum session (Connect)." : "The capture needs an X login."), }; } const stoppedByOperator = diff --git a/common/controller/checkPostAvailability.ts b/common/controller/checkPostAvailability.ts @@ -33,6 +33,7 @@ import { resolveSocialFetcher, } from "../social/fetchers"; import "../social/blueskyFetcher"; +import "../social/xenforoFetcher"; import "../social/xGalleryDlFetcher"; import "../social/xPlaywrightFetcher"; import "../social/xNitterFetcher"; diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts @@ -51,6 +51,8 @@ import "../social/xGalleryDlFetcher"; // only ever used when a channel opts into it via postFetcher: "x-playwright". import "../social/xPlaywrightFetcher"; import "../social/xNitterFetcher"; +// A forum thread (platform "xenforo"): claims only thread URLs. +import "../social/xenforoFetcher"; import type { XCookieSource } from "../social/xCookieSource"; import { resolveXCookieSourceFor, @@ -82,6 +84,9 @@ export type FetchPostsOptions = { // emptyAccountOlderProblem). Only with `older`. force?: boolean; limit?: number; + // Cap on pages read this run, for a fetcher that walks pages (a forum + // thread's "latest N pages"). + pages?: number; onLog?: (line: string) => void; signal?: AbortSignal; // The job's Drain: the fetch stops at its next resume point and keeps it, @@ -299,6 +304,10 @@ export async function fetchPosts( cookieSource: xLogin?.source, browserCookies: xLogin?.browserSpec, limit: opts.limit, + ...(opts.pages ? { pages: opts.pages } : {}), + ...(config.postPagePauseSeconds + ? { pagePauseMs: config.postPagePauseSeconds * 1000 } + : {}), stopAtKnown: !opts.full, onCheckpoint, signal: effectiveSignal, diff --git a/common/controller/importForumPages.test.ts b/common/controller/importForumPages.test.ts @@ -0,0 +1,116 @@ +// Importing browser-saved forum thread pages into a forum-thread channel, end +// to end over a temp corpus: the pages are parsed, posts merged (new ones +// appended, an edited one updated), and pages that are not this thread's are +// skipped by name. SYNTHETIC pages only. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/importForumPages.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "forum-import-")); +process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); + +const { getPaths } = await import("../lib/paths"); +const { importForumPages } = await import("./importForumPages"); +const { readAllPosts } = await import("../lib/posts-server"); +const { CHALLENGE_PAGE, THREAD_URL, threadPage } = await import("../social/__fixtures__/xenforoPages"); + +const T0 = 1_760_000_000; +const post = (id: number, position: number, body: string, editedTs?: number) => ({ + id, + author: `Member${id % 3}`, + userId: 10 + (id % 3), + ts: T0 + position * 60, + position, + body, + ...(editedTs ? { editedTs } : {}), +}); + +async function channel(slug: string, config: Record<string, unknown>) { + const dir = path.join(getPaths().channelsDir, slug); + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, "config.json"), JSON.stringify(config)); +} + +await channel("teapots", { + handling: "youtube", + sourceKind: "social", + platform: "xenforo", + postFetcher: "xenforo-thread", + url: THREAD_URL, + socialHandle: "the-teapot-collectors-thread.4242", +}); +await channel("someone-on-bluesky", { + handling: "youtube", + sourceKind: "social", + platform: "bluesky", + url: "https://bsky.app/profile/someone.example", +}); + +const SAVES = path.join(ROOT, "saves"); +await mkdir(path.join(SAVES, "The Teapot Collectors Thread_files"), { recursive: true }); +await writeFile( + path.join(SAVES, "The Teapot Collectors Thread _ Page 2.html"), + threadPage({ page: 2, last: 2, saved: true, posts: [post(21, 4, "Fourth."), post(22, 5, "Fifth.")] }), +); +await writeFile( + path.join(SAVES, "The Teapot Collectors Thread.html"), + threadPage({ page: 1, last: 2, saved: true, posts: [post(11, 1, "First."), post(12, 2, "Second."), post(13, 3, "Third.")] }), +); +await writeFile(path.join(SAVES, "check.html"), CHALLENGE_PAGE); +await writeFile( + path.join(SAVES, "other-thread.html"), + threadPage({ page: 1, last: 1, posts: [post(99, 1, "Elsewhere.")] }).replace(/4242/g, "5151"), +); +await writeFile(path.join(SAVES, "The Teapot Collectors Thread_files", "x.jpg"), "not a page"); + +test("imports a folder of saved pages: posts merged, strangers skipped by name", async () => { + const log: string[] = []; + const res = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [SAVES], onLog: (l) => log.push(l) }); + assert.equal(res.ok, true, res.error); + assert.equal(res.parsed, 5); + assert.equal(res.written, 5); + const byFile = Object.fromEntries(res.pages.map((p) => [p.file, p])); + assert.equal(byFile["check.html"].skipped?.startsWith("not a thread page (challenge"), true); + assert.match(byFile["other-thread.html"].skipped ?? "", /another thread \(5151, not 4242\)/); + assert.equal(byFile["The Teapot Collectors Thread _ Page 2.html"].page, 2); + assert.equal(res.pages.length, 4, "the _files folder is not walked"); + + const posts = await readAllPosts(path.join(getPaths().channelsDir, "teapots")); + assert.deepEqual(posts.map((p) => p.id), ["22", "21", "13", "12", "11"]); + assert.equal(posts[0].forum?.page, 2); + assert.equal(posts[0].channelSlug, "teapots"); +}); + +test("a later save of an edited post updates it; the same save again changes nothing", async () => { + const file = path.join(ROOT, "page-2-again.html"); + await writeFile( + file, + threadPage({ page: 2, last: 2, saved: true, posts: [post(21, 4, "Fourth, edited.", T0 + 5000), post(22, 5, "Fifth.")] }), + ); + const res = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [file] }); + assert.deepEqual([res.written, res.updated, res.unchanged], [0, 1, 1]); + const posts = await readAllPosts(path.join(getPaths().channelsDir, "teapots")); + assert.equal(posts.find((p) => p.id === "21")?.text, "Fourth, edited."); + assert.equal(posts.length, 5); + + const again = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [file] }); + assert.deepEqual([again.written, again.updated, again.unchanged], [0, 0, 2]); +}); + +test("a dry run writes nothing; refusals name the problem", async () => { + const dry = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [SAVES], dryRun: true }); + assert.equal(dry.ok, true); + assert.equal(dry.written, 0); + + const notForum = await importForumPages({ paths: getPaths(), slug: "someone-on-bluesky", inputs: [SAVES] }); + assert.match(notForum.error ?? "", /not a forum-thread channel/); + const missing = await importForumPages({ paths: getPaths(), slug: "teapots", inputs: [path.join(ROOT, "nope")] }); + assert.match(missing.error ?? "", /No such file or directory/); + const noChannel = await importForumPages({ paths: getPaths(), slug: "nobody", inputs: [SAVES] }); + assert.match(noChannel.error ?? "", /No such channel/); +}); diff --git a/common/controller/importForumPages.ts b/common/controller/importForumPages.ts @@ -0,0 +1,153 @@ +// Import forum thread pages the operator SAVED from their own browser ("Save +// page as", complete or HTML only) into a forum-thread channel. The second way +// in beside the live fetcher, through the same parser (xenforoParse.ts) and +// the same store (posts-server.ts): new posts are appended, a post already +// archived is updated when the save holds a newer record of it (an edit), and +// one that arrives unchanged is left alone. +// +// Server-only (node:fs). Nothing here touches the network: a saved page's +// assets are read from the page as it is, never fetched. + +import { readdir, readFile, stat } from "node:fs/promises"; +import path from "node:path"; +import { readChannelConfig } from "./channels"; +import { isSocialChannel } from "../lib/channelConfig"; +import type { Paths } from "../lib/paths"; +import type { Post } from "../lib/posts"; +import { upsertPosts } from "../lib/posts-server"; +import { + classifyForumPage, + parseXenforoThreadPage, + parseXenforoThreadUrl, +} from "../social/xenforoParse"; + +export type ImportForumPagesOptions = { + paths: Paths; + slug: string; + // Files and directories. A directory contributes the .html/.htm files in + // it (not the `<name>_files/` asset folders a browser saves beside them). + inputs: ReadonlyArray<string>; + // Parse and count, write nothing. + dryRun?: boolean; + onLog?: (line: string) => void; +}; + +export type ImportedPage = { + file: string; + page?: number; + posts: number; + // Why the file contributed nothing, when it did not. + skipped?: string; +}; + +export type ImportForumPagesResult = { + ok: boolean; + error?: string; + pages: ImportedPage[]; + parsed: number; + written: number; + updated: number; + unchanged: number; +}; + +const PAGE_EXT = /\.(html?|xhtml)$/i; + +// The page files the inputs name, sorted, each once. +export async function listSavedPages(inputs: ReadonlyArray<string>): Promise<string[]> { + const out = new Set<string>(); + for (const input of inputs) { + const abs = path.resolve(input); + const st = await stat(abs).catch(() => null); + if (!st) throw new Error(`No such file or directory: ${input}`); + if (st.isDirectory()) { + for (const e of await readdir(abs, { withFileTypes: true })) { + if (e.isFile() && PAGE_EXT.test(e.name)) out.add(path.join(abs, e.name)); + } + } else { + out.add(abs); + } + } + return [...out].sort(); +} + +export async function importForumPages(opts: ImportForumPagesOptions): Promise<ImportForumPagesResult> { + const log = (line: string) => opts.onLog?.(line); + const fail = (error: string): ImportForumPagesResult => ({ + ok: false, + error, + pages: [], + parsed: 0, + written: 0, + updated: 0, + unchanged: 0, + }); + const config = await readChannelConfig(opts.paths, opts.slug); + if (!config) return fail(`No such channel: ${opts.slug}`); + if (!isSocialChannel(config) || config.platform !== "xenforo") { + return fail(`${opts.slug} is not a forum-thread channel (platform "xenforo").`); + } + const thread = parseXenforoThreadUrl(config.url ?? ""); + if (!thread) return fail(`${opts.slug}'s URL is not a XenForo thread URL.`); + + let files: string[]; + try { + files = await listSavedPages(opts.inputs); + } catch (err) { + return fail((err as Error).message); + } + if (files.length === 0) return fail("No saved pages (.html) found in what was given."); + + const pages: ImportedPage[] = []; + const posts: Post[] = []; + for (const file of files) { + const name = path.basename(file); + let html: string; + try { + html = await readFile(file, "utf8"); + } catch (err) { + pages.push({ file: name, posts: 0, skipped: `unreadable: ${(err as Error).message}` }); + continue; + } + const block = classifyForumPage(html); + if (block) { + pages.push({ file: name, posts: 0, skipped: `not a thread page (${block.kind}: ${block.detail})` }); + continue; + } + const parsed = parseXenforoThreadPage(html, { channelSlug: opts.slug, pageUrl: thread.base }); + if (parsed.threadId && parsed.threadId !== thread.threadId) { + pages.push({ + file: name, + page: parsed.page, + posts: 0, + skipped: `a page of another thread (${parsed.threadId}, not ${thread.threadId})`, + }); + continue; + } + if (parsed.host && parsed.host !== thread.host) { + pages.push({ file: name, page: parsed.page, posts: 0, skipped: `a page of another forum (${parsed.host})` }); + continue; + } + pages.push({ file: name, page: parsed.page, posts: parsed.posts.length }); + posts.push(...parsed.posts); + log(`${name}: page ${parsed.page}/${parsed.lastPage}, ${parsed.posts.length} post(s).`); + } + + if (opts.dryRun) { + log(`Dry run: ${posts.length} post(s) parsed from ${files.length} file(s); nothing written.`); + return { ok: true, pages, parsed: posts.length, written: 0, updated: 0, unchanged: 0 }; + } + const channelRoot = path.join(opts.paths.channelsDir, opts.slug); + const res = await upsertPosts(channelRoot, posts); + log( + `Imported ${posts.length} post(s) from ${files.length} file(s): ${res.written} new, ` + + `${res.updated} updated, ${res.unchanged} unchanged.`, + ); + return { + ok: true, + pages, + parsed: posts.length, + written: res.written, + updated: res.updated, + unchanged: res.unchanged, + }; +} diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -523,6 +523,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: true, queueKeyStrategy: "platform", }, + // Forum thread pages the operator saved from a browser, imported into a + // forum-thread channel (controller/importForumPages.ts). Offline — nothing is + // fetched — but on the PLATFORM queue so it never writes the channel's posts + // beside a fetch writing the same shards. Not drainable (one parse and one + // write); not replayable (the files named are the operator's, and may be + // gone). + "import-forum-pages": { + kind: "import-forum-pages", + label: "Import saved forum pages", + drainable: false, + replayable: false, + queueKeyStrategy: "platform", + }, // A screenshot and the attached media of specific archived posts (and the X // Article a post links to), into the channel's `posts-media/` // (controller/capturePosts.ts). On the PLATFORM queue, like fetch-posts: diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -118,6 +118,7 @@ export type ChannelConfig = { sourceKind?: ChannelSourceKind; postFetcher?: string; socialHandle?: string; + postPagePauseSeconds?: number; platform?: Platform; name?: string; url?: string; @@ -157,6 +158,8 @@ export const CHANNEL_CONFIG_FIELD_DOCS: FieldDocs<ChannelConfig> = { 'Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed.', socialHandle: 'Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account.', + postPagePauseSeconds: + "Social channels that are read page by page (a forum thread) only: the pause between two page loads, in seconds; each pause is jittered to 0.85–1.65× of it. Absent = the fetcher's own (12 s, so 10–20 s); floored at 5, capped at 600.", platform: "The source platform (youtube, rumble, …). An unknown value is dropped.", name: "Display name.", url: "The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced.", @@ -363,6 +366,8 @@ export const CHANNEL_CONFIG_COERCIONS: { postFetcher: trimmedNonBlank, socialHandle: (v) => typeof v === "string" && v.trim() ? v.trim().replace(/^@/, "") : undefined, + postPagePauseSeconds: (v) => + isFiniteNumber(v) && v > 0 ? Math.min(Math.max(Math.floor(v), 5), 600) : undefined, platform: (v) => typeof v === "string" && PLATFORM_VALUES.includes(v as Platform) ? (v as Platform) diff --git a/common/lib/channelConfigSchema.ts b/common/lib/channelConfigSchema.ts @@ -47,6 +47,7 @@ export const channelConfigObjectSchema = z.object({ sourceKind: field("sourceKind"), postFetcher: field("postFetcher"), socialHandle: field("socialHandle"), + postPagePauseSeconds: field("postPagePauseSeconds"), platform: field("platform"), name: field("name"), url: field("url"), diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts @@ -94,7 +94,7 @@ const SHARD_SCHEME = { // timestamp fragment. const POST_SCHEME = { description: - "Social posts (X/Twitter, Bluesky) are archived as a PARALLEL corpus to " + + "Social posts (X/Twitter, Bluesky, forum threads) are archived as a PARALLEL corpus to " + "video transcripts and are served as paginated JSON shards under the same " + "scheme: (1) GET the channel's posts manifest; (2) look up the post id in " + "its `slugToPage` map to get a page number N; (3) GET page-<NNNN>.json and " + @@ -105,7 +105,11 @@ const POST_SCHEME = { postPage: "/posts/<slug>/page-<NNNN>.json -> array of { id, slug, channelSlug, author, " + "authorName, createdAt, uploadDate, text, url, platform, threadId, replyTo, " + - "quoted, repostOf, isReply, isRepost, links, mediaCount, engagement }", + "quoted, repostOf, isReply, isRepost, links, mediaCount, media, engagement, " + + "forum }; `forum` (platform \"xenforo\": one channel is one forum thread) " + + "carries { host, threadId, threadTitle, page, position, authorId, editedAt, " + + "quotes: [{ postId, author }] }, and quoted text in a forum post's `text` " + + "is marked with leading \"> \" lines", ordering: "newest first, by `createdAt` (ISO-8601, ms precision where the source provides it)", dateFilter: diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs @@ -8,6 +8,16 @@ /** @typedef {import("./platform").Platform} Platform */ +// Forums known to run XenForo, by host. Any other XenForo forum is recognised +// by its thread URL (THREAD_PATH_RE) instead — a host table is a convenience, +// not the rule. +export const XENFORO_HOSTS = ["kiwifarms.st", "kiwifarms.net"]; + +// A XenForo thread path: /threads/<slug>.<id>/ or /threads/<id>/, optionally +// under a prefix (/community/threads/…) or behind index.php? (non-friendly +// URLs). +export const XENFORO_THREAD_PATH_RE = /(?:^|\/|\?)threads\/(?:[^/?#]*\.)?(\d+)(?:\/|$)/; + /** * @param {string | undefined | null} url * @returns {Platform | null} @@ -24,6 +34,11 @@ export function detectPlatform(url) { if (host === "x.com" || host.endsWith(".x.com")) return "twitter"; if (host.endsWith("twitter.com")) return "twitter"; if (host === "bsky.app" || host.endsWith(".bsky.app")) return "bluesky"; + if (XENFORO_HOSTS.some((h) => host === h || host.endsWith(`.${h}`))) { + return "xenforo"; + } + const u = new URL(url); + if (XENFORO_THREAD_PATH_RE.test(u.pathname + u.search)) return "xenforo"; } catch { /* fall through */ } diff --git a/common/lib/forumPosts.test.ts b/common/lib/forumPosts.test.ts @@ -0,0 +1,147 @@ +// Forum posts (platform "xenforo") in the model and the store: detection, the +// stored-record validator, the conversation a forum post belongs to, and +// upsertPosts — the import's write, which updates an edited post in place +// (append-only, last record wins). SYNTHETIC posts only. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/forumPosts.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { detectPlatform, isSocialPlatform } from "./platform"; +import { + isPostPlatform, + parsePost, + postConversation, + postPermalink, + postPlatformLabel, + uploadDateFromCreatedAt, + type Post, +} from "./posts"; +import { readAllPosts, readSeenPostIds, supersedesPost, upsertPosts } from "./posts-server"; +import { handleFromAccountUrl } from "../social/fetchers"; + +function forumPost(id: string, position: number, extra: Partial<Post> = {}): Post { + const createdAt = new Date(Date.UTC(2026, 0, 1, 0, position)).toISOString(); + return { + id, + slug: `teapots/${id}`, + channelSlug: "teapots", + author: `Member${position}`, + createdAt, + uploadDate: uploadDateFromCreatedAt(createdAt), + text: `Words of post ${position}.`, + url: `https://forum.example/posts/${id}/`, + platform: "xenforo", + isReply: position > 1, + isRepost: false, + links: [], + forum: { host: "forum.example", threadId: "4242", position, page: 1 }, + ...extra, + }; +} + +test("detection: forum thread URLs are the xenforo platform, a social one", () => { + assert.equal(detectPlatform("https://forum.example/threads/a-title.4242/"), "xenforo"); + assert.equal(detectPlatform("https://forum.example/threads/a-title.4242/page-9"), "xenforo"); + assert.equal(detectPlatform("https://forum.example/index.php?threads/a-title.4242/"), "xenforo"); + // Kiwi Farms is known by host (it runs XenForo). + assert.equal(detectPlatform("https://kiwifarms.st/members/someone.1/"), "xenforo"); + assert.equal(detectPlatform("https://forum.example/members/someone.1/"), null); + assert.equal(detectPlatform("https://www.youtube.com/@x/threads"), "youtube"); + assert.equal(isSocialPlatform("xenforo"), true); + assert.equal(isPostPlatform("xenforo"), true); + assert.equal(postPlatformLabel("xenforo"), "Forum"); + assert.equal(handleFromAccountUrl("https://forum.example/threads/a-title.4242/page-3"), "a-title.4242"); + assert.equal(postPermalink("xenforo", "", "77", "https://forum.example/"), "https://forum.example/posts/77/"); +}); + +test("the validator keeps forum facts and media, and drops what is malformed", () => { + const p = forumPost("10", 3, { + media: [ + { kind: "image", url: "https://images.example/a.jpg", name: "a.jpg" }, + { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK", provider: "youtube" }, + ], + mediaCount: 2, + forum: { + host: "forum.example", + threadId: "4242", + threadTitle: "Teapots", + position: 3, + page: 1, + authorId: "9", + editedAt: "2026-01-02T00:00:00.000Z", + quotes: [{ postId: "8", author: "Member1" }], + }, + }); + assert.deepEqual(parsePost(JSON.parse(JSON.stringify(p))), p); + const bad = parsePost({ + ...p, + media: [{ kind: "sticker", url: "x" }, { kind: "image" }], + forum: { host: "forum.example" }, + }); + assert.equal(bad?.media, undefined); + assert.equal(bad?.forum, undefined); + // A forum post with no stored url gets its permalink from the forum host. + const { url: _url, ...noUrl } = p; + assert.equal(parsePost(noUrl)?.url, "https://forum.example/posts/10/"); +}); + +test("a forum post's conversation is its quote graph, in thread order", () => { + const a = forumPost("1", 1); + const b = forumPost("2", 2, { forum: { host: "forum.example", threadId: "4242", position: 2, quotes: [{ postId: "1" }] } }); + const c = forumPost("3", 3, { forum: { host: "forum.example", threadId: "4242", position: 3, quotes: [{ postId: "2" }] } }); + const d = forumPost("4", 4, { forum: { host: "forum.example", threadId: "4242", position: 4, quotes: [{ postId: "3" }] } }); + const unrelated = forumPost("5", 5); + const all = [unrelated, d, c, b, a]; + assert.deepEqual(postConversation(c, all).map((p) => p.id), ["1", "2", "3", "4"]); + assert.deepEqual(postConversation(unrelated, all).map((p) => p.id), ["5"]); + assert.deepEqual(postConversation(c, all, 1).map((p) => p.id), ["2", "3", "4"]); +}); + +test("a non-forum post's conversation is still its reply thread", () => { + const base = { ...forumPost("x", 1), platform: "bluesky" as const, forum: undefined }; + const root = { ...base, id: "r", threadId: "r", createdAt: "2026-01-01T00:00:00.000Z" }; + const reply = { ...base, id: "s", threadId: "r", createdAt: "2026-01-01T00:01:00.000Z" }; + const other = { ...base, id: "t", threadId: "t" }; + assert.deepEqual(postConversation(reply, [other, reply, root]).map((p) => p.id), ["r", "s"]); +}); + +test("supersedesPost: a later edit wins, an older one never does", () => { + const stored = forumPost("1", 1, { forum: { host: "h", threadId: "1", editedAt: "2026-02-01T00:00:00.000Z" } }); + const newer = { ...stored, text: "changed", forum: { ...stored.forum!, editedAt: "2026-03-01T00:00:00.000Z" } }; + const older = { ...stored, text: "older words", forum: { ...stored.forum!, editedAt: "2026-01-15T00:00:00.000Z" } }; + const unedited = { ...stored, text: "never edited", forum: { host: "h", threadId: "1" } }; + assert.equal(supersedesPost(stored, { ...stored }), false); + assert.equal(supersedesPost(stored, newer), true); + assert.equal(supersedesPost(stored, older), false); + assert.equal(supersedesPost(stored, unedited), false); + const plain = forumPost("2", 2); + assert.equal(supersedesPost(plain, { ...plain, text: "re-read" }), true); +}); + +test("upsertPosts: new posts appended, an edit updates the post, a repeat changes nothing", async () => { + const root = await mkdtemp(path.join(tmpdir(), "forum-upsert-")); + const first = await upsertPosts(root, [forumPost("1", 1), forumPost("2", 2)]); + assert.deepEqual([first.written, first.updated, first.unchanged], [2, 0, 0]); + + const edited = forumPost("2", 2, { + text: "Words of post 2, edited.", + forum: { host: "forum.example", threadId: "4242", position: 2, page: 1, editedAt: "2026-01-05T00:00:00.000Z" }, + }); + const second = await upsertPosts(root, [forumPost("1", 1), edited, forumPost("3", 3)]); + assert.deepEqual([second.written, second.updated, second.unchanged], [1, 1, 1]); + + const all = await readAllPosts(root); + assert.deepEqual(all.map((p) => p.id), ["3", "2", "1"]); + assert.equal(all.find((p) => p.id === "2")?.text, "Words of post 2, edited."); + assert.equal((await readSeenPostIds(root)).size, 3); + // The archive lists each id once: an update is not a new post. + const archive = await readFile(path.join(root, "posts-archive"), "utf8"); + assert.equal(archive.trim().split("\n").length, 3); + + const third = await upsertPosts(root, [edited]); + assert.deepEqual([third.written, third.updated, third.unchanged], [0, 0, 1]); +}); diff --git a/common/lib/platform.ts b/common/lib/platform.ts @@ -9,7 +9,8 @@ export type Platform = | "twitch" | "kick" | "twitter" - | "bluesky"; + | "bluesky" + | "xenforo"; export const PLATFORM_VALUES: ReadonlyArray<Platform> = [ "youtube", @@ -19,6 +20,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [ "kick", "twitter", "bluesky", + "xenforo", ]; // The subset that carries social posts rather than videos. A channel on one of @@ -27,10 +29,11 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [ export const SOCIAL_PLATFORM_VALUES: ReadonlyArray<Platform> = [ "twitter", "bluesky", + "xenforo", ]; export function isSocialPlatform(platform: Platform | null | undefined): boolean { - return platform === "twitter" || platform === "bluesky"; + return platform === "twitter" || platform === "bluesky" || platform === "xenforo"; } // Host → platform. Lives in `detectPlatform.mjs` (plain JS so umtool's `.mjs` @@ -54,6 +57,10 @@ export function defaultWebpageUrl(platform: Platform, id: string): string { // every archived post carries its own canonical `url` (see postPermalink). if (platform === "twitter") return `https://x.com/i/status/${id}`; if (platform === "bluesky") return `https://bsky.app/profile/${id}`; + // A forum post's URL needs its forum's host, which an id alone does not + // carry; every archived forum post has its own `url`. A thread URL passed as + // the id is returned as it is. + if (platform === "xenforo") return /^https?:\/\//.test(id) ? id : ""; return `https://www.youtube.com/watch?v=${id}`; } diff --git a/common/lib/posts-server.ts b/common/lib/posts-server.ts @@ -129,6 +129,96 @@ export async function writePosts( }; } +export type UpsertPostsResult = WritePostsResult & { + // Archived posts whose stored record was superseded by a newer one. + updated: number; + // Archived posts that arrived again unchanged (or older than what is + // stored). + unchanged: number; +}; + +// What makes two records of one post differ, for an update. +function postContentKey(p: Post): string { + return JSON.stringify([ + p.text, + p.author, + p.links, + p.media ?? null, + p.forum?.editedAt ?? null, + p.forum?.position ?? null, + p.forum?.threadTitle ?? null, + ]); +} + +// Is `incoming` a newer record of the same post than `stored`? A later edit +// time wins; an older one never replaces a newer one; with no edit times to +// compare (or equal ones), different content is taken as the later read. +export function supersedesPost(stored: Post, incoming: Post): boolean { + if (postContentKey(stored) === postContentKey(incoming)) return false; + const a = stored.forum?.editedAt; + const b = incoming.forum?.editedAt; + if (a && b && a !== b) return b > a; + if (a && !b) return false; + return true; +} + +// writePosts, plus UPDATES: a post already archived whose incoming record is +// newer (supersedesPost — an edit made since it was archived) is appended +// again to its month shard. The JSONL stays append-only; every reader takes the +// LAST record of an id (readAllPosts, the index build), so the newer one wins. +// For sources that re-read posts they already hold (a forum page saved twice). +export async function upsertPosts( + channelRoot: string, + posts: ReadonlyArray<Post>, +): Promise<UpsertPostsResult> { + // One record per id from the batch: the newest of them. + const batch = new Map<string, Post>(); + for (const p of posts) { + const prev = batch.get(p.id); + if (!prev || supersedesPost(prev, p)) batch.set(p.id, p); + } + const seen = await readSeenPostIds(channelRoot); + const fresh = [...batch.values()].filter((p) => !seen.has(p.id)); + const known = [...batch.values()].filter((p) => seen.has(p.id)); + const written = await writePosts(channelRoot, fresh); + let updated = 0; + let unchanged = 0; + const shards = new Set(written.shards); + if (known.length > 0) { + const stored = new Map<string, Post>(); + for (const p of await readAllPosts(channelRoot)) stored.set(p.id, p); + const byShard = new Map<string, Post[]>(); + for (const p of known) { + const old = stored.get(p.id); + if (old && !supersedesPost(old, p)) { + unchanged++; + continue; + } + // Keep the shard the post already lives in: its createdAt does not + // change with an edit. + const shard = monthShardFromCreatedAt(old?.createdAt ?? p.createdAt); + const bucket = byShard.get(shard); + const rec = old ? { ...p, createdAt: old.createdAt, uploadDate: old.uploadDate } : p; + if (bucket) bucket.push(rec); + else byShard.set(shard, [rec]); + updated++; + } + if (byShard.size > 0) await mkdir(channelPostsDir(channelRoot), { recursive: true }); + for (const [shard, shardPosts] of byShard) { + const body = shardPosts.map((p) => JSON.stringify(p)).join("\n") + "\n"; + await appendFile(shardPath(channelRoot, shard), body, "utf8"); + shards.add(shard); + } + } + return { + written: written.written, + skipped: written.skipped, + shards: [...shards].sort(), + updated, + unchanged, + }; +} + // List the channel's month shards, oldest first. export async function listPostShards( channelRoot: string, diff --git a/common/lib/posts.ts b/common/lib/posts.ts @@ -10,17 +10,71 @@ import { pageFileName } from "./manifest"; -export type PostPlatform = "twitter" | "bluesky"; +// "xenforo" is a forum THREAD read as a posts source: the channel is one +// thread, each forum post a Post. It is named for the forum software, not a +// host — which forum a thread is on is its URL (and `Post.forum.host`). +export type PostPlatform = "twitter" | "bluesky" | "xenforo"; export const POST_PLATFORM_VALUES: ReadonlyArray<PostPlatform> = [ "twitter", "bluesky", + "xenforo", ]; export function isPostPlatform(v: unknown): v is PostPlatform { - return v === "twitter" || v === "bluesky"; + return v === "twitter" || v === "bluesky" || v === "xenforo"; } +// The platform's name as a reader sees it. +export function postPlatformLabel(platform: PostPlatform | string | undefined): string { + if (platform === "twitter") return "X"; + if (platform === "bluesky") return "Bluesky"; + if (platform === "xenforo") return "Forum"; + return "Post"; +} + +// Media a post carries, as LINKS (the archive keeps the URLs, not the bytes — +// a post capture downloads them). Recorded by the forum parser; the X and +// Bluesky normalizers count media instead (`mediaCount`). +export type PostMediaKind = "image" | "video" | "attachment" | "embed" | "link-card"; + +export const POST_MEDIA_KINDS: ReadonlyArray<PostMediaKind> = [ + "image", + "video", + "attachment", + "embed", + "link-card", +]; + +export type PostMedia = { + kind: PostMediaKind; + url: string; + // A file name (an attachment's), or an image's alt text. + name?: string; + // An embed's provider ("youtube", "twitter", …), as the forum named it. + provider?: string; +}; + +// What a forum post carries beyond the common shape (platform "xenforo"). +export type ForumPostInfo = { + host: string; + // The thread's numeric id, and its title and canonical URL when read. + threadId: string; + threadTitle?: string; + threadUrl?: string; + // The thread page the post was read on, and its position in the thread + // (the "#N" the forum shows). + page?: number; + position?: number; + // The author's numeric member id. + authorId?: string; + // When the forum says the post was last edited (ISO-8601). + editedAt?: string; + // The posts this one quotes, in order: the quoted post's id and author + // when the quote names them. + quotes?: { postId?: string; author?: string; authorId?: string }[]; +}; + // A pointer to another post, which may or may not itself be archived. Used for // reply/quote/repost edges so thread structure survives even when the // referenced post is outside the archived account. @@ -68,6 +122,10 @@ export type Post = { isRepost: boolean; links: string[]; // expanded outbound urls mediaCount?: number; // counted, not archived (v1) + // The media's URLs, where the source gives them (forum posts). + media?: PostMedia[]; + // Forum-specific facts (platform "xenforo"). + forum?: ForumPostInfo; engagement?: PostEngagement; // Merged in at index time from the channel's availability sidecar (the same // shape videos use: the stored record is the source of truth, the flag on @@ -209,15 +267,21 @@ export function parsePostSlug( // Canonical permalink for a post. Bluesky needs the handle (its URLs are // /profile/<handle>/post/<rkey>); X only needs the numeric id but includes the -// handle for readability. +// handle for readability. A forum post needs its forum's origin (`base`, +// "https://forum.example"): XenForo's /posts/<id>/ resolves to the post in its +// thread wherever the thread has moved. export function postPermalink( platform: PostPlatform, author: string, id: string, + base?: string, ): string { if (platform === "bluesky") { return `https://bsky.app/profile/${author}/post/${id}`; } + if (platform === "xenforo") { + return `${(base ?? "").replace(/\/+$/, "")}/posts/${id}/`; + } return `https://x.com/${author || "i"}/status/${id}`; } @@ -251,7 +315,12 @@ export function parsePost(raw: unknown): Post | null { url: typeof r.url === "string" && r.url ? r.url - : postPermalink(r.platform, author, r.id), + : postPermalink( + r.platform, + author, + r.id, + r.platform === "xenforo" ? forumBaseOf(r.forum) : undefined, + ), platform: r.platform, isReply: r.isReply === true, isRepost: r.isRepost === true, @@ -279,9 +348,77 @@ export function parsePost(raw: unknown): Post | null { } const engagement = parseEngagement(r.engagement); if (engagement) post.engagement = engagement; + const media = parseMediaList(r.media); + if (media) post.media = media; + const forum = parseForumInfo(r.forum); + if (forum) post.forum = forum; return post; } +function forumBaseOf(raw: unknown): string | undefined { + const host = (raw as { host?: unknown } | null)?.host; + return typeof host === "string" && host ? `https://${host}` : undefined; +} + +function parseMediaList(raw: unknown): PostMedia[] | null { + if (!Array.isArray(raw)) return null; + const out: PostMedia[] = []; + for (const item of raw) { + if (!item || typeof item !== "object") continue; + const m = item as Record<string, unknown>; + if (typeof m.url !== "string" || !m.url) continue; + if (!(POST_MEDIA_KINDS as string[]).includes(m.kind as string)) continue; + const media: PostMedia = { kind: m.kind as PostMediaKind, url: m.url }; + if (typeof m.name === "string" && m.name) media.name = m.name; + if (typeof m.provider === "string" && m.provider) media.provider = m.provider; + out.push(media); + } + return out.length > 0 ? out : null; +} + +const posInt = (v: unknown): number | undefined => + typeof v === "number" && Number.isFinite(v) && v >= 1 ? Math.floor(v) : undefined; +const nonBlank = (v: unknown): string | undefined => + typeof v === "string" && v ? v : undefined; + +function parseForumInfo(raw: unknown): ForumPostInfo | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + const host = nonBlank(r.host); + const threadId = nonBlank(r.threadId); + if (!host || !threadId) return null; + const out: ForumPostInfo = { host, threadId }; + const threadTitle = nonBlank(r.threadTitle); + if (threadTitle) out.threadTitle = threadTitle; + const threadUrl = nonBlank(r.threadUrl); + if (threadUrl) out.threadUrl = threadUrl; + const page = posInt(r.page); + if (page) out.page = page; + const position = posInt(r.position); + if (position) out.position = position; + const authorId = nonBlank(r.authorId); + if (authorId) out.authorId = authorId; + const editedAt = nonBlank(r.editedAt); + if (editedAt) out.editedAt = editedAt; + if (Array.isArray(r.quotes)) { + const quotes: NonNullable<ForumPostInfo["quotes"]> = []; + for (const q of r.quotes) { + if (!q || typeof q !== "object") continue; + const qq = q as Record<string, unknown>; + const one: { postId?: string; author?: string; authorId?: string } = {}; + const postId = nonBlank(qq.postId); + if (postId) one.postId = postId; + const author = nonBlank(qq.author); + if (author) one.author = author; + const authorId = nonBlank(qq.authorId); + if (authorId) one.authorId = authorId; + quotes.push(one); + } + if (quotes.length > 0) out.quotes = quotes; + } + return out; +} + function parsePostRef(raw: unknown): PostRef | null { if (!raw || typeof raw !== "object") return null; const r = raw as Record<string, unknown>; @@ -316,6 +453,56 @@ export function comparePostsNewestFirst(a: Post, b: Post): number { return a.id < b.id ? 1 : a.id > b.id ? -1 : 0; } +// The CONVERSATION around a post, oldest first: the "more context" unit the +// viewer's thread view and the MCP get_thread tool show. +// +// For X and Bluesky that is the reply thread (threadId). A forum thread is one +// channel of possibly thousands of posts, so a forum post's conversation is +// the quote graph instead: the posts it quotes (and theirs, to `depth`), and +// the posts that quote it — read in thread order. +export function postConversation( + post: Post, + candidates: ReadonlyArray<Post>, + depth = 3, +): Post[] { + if (post.platform !== "xenforo") { + const threadId = post.threadId || post.id; + const thread = candidates.filter((p) => (p.threadId || p.id) === threadId); + if (!thread.some((p) => p.id === post.id)) thread.push(post); + return thread.sort((a, b) => -comparePostsNewestFirst(a, b)); + } + const byId = new Map<string, Post>(); + for (const p of candidates) byId.set(p.id, p); + byId.set(post.id, post); + const keep = new Map<string, Post>([[post.id, post]]); + let frontier: Post[] = [post]; + for (let d = 0; d < depth && frontier.length > 0; d++) { + const next: Post[] = []; + for (const p of frontier) { + for (const q of p.forum?.quotes ?? []) { + const hit = q.postId ? byId.get(q.postId) : undefined; + if (hit && !keep.has(hit.id)) { + keep.set(hit.id, hit); + next.push(hit); + } + } + } + frontier = next; + } + for (const p of candidates) { + if (p.forum?.quotes?.some((q) => q.postId === post.id)) keep.set(p.id, p); + } + return [...keep.values()].sort(compareForumOrder); +} + +// Thread order for forum posts: position when both have one, else time. +function compareForumOrder(a: Post, b: Post): number { + const pa = a.forum?.position; + const pb = b.forum?.position; + if (pa && pb && pa !== pb) return pa - pb; + return -comparePostsNewestFirst(a, b); +} + // Group a flat post list into threads keyed by threadId (falling back to the // post's own id for a root/standalone post). Used by the viewer's thread // context and the MCP get_thread tool. diff --git a/common/lib/report/convert-server.ts b/common/lib/report/convert-server.ts @@ -112,7 +112,7 @@ export function diskPostOf(channelsDir: string): PostOf { const post = (await p).get(id); if (!post) return null; const rec: PostRecord = { - platform: post.platform === "bluesky" ? "bluesky" : "x", + platform: post.platform === "bluesky" ? "bluesky" : post.platform === "xenforo" ? "forum" : "x", url: post.url, createdAt: post.createdAt, author: post.author, diff --git a/common/lib/report/convertManifest.ts b/common/lib/report/convertManifest.ts @@ -355,7 +355,9 @@ export async function manifestToReport(manifest: unknown, opts: ManifestConvertO // What a post record says that a post citation does not (the CLI reads it off // the channel's archive). export type PostRecord = { - platform: "x" | "bluesky"; + // "forum": a forum-thread post (platform "xenforo"), which a video manifest + // does not carry yet — it is left out with a warning, never relabelled. + platform: "x" | "bluesky" | "forum"; url: string; createdAt?: string; author?: string; @@ -489,6 +491,10 @@ export async function reportToManifest(raw: unknown, opts: ManifestOptions = {}) if (postsDone.has(cid)) return; postsDone.add(cid); const rec = opts.postOf ? await opts.postOf(c.channel, c.id) : null; + if (rec?.platform === "forum") { + warnings.push(`${cid}: post ${c.channel}/${c.id} is a forum post, which a video manifest does not carry — left out of posts`); + return; + } const platform = rec?.platform ?? (/^\d+$/.test(c.id) ? "x" : null); const url = rec?.url ?? (platform === "x" ? `https://x.com/i/status/${c.id}` : null); const date = c.date ?? rec?.createdAt; diff --git a/common/social/__fixtures__/xenforoPages.ts b/common/social/__fixtures__/xenforoPages.ts @@ -0,0 +1,222 @@ +// SYNTHETIC XenForo 2 thread pages for the forum-thread tests. The forum, the +// thread, the members and every word are invented; the markup follows the +// shape XenForo 2 serves (article.message, .bbWrapper, the attribution header, +// the page nav), and one variant is shaped the way a browser's "Save page as" +// rewrites a page. + +export const ORIGIN = "https://forum.example"; +export const THREAD_PATH = "/threads/the-teapot-collectors-thread.4242/"; +export const THREAD_URL = `${ORIGIN}${THREAD_PATH}`; + +export type FakePost = { + id: number; + author: string; + userId: number; + // Epoch seconds. + ts: number; + position: number; + body: string; + editedTs?: number; + attachments?: { href: string; name: string }[]; +}; + +export function message(p: FakePost, opts: { savedAssets?: boolean; article?: boolean } = {}): string { + const iso = new Date(p.ts * 1000).toISOString().replace(".000Z", "+0000"); + const avatar = opts.savedAssets + ? `./The Teapot Collectors Thread_files/${p.userId}.jpg` + : `/data/avatars/m/0/${p.userId}.jpg?1700000000`; + const attachments = p.attachments?.length + ? `<section class="message-attachments"> + <h4 class="block-textHeader">Attachments</h4> + <ul class="attachmentList">${p.attachments + .map( + (a) => `<li class="file file--linked"> + <a class="u-anchorTarget" id="attachment-${p.id}"></a> + <a class="file-preview js-lbImage" href="${a.href}" target="_blank"> + <img src="${opts.savedAssets ? "./The Teapot Collectors Thread_files/thumb.jpg" : "/data/attachments/thumb.jpg"}" alt="${a.name}" width="200" loading="lazy"> + </a> + <div class="file-content"><div class="file-info"> + <span class="file-name" title="${a.name}">${a.name}</span> + <div class="file-meta">50 KB · Views: 3</div> + </div></div> + </li>`, + ) + .join("")}</ul> + </section>` + : ""; + const edited = p.editedTs + ? `<div class="message-lastEdit">Last edited: <time class="u-dt" dir="auto" datetime="${new Date(p.editedTs * 1000).toISOString().replace(".000Z", "+0000")}" data-timestamp="${p.editedTs}" data-date-string="x" data-time-string="y" title="x">x</time></div>` + : ""; + return ` +<article class="message ${opts.article ? "message--article" : "message--post"} js-post js-inlineModContainer " data-author="${p.author}" data-content="post-${p.id}" id="js-post-${p.id}" itemscope itemtype="https://schema.org/Comment" itemid="${ORIGIN}/posts/${p.id}/"> + <span class="u-anchorTarget" id="post-${p.id}"></span> + <div class="message-inner"> + <div class="message-cell message-cell--user"> + <section class="message-user" itemprop="author" itemscope itemtype="https://schema.org/Person" itemid="${ORIGIN}/members/${p.author.toLowerCase()}.${p.userId}/"> + <div class="message-avatar"><div class="message-avatar-wrapper"> + <a href="/members/${p.author.toLowerCase()}.${p.userId}/" class="avatar avatar--m" data-user-id="${p.userId}" data-xf-init="member-tooltip"> + <img src="${avatar}" alt="${p.author}" class="avatar-u${p.userId}-m" width="96" height="96" loading="lazy" itemprop="image"> + </a> + </div></div> + <div class="message-userDetails"> + <h4 class="message-name"><a href="/members/${p.author.toLowerCase()}.${p.userId}/" class="username " dir="auto" data-user-id="${p.userId}" data-xf-init="member-tooltip"><span itemprop="name">${p.author}</span></a></h4> + <h5 class="userTitle message-userTitle" dir="auto" itemprop="jobTitle">Collector</h5> + </div> + </section> + </div> + <div class="message-cell message-cell--main"> + <div class="message-main js-quickEditTarget"> + <header class="message-attribution message-attribution--split"> + <ul class="message-attribution-main listInline "> + <li class="u-concealed"> + <a href="${THREAD_PATH}post-${p.id}" rel="nofollow" itemprop="url"> + <time class="u-dt" dir="auto" datetime="${iso}" data-timestamp="${p.ts}" data-date-string="d" data-time-string="t" title="t" itemprop="datePublished">d</time> + </a> + </li> + </ul> + <ul class="message-attribution-opposite message-attribution-opposite--list "> + <li><a href="${THREAD_PATH}post-${p.id}" class="message-attribution-gadget" data-xf-init="share-tooltip" rel="nofollow"><i class="fa--xf fal fa-share-alt"><svg><use href="#share-alt"></use></svg></i></a></li> + <li><a href="${THREAD_PATH}post-${p.id}" class="message-attribution-gadget bookmarkLink" rel="nofollow"><span class="js-bookmarkText u-srOnly">Add bookmark</span></a></li> + ${opts.article ? "" : `<li> + <a href="${THREAD_PATH}post-${p.id}" rel="nofollow"> + #${p.position.toLocaleString("en-US")} + </a> + </li>`} + </ul> + </header> + <div class="message-content js-messageContent"> + <div class="message-userContent lbContainer js-lbContainer " data-lb-id="post-${p.id}"> + <article class="message-body js-selectToQuote"> + <div itemprop="text"> + <div class="bbWrapper">${p.body}</div> + </div> + <div class="js-selectToQuoteEnd">&nbsp;</div> + </article> + ${attachments} + </div> + ${edited} + </div> + <footer class="message-footer"> + <div class="message-actionBar actionBar"><div class="actionBar-set actionBar-set--external"> + <a href="${THREAD_PATH}post-${p.id}/react" class="reaction actionBar-action">Like</a> + </div></div> + <div class="reactionsBar js-reactionsList is-active"><a class="reactionsBar-link" href="/posts/${p.id}/reactions">Someone and 2 others</a></div> + </footer> + </div> + </div> + </div> +</article>`; +} + +export function pageNav(page: number, last: number): string { + if (last <= 1) return ""; + const items: string[] = []; + for (const n of [1, page - 1, page, page + 1, last]) { + if (n < 1 || n > last || items.some((i) => i.includes(`>${n}<`))) continue; + items.push( + `<li class="pageNav-page ${n === page ? "pageNav-page--current " : ""}"><a href="${THREAD_PATH}${n === 1 ? "" : `page-${n}`}">${n}</a></li>`, + ); + } + return `<nav class="pageNavWrapper pageNavWrapper--mixed"> + <div class="pageNav"> + <ul class="pageNav-main">${items.join("")}</ul> + </div> + <div class="pageNavSimple"> + <a class="pageNavSimple-el pageNavSimple-el--current" data-xf-init="tooltip" title="Go to page">${page} of ${last}</a> + </div> + <div class="js-pageJumpPage-wrap"><input type="number" class="input input--number js-pageJumpPage" value="${page}" min="1" max="${last}" step="1" required="required" data-menu-autofocus="true"></div> + </nav>`; +} + +export function threadPage(opts: { + page: number; + last: number; + posts: FakePost[]; + saved?: boolean; + title?: string; + // An article thread: this post (its first) shown atop the page. + article?: FakePost; +}): string { + const canonical = `${ORIGIN}${THREAD_PATH}${opts.page > 1 ? `page-${opts.page}` : ""}`; + const savedComment = opts.saved ? `<!-- saved from url=(0068)${canonical} -->\n` : ""; + // A browser's save keeps the canonical link; the variant drops it so the + // saved-from comment is what names the page. + const canonicalLink = opts.saved ? "" : `<link rel="canonical" href="${canonical}" />`; + return `<!DOCTYPE html> +${savedComment}<html id="XF" lang="en-US" dir="LTR" data-xf="2.2" data-app="public" data-template="thread_view" data-container-key="node-7" data-content-key="thread-4242" data-logged-in="false"> +<head> + <meta charset="utf-8" /> + <title>${opts.title ?? "The Teapot Collectors Thread"}${opts.page > 1 ? ` | Page ${opts.page}` : ""} | Example Forum</title> + ${canonicalLink} + <meta property="og:title" content="${opts.title ?? "The Teapot Collectors Thread"}" /> + <script>window.XF = {}; if (1 < 2) { var x = "</div>"; }</script> + <style>.message { color: red; }</style> +</head> +<body data-template="thread_view"> + <div class="p-body-header"> + <div class="p-title "><h1 class="p-title-value"><span class="label label--blue" dir="auto">Hobby</span><span class="label-append">&nbsp;</span>${opts.title ?? "The Teapot Collectors Thread"}</h1></div> + </div> + ${pageNav(opts.page, opts.last)} + <div class="block block--messages" data-xf-init="lightbox select-to-quote"> + <div class="block-body js-replyNewMessageContainer"> + ${opts.article ? message(opts.article, { savedAssets: opts.saved, article: true }) : ""} + ${opts.posts.map((p) => message(p, { savedAssets: opts.saved })).join("\n")} + </div> + </div> + ${pageNav(opts.page, opts.last)} +</body> +</html>`; +} + +// A browser check page of the KiwiFlare kind: no posts, a proof-of-work script +// and its marker. +export const CHALLENGE_PAGE = `<!DOCTYPE html> +<html><head><title>Checking your browser</title> +<script src="/.sssg/api/challenge.js"></script></head> +<body><div id="sssg-root">Please wait while your browser solves a proof-of-work challenge.</div> +<noscript>This check needs JavaScript.</noscript></body></html>`; + +export const CAPTCHA_PAGE = `<!DOCTYPE html> +<html><head><title>One more step</title></head> +<body><div class="h-captcha" data-sitekey="00000000-0000-0000-0000-000000000000"></div></body></html>`; + +export const LOGIN_PAGE = `<!DOCTYPE html> +<html data-template="login"><head><title>Log in | Example Forum</title></head> +<body><div class="blockMessage">You must be logged-in to do that.</div></body></html>`; + +export const NOT_FOUND_PAGE = `<!DOCTYPE html> +<html data-template="error"><head><title>Oops! We ran into some problems. | Example Forum</title></head> +<body><div class="blockMessage">The requested thread could not be found.</div></body></html>`; + +// A body exercising everything the body reader handles. +export const RICH_BODY = ` +<blockquote data-attributes="member: 7" data-quote="Marigold" data-source="post: 1001" class="bbCodeBlock bbCodeBlock--expandable bbCodeBlock--quote js-expandWatch"> + <div class="bbCodeBlock-title"><a href="/goto/post?id=1001" class="bbCodeBlock-sourceJump" rel="nofollow" data-xf-click="attribution" data-content-selector="#post-1001">Marigold said:</a></div> + <div class="bbCodeBlock-content"> + <div class="bbCodeBlock-expandContent js-expandContent "> + The blue one is a reproduction.<br /> + Look at the glaze. + </div> + <div class="bbCodeBlock-expandLink js-expandLink"><a role="button" tabindex="0">Click to expand...</a></div> + </div> +</blockquote> +I disagree, <a href="/members/marigold.7/" class="username" data-xf-init="member-tooltip" data-user-id="7" data-username="@Marigold">@Marigold</a>.<br /> +<br /> +Here is the catalogue: <a href="https://archive.example/AbCd1" target="_blank" class="link link--external" rel="nofollow ugc noopener">https://archive.example/AbCd1</a><br /> +And the <a href="https://museum.example/teapots?id=9" target="_blank" class="link link--external" rel="nofollow ugc noopener">museum page</a> <img src="/styles/default/smilies/smile.png" class="smilie" alt=":)" title="Smile :)" data-shortname=":)" /> +<div class="bbImageWrapper js-lbImage" title="teapot.jpg" data-src="https://images.example/teapot.jpg" data-lb-sidebar-href="" data-lb-caption-extra-html="" data-single-image="1"> + <img src="https://forum.example/proxy.php?image=https%3A%2F%2Fimages.example%2Fteapot.jpg" data-url="https://images.example/teapot.jpg" class="bbImage" data-zoom-target="1" style="" alt="teapot.jpg" title="" width="" height="" loading="lazy" /> +</div> +<span data-s9e-mediaembed="youtube" style="display:inline-block;width:100%;max-width:640px"><span style="display:block;overflow:hidden;position:relative;padding-bottom:56.25%"><iframe allowfullscreen="" loading="lazy" scrolling="no" style="border:0;height:100%;left:0;position:absolute;width:100%" src="https://www.youtube.com/embed/AbCdEfGhIjK"></iframe></span></span> +<iframe data-s9e-mediaembed="twitter" allow="autoplay *" allowfullscreen="" loading="lazy" scrolling="no" src="https://s9e.github.io/iframe/2/twitter.min.html#1234567890123456789" style="background:url(https://abs.twimg.com/favicons/favicon.ico) no-repeat 50% 50%;border:0;height:350px;max-width:550px;width:100%"></iframe> +<div class="bbCodeSpoiler"> + <button type="button" class="bbCodeSpoiler-button button--longText button" data-xf-click="toggle" data-xf-init="tooltip" title="Click to reveal or hide spoiler"><span class="button-text"><span>Spoiler: <span class="bbCodeSpoiler-button-title">the ending</span></span></span></button> + <div class="bbCodeSpoiler-content"><div class="bbCodeBlock bbCodeBlock--spoiler"><div class="bbCodeBlock-content">The lid was glued on.</div></div></div> +</div> +<div class="bbMediaWrapper"><div class="bbMediaWrapper-inner"><video controls="" data-xf-init="video-init"><source src="/data/video/12/12345-abc.mp4" /><div class="bbMediaWrapper-fallback">Your browser is not able to display this video.</div></video></div></div> +<div class="bbCodeBlock bbCodeBlock--unfurl js-unfurl fauxBlockLink" data-unfurl="true" data-result-id="77" data-url="https://news.example/teapot-auction" data-host="news.example" data-pending="false"> + <div class="contentRow"><div class="contentRow-main"><h3 class="contentRow-header js-unfurl-title"><a href="https://news.example/teapot-auction" class="link link--external fauxBlockLink-blockLink" target="_blank" rel="nofollow ugc noopener" data-proxy-href="">Teapot sells for a fortune</a></h3><div class="contentRow-snippet js-unfurl-desc">An auction report.</div></div></div> +</div> +<ul><li>first point</li><li>second point</li></ul> +<a href="https://forum.example/attachments/receipt-png.555/" target="_blank"><img src="https://forum.example/data/attachments/0/555-receipt.jpg" class="bbImage" alt="receipt.png" /></a> +`; diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts @@ -54,6 +54,12 @@ export type PostFetchInput = { // Soft cap on how many posts to return in one run. Undefined = no cap // beyond the watermark/seen-id stop conditions. limit?: number; + // Cap on how many PAGES one run reads, for a fetcher that walks pages (a + // forum thread: "the latest N pages"). Ignored by the others. + pages?: number; + // The pause between two page loads, for a paced page walker (a forum + // thread). Default: the fetcher's own. + pagePauseMs?: number; // Stop once the walk reaches posts that are already archived. False for a // --full repair, which re-walks everything on purpose. Undefined: the // fetcher's own default. @@ -181,8 +187,10 @@ export type PostAvailabilityInput = { // (postCapture.ts owns the layout). export type PostCaptureInput = Pick< PostFetchInput, - "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog" + "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog" | "pagePauseMs" > & { + // The channel's URL (a forum capture needs the forum's origin). + accountUrl?: string; ids: ReadonlyArray<string>; handle: string; outDir: string; @@ -197,7 +205,17 @@ export type PostCaptureInput = Pick< // What the posts archive holds for each id — its text and expanded links — // for the links a capture follows (an X Article's). Ids absent are read // from the page alone. - archived?: ReadonlyMap<string, { text: string; links?: ReadonlyArray<string> }>; + archived?: ReadonlyMap< + string, + { + text: string; + links?: ReadonlyArray<string>; + // The post's own URL and media, where the archive has them (forum + // posts: the capture opens the URL and downloads the media). + url?: string; + media?: ReadonlyArray<{ kind: string; url: string; name?: string }>; + } + >; // A soft stop: no new post is started once it fires, and the one in hand // finishes (`signal` cancels outright). drain?: AbortSignal; @@ -341,6 +359,11 @@ export function handleFromAccountUrl(input: string): string | null { } const segments = url.pathname.split("/").filter(Boolean); if (segments.length === 0) return null; + // A forum thread: its "<slug>.<id>" key is the channel's handle. + const threadIdx = segments.findIndex((s) => s === "threads"); + if (threadIdx >= 0 && /(?:^|\.)\d+$/.test(segments[threadIdx + 1] ?? "")) { + return decodeURIComponent(segments[threadIdx + 1]); + } // bsky.app/profile/<handle-or-did> if (segments[0] === "profile" && segments[1]) { return decodeURIComponent(segments[1]).replace(/^@/, ""); diff --git a/common/social/forumSession.ts b/common/social/forumSession.ts @@ -0,0 +1,406 @@ +// THE FORUM BROWSER: one persistent Chromium profile per forum host, opened +// headless by the thread fetcher and the post capture, and HEADED by Connect. +// +// Why a persistent profile: a forum behind a browser check (Kiwi Farms' +// KiwiFlare — a JavaScript proof of work — or a "just a moment" page) sets a +// clearance cookie once the check passes. A real browser clears it by itself; +// keeping the profile keeps the cookie, so every later page and run reuses it +// instead of meeting the check again. +// +// Connect is the operator's way through anything the headless browser cannot +// pass (a captcha, a login a members-only thread needs): a headed window on the +// same profile, at the thread, which the operator clears by hand and closes. +// Mirrors the X session broker (xSessionBroker.ts): the same browser choice +// and launch options (xBrowser.ts — the operator's own Chromium when one is +// installed, without the automation bar), the same record of which browser +// wrote the profile, and the same rule that a window opens only on an explicit +// operator action, never from a fetch. +// +// The user agent is the browser's own. Headless Chromium names itself +// "HeadlessChrome" there; that one word is replaced with "Chrome" so the +// headless profile presents as the browser it is. Nothing else is disguised, +// and a captcha is never answered by code. +// +// Layout, per host: transcripts/.forum-session/<host>/profile/ (the profile), +// transcripts/.forum-session/<host>/session.json (the last Connect). + +import path from "node:path"; +import { mkdir, rm, stat } from "node:fs/promises"; +import { execa } from "execa"; +import type { Paths } from "../lib/paths"; +import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; +import { + importPlaywright, + type BrowserContextLike, + type ChromiumLike, + type PageLike, +} from "./playwrightRuntime"; +import { + buildXBrowserLaunchOptions, + describeXBrowser, + findXBrowser, + readXBrowserRecord, + recordedXBrowser, + writeXBrowserRecord, + type XBrowserChoice, +} from "./xBrowser"; +import { classifyForumPage, type ForumBlock } from "./xenforoParse"; + +// A host as a directory name: lowercased, nothing but [a-z0-9.-]. +export function forumHostKey(host: string): string { + const key = host.trim().toLowerCase().replace(/[^a-z0-9.-]/g, "_"); + if (!key || key.startsWith(".")) throw new Error(`"${host}" is not a forum host`); + return key; +} + +export function forumSessionDir(paths: Pick<Paths, "transcriptsDir">, host: string): string { + return path.join(paths.transcriptsDir, ".forum-session", forumHostKey(host)); +} + +export function forumProfileDir(paths: Pick<Paths, "transcriptsDir">, host: string): string { + return path.join(forumSessionDir(paths, host), "profile"); +} + +export type ForumSessionRecord = { + host: string; + connectedAt: string; + // The window's last look at the forum before it closed: did a thread page + // show (the check was cleared), and how many cookies the host had set. + cleared: boolean; + cookies: number; + browser?: string; +}; + +export type ForumSessionStatus = { + host: string; + hasProfile: boolean; + lastConnect?: ForumSessionRecord; +}; + +export async function readForumSessionStatus( + paths: Pick<Paths, "transcriptsDir">, + host: string, +): Promise<ForumSessionStatus> { + const dir = forumSessionDir(paths, host); + const hasProfile = await stat(path.join(dir, "profile")) + .then(() => true) + .catch(() => false); + const read = await readJsonFile(path.join(dir, "session.json")); + const rec = read.ok ? (read.value as ForumSessionRecord | null) : null; + return { + host, + hasProfile, + ...(rec && typeof rec.connectedAt === "string" ? { lastConnect: rec } : {}), + }; +} + +export async function clearForumSession(paths: Pick<Paths, "transcriptsDir">, host: string): Promise<void> { + await rm(forumSessionDir(paths, host), { recursive: true, force: true }); +} + +function firstLine(err: unknown): string { + return ((err as Error)?.message ?? String(err)).split("\n")[0]; +} + +// --- opening the profile --------------------------------------------------------- + +const BUNDLED: XBrowserChoice = { kind: "bundled" }; + +// The headless user agent with its one tell removed, learned once per process. +let normalUserAgent: string | undefined; + +export function presentableUserAgent(ua: string): string { + return ua.replace(/HeadlessChrome/g, "Chrome"); +} + +async function launchHeadless( + chromium: ChromiumLike, + profileDir: string, + browser: XBrowserChoice, + userAgent?: string, +): Promise<BrowserContextLike> { + return chromium.launchPersistentContext(profileDir, { + ...buildXBrowserLaunchOptions({ browser, headless: true }), + viewport: { width: 1280, height: 900 }, + ...(userAgent ? { userAgent } : {}), + }); +} + +// The host's profile, headless. The bundled build first; the browser the +// profile records (the one Connect opened) when the bundled build cannot open +// it — as launchXProfile does for X. +export async function openForumProfile( + profileDir: string, + opts: { onLog?: (line: string) => void; chromium?: ChromiumLike } = {}, +): Promise<BrowserContextLike> { + const log = opts.onLog ?? (() => {}); + await mkdir(profileDir, { recursive: true }); + const chromium = opts.chromium ?? (await importPlaywright()).chromium; + let browser: XBrowserChoice = BUNDLED; + let context: BrowserContextLike; + try { + context = await launchHeadless(chromium, profileDir, browser, normalUserAgent); + } catch (err) { + const record = await readXBrowserRecord(profileDir); + const recorded = recordedXBrowser(record); + if (!recorded) throw err; + log( + `Playwright's bundled Chromium could not open the forum profile (${firstLine(err)}); ` + + `using ${describeXBrowser(recorded, record?.version)}, headless.`, + ); + browser = recorded; + context = await launchHeadless(chromium, profileDir, browser, normalUserAgent); + } + if (normalUserAgent) return context; + // First launch in this process: read the browser's own agent; if it names + // itself headless, reopen once with that word replaced. + try { + const page = context.pages()[0] ?? (await context.newPage()); + const ua = String(await page.evaluate("navigator.userAgent")); + const fixed = presentableUserAgent(ua); + normalUserAgent = fixed; + if (fixed !== ua) { + await context.close().catch(() => {}); + context = await launchHeadless(chromium, profileDir, browser, fixed); + } + } catch (err) { + log(`Could not read the browser's user agent (${firstLine(err)}); keeping its default.`); + } + return context; +} + +// --- loading one page through a browser check --------------------------------------- + +export type ForumPageLoad = { + html: string; + // The response status, when the page came straight back (no check waited + // out — after a check the first response's status no longer describes the + // page). + status?: number; + url: string; + // How long a browser check took to clear, when there was one. + challengeMs?: number; +}; + +export type ForumPageLoader = { + load(url: string, signal: AbortSignal): Promise<ForumPageLoad>; + close(): Promise<void>; +}; + +export const CHALLENGE_TIMEOUT_MS = 60_000; +const CHALLENGE_POLL_MS = 2_000; +const OUTER_HTML = "document.documentElement ? document.documentElement.outerHTML : ''"; + +type ResponseLike = { status?: () => number } | null | undefined; + +// Go to `url` in `page` and come back with what it shows. A browser check +// (classifyForumPage → "challenge") is WAITED OUT, up to `timeoutMs`: the +// check's own script runs, sets its cookie and reloads, and the poll reads +// the page that follows. Nothing is clicked or solved. What the page shows at +// the end — the thread, or the check still standing — is the caller's to +// judge. +export async function loadForumPage( + page: PageLike, + url: string, + opts: { + signal: AbortSignal; + onLog?: (line: string) => void; + timeoutMs?: number; + pollMs?: number; + now?: () => number; + }, +): Promise<ForumPageLoad> { + const now = opts.now ?? Date.now; + const timeoutMs = opts.timeoutMs ?? CHALLENGE_TIMEOUT_MS; + const pollMs = opts.pollMs ?? CHALLENGE_POLL_MS; + let status: number | undefined; + try { + const res = (await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 })) as ResponseLike; + status = typeof res?.status === "function" ? res.status() : undefined; + } catch (err) { + // A navigation the check's own reload interrupted still leaves a page to + // read; anything else is a load failure. + if (!/interrupted|ERR_ABORTED|frame was detached/i.test(firstLine(err))) { + throw new Error(`Could not load ${url}: ${firstLine(err)}`); + } + } + const read = async (): Promise<string | null> => { + try { + return String(await page.evaluate(OUTER_HTML)); + } catch { + return null; // mid-navigation + } + }; + let html = (await read()) ?? ""; + let block: ForumBlock | null = classifyForumPage(html, status); + if (block?.kind !== "challenge") { + return { html, ...(status !== undefined ? { status } : {}), url: await currentUrl(page, url) }; + } + const started = now(); + opts.onLog?.( + `The forum served a browser check (${block.detail}); waiting up to ${Math.round(timeoutMs / 1000)} s for it to clear.`, + ); + while (now() - started < timeoutMs) { + if (opts.signal.aborted) break; + await page.waitForTimeout(pollMs); + const next = await read(); + if (next === null) continue; + html = next; + block = classifyForumPage(html); + if (block?.kind !== "challenge") break; + } + const challengeMs = now() - started; + if (block?.kind === "challenge") { + opts.onLog?.(`The browser check had not cleared after ${Math.round(challengeMs / 1000)} s.`); + } else { + opts.onLog?.(`The browser check cleared in ${Math.round(challengeMs / 1000)} s.`); + } + return { html, url: await currentUrl(page, url), challengeMs }; +} + +async function currentUrl(page: PageLike, fallback: string): Promise<string> { + try { + const href = String(await page.evaluate("location.href")); + return href && href !== "about:blank" ? href : fallback; + } catch { + return fallback; + } +} + +// A loader over the host's headless profile, opened on the first load and +// kept for the run (one browser, one page, every page in turn). +export function browserForumLoader( + profileDir: string, + opts: { onLog?: (line: string) => void; timeoutMs?: number; chromium?: ChromiumLike } = {}, +): ForumPageLoader & { page(): Promise<PageLike> } { + let context: BrowserContextLike | undefined; + let page: PageLike | undefined; + const ensure = async (): Promise<PageLike> => { + if (!context) { + context = await openForumProfile(profileDir, { onLog: opts.onLog, chromium: opts.chromium }); + opts.onLog?.("Opened the forum browser profile (headless)."); + } + page ??= context.pages()[0] ?? (await context.newPage()); + return page; + }; + return { + page: ensure, + async load(url, signal) { + const p = await ensure(); + return loadForumPage(p, url, { signal, onLog: opts.onLog, timeoutMs: opts.timeoutMs }); + }, + async close() { + await context?.close().catch(() => {}); + context = undefined; + page = undefined; + }, + }; +} + +// --- Connect: the headed window ---------------------------------------------------------- + +async function browserVersion(b: XBrowserChoice): Promise<string | undefined> { + if (b.kind !== "system") return undefined; + try { + const res = await execa(b.executablePath, ["--version"], { reject: false, timeout: 5_000 }); + const line = `${res.stdout ?? ""}`.trim().split("\n")[0]; + return res.exitCode === 0 && line ? line : undefined; + } catch { + return undefined; + } +} + +type CookieLike = { domain?: string }; + +function cookiesForHost(cookies: ReadonlyArray<CookieLike>, host: string): number { + const h = host.toLowerCase(); + return cookies.filter((c) => { + const d = (c.domain ?? "").replace(/^\./, "").toLowerCase(); + return d === h || h.endsWith(`.${d}`) || d.endsWith(`.${h}`); + }).length; +} + +// Open a HEADED browser on the host's profile at `url` (the thread), for the +// operator to clear the check, answer a captcha or log in. Resolves when the +// window is closed, with the window's last look at the forum. The window opens +// on the display of the machine running the editor (as Connect X does). +export async function connectForumSession( + paths: Pick<Paths, "transcriptsDir">, + url: string, + opts: { onLog?: (line: string) => void; timeoutMs?: number; env?: Record<string, string | undefined> } = {}, +): Promise<ForumSessionRecord> { + const log = opts.onLog ?? (() => {}); + const host = new URL(url).hostname.toLowerCase(); + const profileDir = forumProfileDir(paths, host); + await mkdir(profileDir, { recursive: true }); + const browser = findXBrowser({ env: opts.env }); + const version = await browserVersion(browser); + const label = describeXBrowser(browser, version); + const { chromium } = await importPlaywright(); + log(`Opening ${label} at ${host}. Clear the check (or log in), wait for the thread to show, then close the window.`); + let context: BrowserContextLike; + try { + context = await chromium.launchPersistentContext( + profileDir, + buildXBrowserLaunchOptions({ browser, headless: false, sandbox: true }), + ); + } catch (err) { + log(`The sandboxed launch failed (${firstLine(err)}); opening without the sandbox.`); + context = await chromium.launchPersistentContext( + profileDir, + buildXBrowserLaunchOptions({ browser, headless: false, sandbox: false }), + ); + } + await writeXBrowserRecord(profileDir, browser, version).catch((err) => { + log(`Could not record the browser in the profile: ${firstLine(err)}`); + }); + + let cleared = false; + let cookies = 0; + const look = async () => { + try { + const page = context.pages()[0]; + if (page) { + const html = String(await page.evaluate(OUTER_HTML)); + cleared = classifyForumPage(html) === null; + } + cookies = cookiesForHost((await context.cookies()) as CookieLike[], host); + } catch { + /* closing, or mid-navigation */ + } + }; + try { + const page = context.pages()[0] ?? (await context.newPage()); + await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 }).catch((err: unknown) => { + log(`The first load did not finish (${firstLine(err)}); the window stays open.`); + }); + await new Promise<void>((resolve) => { + let settled = false; + const timer = setInterval(() => void look(), 2_000); + const finish = () => { + if (settled) return; + settled = true; + clearInterval(timer); + resolve(); + }; + context.on("close", finish); + if (opts.timeoutMs) setTimeout(finish, opts.timeoutMs); + }); + } finally { + await context.close().catch(() => {}); + } + const record: ForumSessionRecord = { + host, + connectedAt: new Date().toISOString(), + cleared, + cookies, + browser: label, + }; + await writeJsonAtomic(path.join(forumSessionDir(paths, host), "session.json"), record, { mkdir: true }); + log( + cleared + ? `The thread showed before the window closed; ${cookies} cookie(s) for ${host} kept in the profile.` + : `The window closed without a thread page showing; ${cookies} cookie(s) for ${host} in the profile.`, + ); + return record; +} diff --git a/common/social/htmlReader.ts b/common/social/htmlReader.ts @@ -0,0 +1,121 @@ +// A small HTML reader, pure: no DOM, no dependency. Shared by the X Article +// reader (xArticle.ts) and the XenForo thread parser (xenforoParse.ts). +// +// For HTML a browser serialised (outerHTML): attributes are double-quoted, void +// elements are unclosed, text escapes only & < > and nbsp. It tolerates more +// (single quotes, bare values, stray end tags), but it is not a general parser. + +export type HtmlElement = { + tag: string; + attrs: Record<string, string>; + children: HtmlNode[]; +}; +export type HtmlNode = HtmlElement | string; + +const VOID_TAGS = new Set([ + "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", + "param", "source", "track", "wbr", +]); +const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]); + +const NAMED_ENTITIES: Record<string, string> = { + amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0", + hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“", + rdquo: "”", copy: "©", reg: "®", trade: "™", +}; + +export function decodeEntities(s: string): string { + return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => { + if (name[0] === "#") { + const code = + name[1] === "x" || name[1] === "X" + ? parseInt(name.slice(2), 16) + : parseInt(name.slice(1), 10); + return Number.isFinite(code) && code > 0 && code <= 0x10ffff + ? String.fromCodePoint(code) + : whole; + } + return NAMED_ENTITIES[name.toLowerCase()] ?? whole; + }); +} + +const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y; +const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y; + +// The document's top-level nodes, under a synthetic root element. +export function parseHtml(html: string): HtmlElement { + const root: HtmlElement = { tag: "#root", attrs: {}, children: [] }; + // Lowered once: a raw-text element's end is searched for in it (a page holds + // dozens of scripts, and lowering the whole page per script is quadratic). + let lower: string | undefined; + const stack: HtmlElement[] = [root]; + const top = () => stack[stack.length - 1]; + let i = 0; + while (i < html.length) { + const lt = html.indexOf("<", i); + if (lt < 0) { + top().children.push(decodeEntities(html.slice(i))); + break; + } + if (lt > i) top().children.push(decodeEntities(html.slice(i, lt))); + i = lt; + if (html.startsWith("<!--", i)) { + const end = html.indexOf("-->", i + 4); + i = end < 0 ? html.length : end + 3; + continue; + } + if (html[i + 1] === "!" || html[i + 1] === "?") { + const end = html.indexOf(">", i); + i = end < 0 ? html.length : end + 1; + continue; + } + if (html[i + 1] === "/") { + const end = html.indexOf(">", i); + const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase(); + i = end < 0 ? html.length : end + 1; + // Close up to the matching open element; a stray end tag is ignored. + for (let k = stack.length - 1; k > 0; k--) { + if (stack[k].tag === name) { + stack.length = k; + break; + } + } + continue; + } + TAG_NAME_RE.lastIndex = i + 1; + const nameMatch = TAG_NAME_RE.exec(html); + if (!nameMatch) { + // A "<" that opens no tag is text. + top().children.push("<"); + i++; + continue; + } + const tag = nameMatch[0].toLowerCase(); + let j = TAG_NAME_RE.lastIndex; + const attrs: Record<string, string> = {}; + for (;;) { + ATTR_RE.lastIndex = j; + const a = ATTR_RE.exec(html); + if (!a || a[0].length === 0) break; + attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? ""); + j = ATTR_RE.lastIndex; + } + const close = html.indexOf(">", j); + const selfClosing = close > 0 && html[close - 1] === "/"; + i = close < 0 ? html.length : close + 1; + const el: HtmlElement = { tag, attrs, children: [] }; + top().children.push(el); + if (RAW_TEXT_TAGS.has(tag)) { + lower ??= html.toLowerCase(); + const end = lower.indexOf(`</${tag}`, i); + const stop = end < 0 ? html.length : end; + if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop))); + const gt = end < 0 ? -1 : html.indexOf(">", end); + i = gt < 0 ? html.length : gt + 1; + continue; + } + if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el); + } + return root; +} + diff --git a/common/social/xArticle.ts b/common/social/xArticle.ts @@ -19,6 +19,13 @@ // and an article whose body marker is missing is read by the fallback — the // root's text split at block elements — and says so (`extraction`). +import { + decodeEntities, + parseHtml, + type HtmlElement, + type HtmlNode, +} from "./htmlReader"; + // --- the link ------------------------------------------------------------------ export type XArticleLink = { @@ -60,121 +67,13 @@ export function xArticleLinkFromArchive( return findXArticleLink([archived.text, ...(archived.links ?? [])]); } -// --- a small HTML reader ------------------------------------------------------- +// --- the HTML reader ----------------------------------------------------------- // -// For HTML a browser serialised (outerHTML): attributes are double-quoted, void -// elements are unclosed, text escapes only & < > and nbsp. It tolerates more -// (single quotes, bare values, stray end tags), but it is not a general parser. - -export type HtmlElement = { - tag: string; - attrs: Record<string, string>; - children: HtmlNode[]; -}; -export type HtmlNode = HtmlElement | string; - -const VOID_TAGS = new Set([ - "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", - "param", "source", "track", "wbr", -]); -const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]); - -const NAMED_ENTITIES: Record<string, string> = { - amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0", - hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“", - rdquo: "”", copy: "©", reg: "®", trade: "™", -}; +// Shared with the forum-thread parser (xenforoParse.ts): it lives in +// htmlReader.ts and is re-exported here for the callers that import it from +// this module. -export function decodeEntities(s: string): string { - return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => { - if (name[0] === "#") { - const code = - name[1] === "x" || name[1] === "X" - ? parseInt(name.slice(2), 16) - : parseInt(name.slice(1), 10); - return Number.isFinite(code) && code > 0 && code <= 0x10ffff - ? String.fromCodePoint(code) - : whole; - } - return NAMED_ENTITIES[name.toLowerCase()] ?? whole; - }); -} - -const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y; -const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y; - -// The document's top-level nodes, under a synthetic root element. -export function parseHtml(html: string): HtmlElement { - const root: HtmlElement = { tag: "#root", attrs: {}, children: [] }; - const stack: HtmlElement[] = [root]; - const top = () => stack[stack.length - 1]; - let i = 0; - while (i < html.length) { - const lt = html.indexOf("<", i); - if (lt < 0) { - top().children.push(decodeEntities(html.slice(i))); - break; - } - if (lt > i) top().children.push(decodeEntities(html.slice(i, lt))); - i = lt; - if (html.startsWith("<!--", i)) { - const end = html.indexOf("-->", i + 4); - i = end < 0 ? html.length : end + 3; - continue; - } - if (html[i + 1] === "!" || html[i + 1] === "?") { - const end = html.indexOf(">", i); - i = end < 0 ? html.length : end + 1; - continue; - } - if (html[i + 1] === "/") { - const end = html.indexOf(">", i); - const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase(); - i = end < 0 ? html.length : end + 1; - // Close up to the matching open element; a stray end tag is ignored. - for (let k = stack.length - 1; k > 0; k--) { - if (stack[k].tag === name) { - stack.length = k; - break; - } - } - continue; - } - TAG_NAME_RE.lastIndex = i + 1; - const nameMatch = TAG_NAME_RE.exec(html); - if (!nameMatch) { - // A "<" that opens no tag is text. - top().children.push("<"); - i++; - continue; - } - const tag = nameMatch[0].toLowerCase(); - let j = TAG_NAME_RE.lastIndex; - const attrs: Record<string, string> = {}; - for (;;) { - ATTR_RE.lastIndex = j; - const a = ATTR_RE.exec(html); - if (!a || a[0].length === 0) break; - attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? ""); - j = ATTR_RE.lastIndex; - } - const close = html.indexOf(">", j); - const selfClosing = close > 0 && html[close - 1] === "/"; - i = close < 0 ? html.length : close + 1; - const el: HtmlElement = { tag, attrs, children: [] }; - top().children.push(el); - if (RAW_TEXT_TAGS.has(tag)) { - const end = html.toLowerCase().indexOf(`</${tag}`, i); - const stop = end < 0 ? html.length : end; - if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop))); - const gt = end < 0 ? -1 : html.indexOf(">", end); - i = gt < 0 ? html.length : gt + 1; - continue; - } - if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el); - } - return root; -} +export { decodeEntities, parseHtml, type HtmlElement, type HtmlNode }; // --- reading the article ------------------------------------------------------ diff --git a/common/social/xenforoFetcher.test.ts b/common/social/xenforoFetcher.test.ts @@ -0,0 +1,523 @@ +// The forum-thread walk, capture and page loader over fakes: no browser, no +// network. SYNTHETIC pages (__fixtures__/xenforoPages.ts). +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xenforoFetcher.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile, readdir } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import type { Post } from "../lib/posts"; +import type { PostFetchInput } from "./fetchers"; +import { loadForumPage, presentableUserAgent, type ForumPageLoader } from "./forumSession"; +import type { PageLike } from "./playwrightRuntime"; +import { + captureForumPosts, + DEFAULT_PAGE_PAUSE_MS, + downloadableMedia, + forumBlockFailure, + jitteredPause, + LAST_PAGE_PROBE, + MIN_PAGE_PAUSE_MS, + walkXenforoThread, + xenforoFetcher, +} from "./xenforoFetcher"; +import { + CHALLENGE_PAGE, + CAPTCHA_PAGE, + NOT_FOUND_PAGE, + ORIGIN, + THREAD_URL, + threadPage, + type FakePost, +} from "./__fixtures__/xenforoPages"; + +const T0 = 1_760_000_000; +const LAST = 5; +const PER_PAGE = 3; + +// Post k (1-based position) of the thread, on page ceil(k / 3). +function fakePost(position: number): FakePost { + return { + id: 9000 + position, + author: `Member${position % 4}`, + userId: 50 + (position % 4), + ts: T0 + position * 600, + position, + body: `Words of post ${position}.`, + }; +} + +function pageHtml(page: number): string { + const posts: FakePost[] = []; + for (let k = (page - 1) * PER_PAGE + 1; k <= page * PER_PAGE; k++) posts.push(fakePost(k)); + return threadPage({ page, last: LAST, posts }); +} + +const pageOf = (url: string): number => { + const m = /page-(\d+)$/.exec(url); + if (!m) return 1; + return Math.min(Number(m[1]), LAST); // past the end → the last page +}; + +type FakeLoader = ForumPageLoader & { urls: string[]; closed: number }; + +function fakeLoader( + override: (url: string, page: number) => { html: string; status?: number } | undefined = () => undefined, +): FakeLoader { + const loader: FakeLoader = { + urls: [], + closed: 0, + async load(url) { + loader.urls.push(url); + const page = pageOf(url); + const o = override(url, page); + return { + html: o?.html ?? pageHtml(page), + ...(o?.status ? { status: o.status } : { status: 200 }), + url: `${THREAD_URL}${page > 1 ? `page-${page}` : ""}`, + }; + }, + async close() { + loader.closed++; + }, + }; + return loader; +} + +function input(over: Partial<PostFetchInput> = {}): PostFetchInput { + return { + accountUrl: THREAD_URL, + handle: "the-teapot-collectors-thread.4242", + channelSlug: "teapots", + seenIds: new Set(), + signal: new AbortController().signal, + ...over, + }; +} + +function deps(loader: ForumPageLoader) { + const sleeps: number[] = []; + return { + sleeps, + deps: { + loader, + pauseMs: () => 11_000, + sleep: async (ms: number) => { + sleeps.push(ms); + }, + }, + }; +} + +const ids = (posts: Post[]) => posts.map((p) => Number(p.id) - 9000); + +test("first run: starts at the last page, walks newest first to page 1, paced between loads", async () => { + const loader = fakeLoader(); + const { deps: d, sleeps } = deps(loader); + const log: string[] = []; + const res = await walkXenforoThread(input({ onLog: (l) => log.push(l) }), d); + assert.equal(res.complete, true); + assert.equal(res.cursor, undefined); + assert.equal(loader.urls[0], `${THREAD_URL}page-${LAST_PAGE_PROBE}`); + assert.deepEqual(loader.urls.slice(1), [4, 3, 2].map((n) => `${THREAD_URL}page-${n}`).concat([THREAD_URL])); + // Every page's posts, each page in thread order, newest page first. + assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12, 7, 8, 9, 4, 5, 6, 1, 2, 3]); + // One pause before every load but the first, none in parallel. + assert.deepEqual(sleeps, [11_000, 11_000, 11_000, 11_000]); + assert.equal(loader.closed, 1); + assert.match(log[0], /^Page 5\/5: 3 post\(s\) parsed, 3 new\.$/); +}); + +test("a pages cap reads the latest N pages and leaves the cursor at the next one", async () => { + const loader = fakeLoader(); + const res = await walkXenforoThread(input({ pages: 2 }), deps(loader).deps); + assert.equal(res.complete, false); + assert.equal(res.cursor, "3"); + assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12]); + assert.equal(loader.urls.length, 2); +}); + +test("resume: a stored cursor continues backwards, and an archived post there does not end it", async () => { + const loader = fakeLoader(); + // Post 9 (page 3) shifted in from a page already read: archived. + const res = await walkXenforoThread(input({ cursor: "3", seenIds: new Set(["9009", "9013"]) }), deps(loader).deps); + assert.equal(res.complete, true); + assert.deepEqual(loader.urls, [`${THREAD_URL}page-3`, `${THREAD_URL}page-2`, THREAD_URL]); + assert.deepEqual(ids(res.posts), [7, 8, 4, 5, 6, 1, 2, 3]); +}); + +test("incremental: stops on the page holding an already-archived post, keeping its new ones", async () => { + const loader = fakeLoader(); + const seen = new Set(["9001", "9002", "9003", "9004", "9005", "9006", "9007", "9008", "9009", "9010"]); + const res = await walkXenforoThread(input({ seenIds: seen }), deps(loader).deps); + assert.equal(res.complete, true); + assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]); + assert.equal(loader.urls.length, 2); +}); + +test("a full re-walk (stopAtKnown false) does not stop at archived posts", async () => { + const loader = fakeLoader(); + const res = await walkXenforoThread( + input({ seenIds: new Set(["9013", "9010"]), stopAtKnown: false }), + deps(loader).deps, + ); + assert.equal(res.complete, true); + assert.equal(loader.urls.length, 5); + assert.equal(res.posts.length, 13); +}); + +test("the watermark stops the walk", async () => { + const since = new Date((T0 + 10 * 600) * 1000).toISOString(); // post 10's time + const res = await walkXenforoThread(input({ since }), deps(fakeLoader()).deps); + assert.equal(res.complete, true); + assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]); +}); + +test("a post limit is checked after a whole page", async () => { + const res = await walkXenforoThread(input({ limit: 4 }), deps(fakeLoader()).deps); + assert.equal(res.complete, false); + assert.equal(res.cursor, "3"); + assert.equal(res.posts.length, 6); +}); + +test("checkpoints: each page's posts are saved with the next page as the resume point", async () => { + const saved: { posts: number[]; cursor: string }[] = []; + const res = await walkXenforoThread( + input({ + pages: 3, + onCheckpoint: async (cp) => { + saved.push({ posts: ids(cp.posts), cursor: cp.cursor }); + }, + }), + deps(fakeLoader()).deps, + ); + assert.deepEqual(saved, [ + { posts: [13, 14, 15], cursor: "4" }, + { posts: [10, 11, 12], cursor: "3" }, + { posts: [7, 8, 9], cursor: "2" }, + ]); + assert.equal(res.posts.length, 0, "checkpointed posts are not handed back again"); + assert.equal(res.cursor, "2"); +}); + +test("drain stops between pages, keeping the next page; cancel stops too", async () => { + const drain = new AbortController(); + const loader = fakeLoader((_url, page) => { + if (page === 4) drain.abort(); + return undefined; + }); + const res = await walkXenforoThread(input({ drain: drain.signal }), deps(loader).deps); + assert.equal(res.drained, true); + assert.equal(res.complete, false); + assert.equal(res.cursor, "3"); + assert.deepEqual(ids(res.posts), [13, 14, 15, 10, 11, 12]); + + const cancel = new AbortController(); + const loader2 = fakeLoader((_url, page) => { + if (page === 5) cancel.abort(); + return undefined; + }); + const res2 = await walkXenforoThread(input({ signal: cancel.signal }), deps(loader2).deps); + assert.equal(res2.complete, false); + assert.equal(loader2.urls.length, 1); +}); + +test("a browser check that will not clear stops the run, typed, with the cursor kept and Connect named", async () => { + const loader = fakeLoader((_url, page) => (page === 4 ? { html: CHALLENGE_PAGE } : undefined)); + const res = await walkXenforoThread(input(), deps(loader).deps); + assert.equal(res.complete, false); + assert.equal(res.needsCookies, true); + assert.equal(res.cursor, "4"); + assert.match(res.error ?? "", /browser check/); + assert.match(res.error ?? "", /Connect forum session/); + assert.deepEqual(ids(res.posts), [13, 14, 15], "what was read before the check is kept"); + assert.equal(loader.urls.length, 2, "never retried"); +}); + +test("a captcha on the very first load keeps no cursor; a 429 is not a session problem", async () => { + const captcha = await walkXenforoThread( + input(), + deps(fakeLoader(() => ({ html: CAPTCHA_PAGE }))).deps, + ); + assert.equal(captcha.needsCookies, true); + assert.equal(captcha.cursor, undefined); + assert.match(captcha.error ?? "", /captcha/); + + const limited = await walkXenforoThread( + input({ cursor: "2" }), + deps(fakeLoader(() => ({ html: "<html><body>slow down</body></html>", status: 429 }))).deps, + ); + assert.equal(limited.needsCookies, undefined); + assert.equal(limited.cursor, "2"); + assert.match(limited.error ?? "", /HTTP 429/); +}); + +test("an article thread: its first post atop every page never stops the walk, and is taken once", async () => { + const article = fakePost(1); + const withArticle = (page: number) => { + const posts: FakePost[] = []; + for (let k = (page - 1) * PER_PAGE + 1; k <= page * PER_PAGE; k++) if (k > 1) posts.push(fakePost(k)); + return { html: threadPage({ page, last: LAST, posts, article }) }; + }; + // Incremental: the starter is archived, and old — neither ends the walk on + // the last page; the archived post 10 on page 4 does. + const seen = new Set(["9001", "9010"]); + const since = new Date((T0 + 10 * 600) * 1000).toISOString(); + const res = await walkXenforoThread( + input({ seenIds: seen, since }), + deps(fakeLoader((_url, page) => withArticle(page))).deps, + ); + assert.equal(res.complete, true); + assert.deepEqual(ids(res.posts), [13, 14, 15, 11, 12]); + // A first run takes the starter once, not once per page. + const all = await walkXenforoThread(input(), deps(fakeLoader((_url, page) => withArticle(page))).deps); + assert.equal(all.posts.filter((p) => p.id === "9001").length, 1); + assert.equal(all.posts.length, 15); +}); + +test("a page that parses to no posts stops rather than stepping past it", async () => { + const empty = threadPage({ page: 4, last: LAST, posts: [] }); + const res = await walkXenforoThread( + input(), + deps(fakeLoader((_url, page) => (page === 4 ? { html: empty } : undefined))).deps, + ); + assert.equal(res.complete, false); + assert.equal(res.cursor, "4"); + assert.match(res.error ?? "", /no posts/); +}); + +test("a bad cursor and a URL that is not a thread are refused before any load", async () => { + const loader = fakeLoader(); + assert.match((await walkXenforoThread(input({ cursor: "x" }), deps(loader).deps)).error ?? "", /not a page number/); + assert.match( + (await walkXenforoThread(input({ accountUrl: "https://forum.example/members/a.1/" }), deps(loader).deps)).error ?? "", + /not a XenForo thread URL/, + ); + assert.equal(loader.urls.length, 0); +}); + +test("pacing: jittered around the base, never below the floor", () => { + assert.equal(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0), Math.round(DEFAULT_PAGE_PAUSE_MS * 0.85)); + assert.ok(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0.999) <= 20_000); + assert.ok(jitteredPause(DEFAULT_PAGE_PAUSE_MS, () => 0) >= 10_000); + assert.equal(jitteredPause(100, () => 0), Math.round(MIN_PAGE_PAUSE_MS * 0.85)); +}); + +test("block failures: which ones ask for Connect", () => { + assert.equal(forumBlockFailure({ kind: "challenge", detail: "x" }, "forum.example").needsCookies, true); + assert.equal(forumBlockFailure({ kind: "login", detail: "x" }, "forum.example").needsCookies, true); + assert.equal(forumBlockFailure({ kind: "blocked", detail: "HTTP 403" }, "forum.example").needsCookies, false); + assert.equal(forumBlockFailure({ kind: "not-found", detail: "x" }, "forum.example").needsCookies, false); +}); + +test("the fetcher claims thread URLs only, and probes offline", async () => { + assert.equal(xenforoFetcher.detect(THREAD_URL), true); + assert.equal(xenforoFetcher.detect("https://x.com/someone"), false); + assert.equal(xenforoFetcher.detect("https://forum.example/members/a.1/"), false); + const probe = await xenforoFetcher.probe(`${THREAD_URL}page-3`); + assert.deepEqual(probe, { + ok: true, + name: "The Teapot Collectors Thread", + handle: "the-teapot-collectors-thread.4242", + url: THREAD_URL, + }); +}); + +// --- the page loader --------------------------------------------------------------- + +function fakePage(script: { status?: number; htmls: string[]; href?: string }) { + let clock = 0; + let reads = 0; + const page = { + waits: 0, + async goto() { + return { status: () => script.status ?? 200 }; + }, + async waitForSelector() {}, + async waitForTimeout(ms: number) { + page.waits++; + clock += ms; + }, + async evaluate(fn: string) { + if (fn === "location.href") return script.href ?? THREAD_URL; + const html = script.htmls[Math.min(reads, script.htmls.length - 1)]; + reads++; + return html; + }, + on() {}, + async screenshot() { + return new Uint8Array(); + }, + }; + return { page: page as unknown as PageLike & { waits: number }, now: () => clock }; +} + +test("loader: a browser check is waited out and the thread that follows is returned", async () => { + const thread = pageHtml(2); + const { page, now } = fakePage({ status: 403, htmls: [CHALLENGE_PAGE, CHALLENGE_PAGE, thread] }); + const log: string[] = []; + const res = await loadForumPage(page, `${THREAD_URL}page-2`, { + signal: new AbortController().signal, + onLog: (l) => log.push(l), + now, + }); + assert.equal(res.html, thread); + assert.equal(res.status, undefined, "the check's 403 does not describe the page that followed"); + assert.equal(res.challengeMs, 4_000); + assert.match(log.join("\n"), /cleared in 4 s/); +}); + +test("loader: a check still up at the timeout is handed back for the walk to judge", async () => { + const { page, now } = fakePage({ status: 403, htmls: [CHALLENGE_PAGE] }); + const res = await loadForumPage(page, THREAD_URL, { + signal: new AbortController().signal, + timeoutMs: 10_000, + now, + }); + assert.equal(res.html, CHALLENGE_PAGE); + assert.ok((res.challengeMs ?? 0) >= 10_000); +}); + +test("loader: a plain page comes straight back with its status", async () => { + const { page, now } = fakePage({ status: 200, htmls: [pageHtml(1)] }); + const res = await loadForumPage(page, THREAD_URL, { signal: new AbortController().signal, now }); + assert.equal(res.status, 200); + assert.equal(res.challengeMs, undefined); + assert.equal(page.waits, 0); +}); + +test("the headless agent presents as the browser it is", () => { + assert.equal( + presentableUserAgent("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) HeadlessChrome/147.0.0.0 Safari/537.36"), + "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/147.0.0.0 Safari/537.36", + ); +}); + +// --- capture ------------------------------------------------------------------------- + +function captureLoader(pages: Record<string, string>, rect: { x: number; y: number; width: number; height: number } | null) { + const urls: string[] = []; + const got: string[] = []; + const page = { + async goto() { + return null; + }, + async waitForSelector() {}, + async waitForTimeout() {}, + async evaluate(fn: string) { + if (fn.includes("getBoundingClientRect")) return { rect }; + return 0; + }, + on() {}, + async screenshot() { + return new Uint8Array([137, 80, 78, 71]); + }, + request: { + async get(url: string) { + got.push(url); + return { + ok: () => !url.includes("missing"), + status: () => (url.includes("missing") ? 404 : 200), + headers: () => ({ "content-type": "image/jpeg" }), + body: async () => new Uint8Array([1, 2, 3]), + }; + }, + }, + } as unknown as PageLike; + return { + urls, + got, + loader: { + async page() { + return page; + }, + async load(url: string) { + urls.push(url); + return { html: pages[url] ?? pageHtml(1), status: 200, url }; + }, + async close() {}, + }, + }; +} + +test("capture: shoots the post's article and downloads its file media through the profile", async () => { + const outDir = await mkdtemp(path.join(os.tmpdir(), "forum-capture-")); + const { loader, urls, got } = captureLoader({}, { x: 10, y: 400, width: 700, height: 300 }); + const sleeps: number[] = []; + const res = await captureForumPosts( + { + ids: ["9001", "9002"], + handle: "t", + accountUrl: THREAD_URL, + outDir, + archived: new Map([ + [ + "9001", + { + text: "x", + url: `${ORIGIN}/posts/9001/`, + media: [ + { kind: "image", url: "https://images.example/a.jpg" }, + { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK" }, + { kind: "attachment", url: `${ORIGIN}/attachments/scan-jpg.900/` }, + ], + }, + ], + ]), + signal: new AbortController().signal, + }, + { loader, pauseMs: () => 7_000, sleep: async (ms) => void sleeps.push(ms), now: () => new Date(0) }, + ); + assert.deepEqual(res.outcomes.map((o) => [o.id, o.state, o.files]), [ + ["9001", "captured", 3], + ["9002", "captured", 1], + ]); + assert.deepEqual(urls, [`${ORIGIN}/posts/9001/`, `${ORIGIN}/posts/9002/`]); + assert.deepEqual(got, ["https://images.example/a.jpg", `${ORIGIN}/attachments/scan-jpg.900/`], "embeds are pages, not files"); + // Every contact after the first is paced: 2 downloads + the second shot. + assert.equal(sleeps.length, 3); + assert.deepEqual((await readdir(path.join(outDir, "9001"))).sort(), ["capture.json", "media-1.jpg", "media-2.jpg", "shot.png"]); + const rec = JSON.parse(await readFile(path.join(outDir, "9001", "capture.json"), "utf8")); + assert.equal(rec.state, "captured"); + assert.equal(rec.mediaState, "ok"); + assert.equal(rec.url, `${ORIGIN}/posts/9001/`); + const rec2 = JSON.parse(await readFile(path.join(outDir, "9002", "capture.json"), "utf8")); + assert.equal(rec2.mediaState, "none"); +}); + +test("capture: a missing post is deleted; a check that will not clear stops the run for Connect", async () => { + const outDir = await mkdtemp(path.join(os.tmpdir(), "forum-capture-")); + const gone = captureLoader({ [`${ORIGIN}/posts/1/`]: NOT_FOUND_PAGE }, null); + const res = await captureForumPosts( + { ids: ["1"], handle: "t", accountUrl: THREAD_URL, outDir, signal: new AbortController().signal, media: false }, + { loader: gone.loader, pauseMs: () => 0, sleep: async () => {} }, + ); + assert.equal(res.outcomes[0].state, "deleted"); + assert.equal(res.outcomes[0].availability, "deleted"); + + const walled = captureLoader({ [`${ORIGIN}/posts/2/`]: CHALLENGE_PAGE }, null); + const res2 = await captureForumPosts( + { ids: ["2", "3"], handle: "t", accountUrl: THREAD_URL, outDir, signal: new AbortController().signal }, + { loader: walled.loader, pauseMs: () => 0, sleep: async () => {} }, + ); + assert.equal(res2.needsCookies, true); + assert.match(res2.stoppedEarly ?? "", /Connect forum session/); + assert.equal(res2.outcomes.length, 1, "the rest are left for a later run"); +}); + +test("downloadable media: files only", () => { + assert.deepEqual( + downloadableMedia([ + { kind: "image", url: "https://a.example/1.png" }, + { kind: "link-card", url: "https://a.example/page" }, + { kind: "video", url: "./Saved_files/v.mp4" }, + { kind: "video", url: "https://a.example/v.mp4" }, + ]), + [ + { kind: "image", url: "https://a.example/1.png" }, + { kind: "video", url: "https://a.example/v.mp4" }, + ], + ); +}); diff --git a/common/social/xenforoFetcher.ts b/common/social/xenforoFetcher.ts @@ -0,0 +1,626 @@ +// A XenForo forum THREAD as a posts source (platform "xenforo"): the channel is +// one thread, each forum post a Post. Kiwi Farms is the first host; any +// XenForo 2 forum whose thread pages a browser can open works the same way. +// +// THE WALK, NEWEST FIRST. A thread grows at its end, so a run starts on the +// last page (asked for with a page number past the end, which XenForo +// redirects to the last page) and steps back a page at a time. It stops at the +// first page holding an already-archived post (the incremental case), at the +// watermark, at page 1, or at a cap — `pages` ("the latest N pages") or +// `limit` (posts), checked after a whole page so a page is never half-read. +// The cursor is the next page to read, so a capped or drained run resumes +// there; a resumed run walks back to page 1 (pages shift as posts are +// deleted, so meeting an archived post there does not end it). +// +// PACED, SERIAL, ONE BROWSER. One page at a time with a jittered pause between +// loads (about 10–20 s by default; `pagePauseMs` from the channel), in one +// headless Chromium on the host's persistent profile (forumSession.ts). Never +// parallel, never faster on a slow run. +// +// A BROWSER CHECK (KiwiFlare's proof of work and the like) is waited out by the +// loader, up to a minute; the profile keeps the clearance for later pages and +// runs. A check that does not clear, a captcha, a login wall, a ban or a +// 403/429 STOPS the run with a typed failure and keeps the cursor — never a +// retry loop, never an attempt at a captcha. `needsCookies` is set where +// Connect (a headed window on the same profile) is the way through. +// +// One parser for everything: xenforoParse.ts reads the live page here and the +// operator's saved pages in the import (controller/importForumPages.ts). + +import { mkdir, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { getPaths } from "../lib/paths"; +import type { Post } from "../lib/posts"; +import { + registerSocialFetcher, + type PostCaptureInput, + type PostCaptureOutcome, + type PostCaptureResult, + type PostFetchInput, + type PostFetchResult, + type SocialFetcher, + type SocialFetcherProbe, +} from "./fetchers"; +import { + browserForumLoader, + forumProfileDir, + type ForumPageLoader, +} from "./forumSession"; +import { + captureAvailability, + captureWork, + describeCapturedFile, + postCaptureDir, + readPostCapture, + SHOT_FILENAME, + writePostCapture, + type CapturedFile, + type CaptureMediaState, + type PostCaptureRecord, + type PostCaptureState, +} from "./postCapture"; +import type { PageLike } from "./playwrightRuntime"; +import { + classifyForumPage, + parseXenforoThreadPage, + parseXenforoThreadUrl, + xenforoPageUrl, + xenforoThreadHandle, + type ForumBlock, + type XenforoThreadUrl, +} from "./xenforoParse"; + +// A page number past any real thread's end: XenForo redirects it to the last +// page, which saves reading page 1 just to learn where the end is. +export const LAST_PAGE_PROBE = 1_000_000; + +export const DEFAULT_PAGE_PAUSE_MS = 12_000; +// The floor a configured pause is held to. +export const MIN_PAGE_PAUSE_MS = 5_000; + +// The pause before a load: `base` jittered to [0.85, 1.65) of itself — 12 s +// gives 10.2–19.8 s. +export function jitteredPause(base: number, random: () => number): number { + const b = Math.max(MIN_PAGE_PAUSE_MS, base); + return Math.round(b * (0.85 + random() * 0.8)); +} + +export function sleepUnlessAborted(ms: number, signal: AbortSignal): Promise<void> { + return new Promise((resolve) => { + if (signal.aborted) return resolve(); + const timer = setTimeout(done, ms); + function done() { + clearTimeout(timer); + signal.removeEventListener("abort", done); + resolve(); + } + signal.addEventListener("abort", done, { once: true }); + }); +} + +// What a stop on a forum block says, and whether Connect is the way through. +export function forumBlockFailure( + block: ForumBlock, + host: string, +): { error: string; needsCookies: boolean } { + const connect = + `Use “Connect forum session” on the channel page: it opens a browser window on this host's profile ` + + `at the thread — clear it there, wait for the thread to show, close the window, and fetch again.`; + const shown = block.title ? ` (page title “${block.title}”)` : ""; + switch (block.kind) { + case "challenge": + return { + error: `${host} kept its browser check (${block.detail}) up past the wait${shown}; the headless browser could not clear it. ${connect}`, + needsCookies: true, + }; + case "captcha": + return { + error: `${host} asked for a captcha (${block.detail})${shown}, which only a person answers. ${connect}`, + needsCookies: true, + }; + case "login": + return { + error: `${host} shows this thread only to a logged-in member${shown}. ${connect} (log in in that window).`, + needsCookies: true, + }; + case "blocked": + return { + error: `${host} refused the page (${block.detail})${shown}. Stopped; the next run resumes here — do not run it again at once.`, + needsCookies: false, + }; + case "not-found": + return { error: `${host} says the thread could not be found (${block.detail}).`, needsCookies: false }; + default: + return { + error: `${host} answered with a page that is not a thread page (${block.detail})${shown}. Stopped.`, + needsCookies: false, + }; + } +} + +export type ThreadWalkDeps = { + loader: ForumPageLoader; + // The gap before each load after the run's first. + pauseMs: () => number; + sleep: (ms: number, signal: AbortSignal) => Promise<void>; +}; + +// THE WALK. Pure over its deps: the loader hands back HTML, the parser reads it. +export async function walkXenforoThread( + input: PostFetchInput, + deps: ThreadWalkDeps, +): Promise<PostFetchResult> { + const { channelSlug, seenIds, signal, onLog } = input; + const thread = parseXenforoThreadUrl(input.accountUrl); + if (!thread) { + return { posts: [], complete: false, error: `${input.accountUrl} is not a XenForo thread URL (…/threads/<title>.<id>/).` }; + } + const resuming = Boolean(input.cursor); + const startPage = resuming ? Number(input.cursor) : undefined; + if (resuming && (!Number.isInteger(startPage) || startPage! < 1)) { + return { posts: [], complete: false, error: `The stored resume point "${input.cursor}" is not a page number.` }; + } + const watermark = resuming ? undefined : input.since; + const stopAtKnown = (input.stopAtKnown ?? true) && !resuming; + const posts: Post[] = []; + const taken = new Set<string>(); + let returnedCount = 0; + let loads = 0; + let pagesRead = 0; + + const fetchPage = async (url: string) => { + if (loads++ > 0) await deps.sleep(deps.pauseMs(), signal); + return deps.loader.load(url, signal); + }; + + let page = startPage ?? LAST_PAGE_PROBE; + if (resuming) onLog?.(`Resuming the thread walk at page ${page}.`); + let lastPage: number | undefined; + const visited = new Set<number>(); + + try { + for (;;) { + if (signal.aborted) { + return { posts, complete: false, cursor: String(page) }; + } + const url = xenforoPageUrl(thread, page); + let load; + try { + load = await fetchPage(url); + } catch (err) { + return { + posts, + complete: false, + cursor: page === LAST_PAGE_PROBE ? undefined : String(page), + error: (err as Error).message, + }; + } + if (signal.aborted) return { posts, complete: false, cursor: page === LAST_PAGE_PROBE ? undefined : String(page) }; + const block = classifyForumPage(load.html, load.status); + if (block) { + const why = forumBlockFailure(block, thread.host); + onLog?.(`Page ${page === LAST_PAGE_PROBE ? "(last)" : page}: ${block.kind} — ${block.detail}.`); + return { + posts, + complete: false, + ...(page === LAST_PAGE_PROBE ? {} : { cursor: String(page) }), + error: why.error, + ...(why.needsCookies ? { needsCookies: true } : {}), + }; + } + const parsed = parseXenforoThreadPage(load.html, { channelSlug, pageUrl: load.url }); + if (visited.has(parsed.page)) { + return { + posts, + complete: false, + cursor: String(page), + error: `Asked for page ${page} and was shown page ${parsed.page} again; stopped rather than loop.`, + }; + } + visited.add(parsed.page); + page = parsed.page; + lastPage = Math.max(lastPage ?? 0, parsed.lastPage); + pagesRead++; + + let sawKnown = false; + let sawOld = false; + const fresh: Post[] = []; + // Newest first within the page too, so the stops read in time order. + for (const post of [...parsed.posts].reverse()) { + // A post that belongs to another page — an article thread's first + // post, shown atop every page — says nothing about where this page + // stands: it never stops the walk, and is taken once if new. + const repeated = post.forum?.page !== undefined && post.forum.page !== parsed.page; + if (repeated) { + if (!seenIds.has(post.id) && !taken.has(post.id)) { + taken.add(post.id); + fresh.push(post); + } + continue; + } + if (seenIds.has(post.id)) { + if (stopAtKnown) sawKnown = true; + continue; + } + if (watermark && post.createdAt <= watermark) { + sawOld = true; + continue; + } + if (taken.has(post.id)) continue; + taken.add(post.id); + fresh.push(post); + } + fresh.reverse(); + onLog?.( + `Page ${page}/${lastPage}: ${parsed.posts.length} post(s) parsed, ${fresh.length} new` + + (load.challengeMs !== undefined ? ` (browser check cleared in ${Math.round(load.challengeMs / 1000)} s)` : "") + + (sawKnown ? ", reached already-archived posts" : "") + + (sawOld ? ", reached the watermark" : "") + + ".", + ); + if (parsed.posts.length === 0 && page > 1) { + // A thread page with no posts in it is not a page to step past + // silently: it is most likely markup this parser does not know. + return { + posts, + complete: false, + cursor: String(page), + error: `Page ${page} of the thread parsed to no posts; stopped rather than walking on.`, + }; + } + + const next = page - 1; + returnedCount += fresh.length; + if (input.onCheckpoint && fresh.length > 0) { + await input.onCheckpoint({ posts: fresh, cursor: String(Math.max(1, next)) }); + } else { + posts.push(...fresh); + } + + if (sawKnown || sawOld || next < 1) { + return { posts, complete: true }; + } + if (input.pages && pagesRead >= input.pages) { + onLog?.(`Read ${pagesRead} page(s), the cap for this run; the next run continues at page ${next}.`); + return { posts, complete: false, cursor: String(next) }; + } + if (input.limit && returnedCount >= input.limit) { + return { posts, complete: false, cursor: String(next) }; + } + if (input.drain?.aborted) { + onLog?.(`Drained; the next run continues at page ${next}.`); + return { posts, complete: false, cursor: String(next), drained: true }; + } + page = next; + } + } finally { + await deps.loader.close().catch(() => {}); + } +} + +// --- capture ------------------------------------------------------------------------- + +// What the page shows for one post: the post's own article (its box, in +// document coordinates), and whatever the page says instead. +type ForumPostSnapshot = { + rect: { x: number; y: number; width: number; height: number } | null; +}; + +const SNAPSHOT_SCRIPT = (id: string) => `(() => { + const el = document.querySelector('article.message[data-content="post-${id}"]') || + document.getElementById('js-post-${id}'); + if (!el) return { rect: null }; + el.scrollIntoView({ block: "center" }); + const r = el.getBoundingClientRect(); + return { rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height } }; +})()`; + +// Let the post's images load (each capped), and open its spoilers, so the shot +// shows what the post holds. +const PREPARE_SCRIPT = (id: string) => `(() => { + const el = document.querySelector('article.message[data-content="post-${id}"]') || + document.getElementById('js-post-${id}'); + if (!el) return 0; + for (const s of el.querySelectorAll('.bbCodeSpoiler')) s.classList.add('is-active'); + for (const c of el.querySelectorAll('.bbCodeBlock--expandable')) c.classList.add('is-expanded'); + const imgs = Array.from(el.querySelectorAll('img')).filter((i) => !i.complete); + return Promise.all(imgs.map((i) => new Promise((r) => { + i.addEventListener('load', r, { once: true }); + i.addEventListener('error', r, { once: true }); + setTimeout(r, 5000); + }))).then(() => imgs.length); +})()`; + +export type ForumShotResult = { + state: PostCaptureState; + shot?: CapturedFile; + error?: string; + stop?: string; +}; + +export async function shootForumPost( + load: (url: string) => Promise<{ html: string; status?: number }>, + page: PageLike, + postUrl: string, + id: string, + dir: string, + host: string, +): Promise<ForumShotResult> { + let got; + try { + got = await load(postUrl); + } catch (err) { + return { state: "error", error: (err as Error).message }; + } + const block = classifyForumPage(got.html, got.status); + if (block) { + if (block.kind === "not-found") return { state: "deleted" }; + const why = forumBlockFailure(block, host); + return { + state: block.kind === "challenge" || block.kind === "captcha" || block.kind === "login" ? "login-wall" : "error", + error: why.error, + stop: why.error, + }; + } + await page.waitForTimeout(1_000); + await page.evaluate(PREPARE_SCRIPT(id)).catch(() => {}); + const snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as ForumPostSnapshot; + if (!snap.rect) { + // The thread rendered without this post: XenForo shows a deleted post to + // nobody but moderators, and a moved post redirects elsewhere. + return { state: "deleted" }; + } + const r = snap.rect; + if (r.width < 1 || r.height < 1) return { state: "error", error: "The post rendered with no size to shoot." }; + let png: Uint8Array; + try { + png = await page.screenshot({ + type: "png", + fullPage: true, + clip: { x: Math.max(0, Math.floor(r.x)), y: Math.max(0, Math.floor(r.y)), width: Math.ceil(r.width), height: Math.ceil(r.height) }, + }); + } catch (err) { + return { state: "error", error: `The screenshot failed: ${((err as Error).message ?? "").split("\n")[0]}` }; + } + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, SHOT_FILENAME), png); + return { state: "captured", shot: await describeCapturedFile(dir, SHOT_FILENAME, postUrl) }; +} + +const EXT_BY_TYPE: Record<string, string> = { + "image/jpeg": "jpg", + "image/png": "png", + "image/gif": "gif", + "image/webp": "webp", + "image/avif": "avif", + "video/mp4": "mp4", + "video/webm": "webm", + "application/pdf": "pdf", +}; + +function extFor(url: string, contentType: string | undefined): string { + const ct = (contentType ?? "").split(";")[0].trim().toLowerCase(); + if (EXT_BY_TYPE[ct]) return EXT_BY_TYPE[ct]; + const m = /\.([a-z0-9]{2,5})(?:[?#/]|$)/i.exec(new URL(url).pathname.replace(/-([a-z0-9]{2,5})\.\d+\/?$/i, ".$1")); + return m ? m[1].toLowerCase() : "bin"; +} + +// The media a capture downloads: images, attachments and videos the post +// carries as files. Embeds and link cards are pages, not files. +export function downloadableMedia( + media: ReadonlyArray<{ kind: string; url: string }> | undefined, +): { kind: string; url: string }[] { + return (media ?? []).filter( + (m) => (m.kind === "image" || m.kind === "attachment" || m.kind === "video") && /^https?:\/\//.test(m.url), + ); +} + +export type ForumCaptureDeps = { + loader: ForumPageLoader & { page(): Promise<PageLike> }; + pauseMs: () => number; + sleep: (ms: number, signal: AbortSignal) => Promise<void>; + now?: () => Date; +}; + +const STOP_AFTER_ERRORS = 3; + +// The capture loop for forum posts: per id, the post's page (its permalink, +// which XenForo resolves to the post in its thread), the shot of its article, +// then its media fetched through the same browser profile (so a clearance +// cookie covers the downloads too). Every contact after the first is paced. +export async function captureForumPosts( + input: PostCaptureInput, + deps: ForumCaptureDeps, +): Promise<PostCaptureResult> { + const { signal, onLog } = input; + const thread = parseXenforoThreadUrl(input.accountUrl ?? ""); + if (!thread) { + return { outcomes: [], stoppedEarly: `${input.accountUrl ?? "(no URL)"} is not a XenForo thread URL.` }; + } + const wanted = { shots: input.shots ?? true, media: input.media ?? true, force: input.force ?? false }; + const now = deps.now ?? (() => new Date()); + const outcomes: PostCaptureOutcome[] = []; + let contacts = 0; + let errorsInARow = 0; + const contact = async () => { + if (contacts++ > 0) await deps.sleep(deps.pauseMs(), signal); + }; + const stopped = (why: string, extra: Partial<PostCaptureResult> = {}): PostCaptureResult => { + onLog?.(why); + return { outcomes, stoppedEarly: why, ...extra }; + }; + + try { + for (const [i, id] of input.ids.entries()) { + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + if (input.drain?.aborted) return stopped("Drained; the rest are left for a later run."); + const dir = postCaptureDir(input.outDir, id); + const existing = await readPostCapture(dir); + const work = captureWork(existing, { ...wanted, articles: false }); + if (!work.shot && !work.media) { + onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`); + continue; + } + onLog?.(`[${i + 1}/${input.ids.length}] ${id}`); + const archived = input.archived?.get(id); + const postUrl = + archived?.url && /^https?:\/\//.test(archived.url) ? archived.url : `${thread.origin}/posts/${id}/`; + + let state: PostCaptureState | undefined = work.shot ? undefined : existing?.state; + let shot = work.shot ? undefined : existing?.shot; + let error: string | undefined; + let stop: string | undefined; + + if (work.shot) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + const page = await deps.loader.page(); + const res = await shootForumPost( + (url) => deps.loader.load(url, signal), + page, + postUrl, + id, + dir, + thread.host, + ); + state = res.state; + shot = res.shot; + error = res.error; + stop = res.stop; + } + + let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped"); + let media: CapturedFile[] = work.media ? [] : (existing?.media ?? []); + const postIsThere = state === undefined || state === "captured"; + if (work.media && postIsThere && !stop) { + const files = downloadableMedia(archived?.media); + if (files.length === 0) { + mediaState = "none"; + state ??= "captured"; + } else { + const page = await deps.loader.page(); + let failed = 0; + for (const [n, m] of files.entries()) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + try { + const res = await page.request!.get(m.url, { timeout: 60_000, failOnStatusCode: false }); + if (!res.ok()) { + failed++; + onLog?.(`${id}: media ${n + 1} answered HTTP ${res.status()}.`); + continue; + } + const name = `media-${n + 1}.${extFor(m.url, res.headers()["content-type"])}`; + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, name), await res.body()); + media.push(await describeCapturedFile(dir, name, m.url)); + } catch (err) { + failed++; + onLog?.(`${id}: media ${n + 1} failed: ${((err as Error).message ?? "").split("\n")[0]}`); + } + } + mediaState = failed > 0 ? "error" : "ok"; + if (failed > 0) error = error ? `${error}; ${failed} media file(s) failed` : `${failed} media file(s) failed`; + state ??= "captured"; + } + } + + const finalState: PostCaptureState = state ?? "error"; + const record: PostCaptureRecord = { + version: 1, + id, + url: postUrl, + capturedAt: now().toISOString(), + state: finalState, + ...(shot ? { shot } : {}), + mediaState, + media, + ...(error ? { error } : {}), + }; + await writePostCapture(dir, record); + outcomes.push({ + id, + state: finalState, + ...(work.shot && captureAvailability(finalState) ? { availability: captureAvailability(finalState) } : {}), + files: (shot ? 1 : 0) + media.length, + ...(error ? { error } : {}), + }); + onLog?.( + `${id}: ${finalState}` + + (shot ? ", shot" : "") + + (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") + + (error ? ` — ${error}` : ""), + ); + if (stop) { + return stopped(`${stop} Stopped at ${id}; the rest are left for a later run.`, { + needsCookies: finalState === "login-wall", + }); + } + errorsInARow = finalState === "error" ? errorsInARow + 1 : 0; + if (errorsInARow >= STOP_AFTER_ERRORS) { + return stopped(`${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`); + } + } + return { outcomes }; + } finally { + await deps.loader.close().catch(() => {}); + } +} + +// --- the fetcher ---------------------------------------------------------------------- + +function threadOrThrow(url: string): XenforoThreadUrl { + const t = parseXenforoThreadUrl(url); + if (!t) throw new Error(`${url} is not a XenForo thread URL (…/threads/<title>.<id>/).`); + return t; +} + +const livePause = (base?: number) => () => jitteredPause(base ?? DEFAULT_PAGE_PAUSE_MS, Math.random); + +export const xenforoFetcher: SocialFetcher = { + id: "xenforo-thread", + label: "Forum thread (XenForo, headless browser)", + platform: "xenforo", + fields: { limit: true }, + + detect(url: string): boolean { + return parseXenforoThreadUrl(url) !== null; + }, + + // Offline: a probe that loaded the thread would be a paced page load (and a + // browser check) just to fill a form. The URL names the thread. + async probe(url: string): Promise<SocialFetcherProbe> { + const t = parseXenforoThreadUrl(url); + if (!t) return { ok: false, error: "Not a XenForo thread URL (…/threads/<title>.<id>/)." }; + const handle = xenforoThreadHandle(url) ?? t.threadId; + const words = handle.replace(/\.\d+$/, "").replace(/[-_]+/g, " ").trim(); + return { + ok: true, + name: words ? words.replace(/\b\w/g, (c) => c.toUpperCase()) : `Thread ${t.threadId}`, + handle, + url: t.base, + }; + }, + + async fetch(input: PostFetchInput): Promise<PostFetchResult> { + const thread = threadOrThrow(input.accountUrl); + const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog }); + return walkXenforoThread(input, { + loader, + pauseMs: livePause(input.pagePauseMs), + sleep: sleepUnlessAborted, + }); + }, + + async captureByIds(input: PostCaptureInput): Promise<PostCaptureResult> { + const thread = threadOrThrow(input.accountUrl ?? ""); + const loader = browserForumLoader(forumProfileDir(getPaths(), thread.host), { onLog: input.onLog }); + return captureForumPosts(input, { + loader, + pauseMs: livePause(input.pagePauseMs), + sleep: sleepUnlessAborted, + }); + }, +}; + +registerSocialFetcher(xenforoFetcher); diff --git a/common/social/xenforoParse.test.ts b/common/social/xenforoParse.test.ts @@ -0,0 +1,245 @@ +// The XenForo thread parser over SYNTHETIC pages (__fixtures__/xenforoPages.ts): +// posts, quotes, links, media, edits, the page nav, a browser-saved page, and +// the pages that stand in a thread's way. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xenforoParse.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + classifyForumPage, + embedTarget, + parseXenforoThreadPage, + parseXenforoThreadUrl, + resolveForumUrl, + xenforoPageUrl, + xenforoThreadHandle, +} from "./xenforoParse"; +import { parsePost } from "../lib/posts"; +import { + CAPTCHA_PAGE, + CHALLENGE_PAGE, + LOGIN_PAGE, + NOT_FOUND_PAGE, + ORIGIN, + RICH_BODY, + THREAD_URL, + threadPage, + type FakePost, +} from "./__fixtures__/xenforoPages"; + +const T0 = 1_760_000_000; // epoch seconds + +function post(id: number, position: number, body = `Post number ${position}.`, extra: Partial<FakePost> = {}): FakePost { + return { id, author: `Member${id % 5}`, userId: 100 + (id % 5), ts: T0 + position * 60, position, body, ...extra }; +} + +test("thread URLs: friendly, paged, prefixed and non-friendly forms", () => { + const t = parseXenforoThreadUrl(`${THREAD_URL}page-7`); + assert.deepEqual(t, { + origin: ORIGIN, + host: "forum.example", + base: THREAD_URL, + threadId: "4242", + page: 7, + }); + assert.equal(parseXenforoThreadUrl(`${THREAD_URL}post-99`)?.base, THREAD_URL); + assert.equal(parseXenforoThreadUrl("https://forum.example/community/threads/x.12/")?.base, "https://forum.example/community/threads/x.12/"); + assert.equal(parseXenforoThreadUrl("https://forum.example/index.php?threads/x.12/")?.threadId, "12"); + assert.equal(parseXenforoThreadUrl("https://forum.example/threads/12/")?.threadId, "12"); + assert.equal(parseXenforoThreadUrl("https://forum.example/members/someone.3/"), null); + assert.equal(parseXenforoThreadUrl("not a url"), null); + assert.equal(xenforoPageUrl({ base: THREAD_URL }, 1), THREAD_URL); + assert.equal(xenforoPageUrl({ base: THREAD_URL }, 3), `${THREAD_URL}page-3`); + assert.equal(xenforoThreadHandle(`${THREAD_URL}page-2`), "the-teapot-collectors-thread.4242"); +}); + +test("a page: thread metadata, page numbers and every post in page order", () => { + const html = threadPage({ page: 3, last: 9, posts: [post(301, 41), post(302, 42), post(303, 43)] }); + const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); + assert.equal(page.origin, ORIGIN); + assert.equal(page.host, "forum.example"); + assert.equal(page.threadId, "4242"); + assert.equal(page.threadTitle, "The Teapot Collectors Thread"); + assert.equal(page.threadUrl, THREAD_URL); + assert.equal(page.page, 3); + assert.equal(page.lastPage, 9); + assert.deepEqual(page.posts.map((p) => p.id), ["301", "302", "303"]); + const p = page.posts[1]; + assert.equal(p.slug, "teapots/302"); + assert.equal(p.platform, "xenforo"); + assert.equal(p.author, "Member2"); + assert.equal(p.createdAt, new Date((T0 + 42 * 60) * 1000).toISOString()); + assert.equal(p.uploadDate.length, 8); + assert.equal(p.url, `${ORIGIN}/posts/302/`); + assert.equal(p.text, "Post number 42."); + assert.equal(p.isReply, true); + assert.equal(p.isRepost, false); + assert.deepEqual(p.forum, { + host: "forum.example", + threadId: "4242", + page: 3, + threadTitle: "The Teapot Collectors Thread", + threadUrl: THREAD_URL, + position: 42, + authorId: "102", + }); + // Round-trips through the stored-record validator unchanged. + assert.deepEqual(parsePost(JSON.parse(JSON.stringify(p))), p); +}); + +test("positions with thousands separators, the first post, and a single-page thread", () => { + const html = threadPage({ page: 1, last: 1, posts: [post(1, 1, "Opening post."), post(2, 2101)] }); + const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); + assert.equal(page.page, 1); + assert.equal(page.lastPage, 1); + assert.equal(page.posts[0].forum?.position, 1); + assert.equal(page.posts[0].isReply, false); + assert.equal(page.posts[1].forum?.position, 2101); +}); + +test("the body: quotes, mentions, links, smilies, images, embeds, spoilers, video, unfurls, lists, attachments", () => { + const html = threadPage({ page: 2, last: 2, posts: [post(500, 30, RICH_BODY)] }); + const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; + const lines = p.text.split("\n"); + // The quote: a header naming the quoted member and post, then its lines. + assert.deepEqual(lines.slice(0, 3), [ + "> Quoting Marigold (post 1001):", + "> The blue one is a reproduction.", + "> Look at the glaze.", + ]); + assert.ok(!p.text.includes("Click to expand"), "the expand link is chrome"); + assert.ok(!p.text.includes("Marigold said:"), "the quote title is replaced by the header"); + assert.ok(p.text.includes("I disagree, @Marigold."), "a mention is its text"); + // A blank line survives <br><br>. + assert.ok(/@Marigold\.\n\nHere is the catalogue/.test(p.text)); + assert.ok(p.text.includes("Here is the catalogue: https://archive.example/AbCd1")); + assert.ok(p.text.includes("museum page (https://museum.example/teapots?id=9) :)")); + assert.ok(p.text.includes("[image]")); + assert.ok(p.text.includes("[embed: https://www.youtube.com/watch?v=AbCdEfGhIjK]")); + assert.ok(p.text.includes("[embed: https://x.com/i/status/1234567890123456789]")); + assert.ok(p.text.includes("[Spoiler: the ending]\nThe lid was glued on.\n[/Spoiler]")); + assert.ok(p.text.includes("[video]")); + assert.ok(p.text.includes("https://news.example/teapot-auction")); + assert.ok(!p.text.includes("An auction report"), "an unfurl card is its URL, not its blurb"); + assert.ok(p.text.includes("- first point\n- second point")); + assert.ok(!/ /.test(p.text), "no runs of spaces"); + + assert.deepEqual(p.forum?.quotes, [{ postId: "1001", author: "Marigold", authorId: "7" }]); + assert.deepEqual(p.quoted, { + platform: "xenforo", + id: "1001", + url: `${ORIGIN}/posts/1001/`, + author: "Marigold", + }); + assert.deepEqual(p.replyTo, p.quoted); + assert.deepEqual(p.links, [ + "https://archive.example/AbCd1", + "https://museum.example/teapots?id=9", + "https://www.youtube.com/watch?v=AbCdEfGhIjK", + "https://x.com/i/status/1234567890123456789", + "https://news.example/teapot-auction", + ]); + assert.deepEqual(p.media, [ + { kind: "image", url: "https://images.example/teapot.jpg", name: "teapot.jpg" }, + { kind: "embed", url: "https://www.youtube.com/watch?v=AbCdEfGhIjK", provider: "youtube" }, + { kind: "embed", url: "https://x.com/i/status/1234567890123456789", provider: "twitter" }, + { kind: "video", url: `${ORIGIN}/data/video/12/12345-abc.mp4` }, + { kind: "link-card", url: "https://news.example/teapot-auction" }, + { kind: "attachment", url: `${ORIGIN}/attachments/receipt-png.555/` }, + { kind: "image", url: `${ORIGIN}/data/attachments/0/555-receipt.jpg`, name: "receipt.png" }, + ]); + assert.equal(p.mediaCount, p.media?.length); +}); + +test("an edited post records when, and listed attachments are media", () => { + const html = threadPage({ + page: 1, + last: 1, + posts: [ + post(600, 5, "Edited words.", { + editedTs: T0 + 9999, + attachments: [{ href: "/attachments/scan-jpg.900/", name: "scan.jpg" }], + }), + ], + }); + const [p] = parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts; + assert.equal(p.forum?.editedAt, new Date((T0 + 9999) * 1000).toISOString()); + // The edit time is not the post time. + assert.equal(p.createdAt, new Date((T0 + 5 * 60) * 1000).toISOString()); + assert.deepEqual(p.media, [{ kind: "attachment", url: `${ORIGIN}/attachments/scan-jpg.900/`, name: "scan.jpg" }]); +}); + +test("a browser-saved page: no canonical link, rewritten assets, the saved-from comment names it", () => { + const html = threadPage({ page: 4, last: 6, saved: true, posts: [post(700, 61), post(701, 62)] }); + assert.equal(classifyForumPage(html), null); + const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); + assert.equal(page.origin, ORIGIN); + assert.equal(page.threadId, "4242"); + assert.equal(page.page, 4); + assert.equal(page.lastPage, 6); + assert.equal(page.posts.length, 2); + assert.equal(page.posts[0].url, `${ORIGIN}/posts/700/`); + // With no saved-from comment either, the caller's URL is the fallback, and + // the content key still names the thread. + const bare = html.replace(/<!-- saved from[^>]*-->/, ""); + const fromFallback = parseXenforoThreadPage(bare, { channelSlug: "teapots", pageUrl: THREAD_URL }); + assert.equal(fromFallback.origin, ORIGIN); + const noUrl = parseXenforoThreadPage(bare, { channelSlug: "teapots" }); + assert.equal(noUrl.threadId, "4242"); + assert.equal(noUrl.posts.length, 2); +}); + +test("URL resolution keeps saved-asset paths and absolutises forum paths", () => { + assert.equal(resolveForumUrl("/attachments/a.1/", ORIGIN), `${ORIGIN}/attachments/a.1/`); + assert.equal(resolveForumUrl("./Thread_files/a.jpg", ORIGIN), "./Thread_files/a.jpg"); + assert.equal(resolveForumUrl("Thread_files/a.jpg", ORIGIN), "Thread_files/a.jpg"); + assert.equal(resolveForumUrl("//cdn.example/x.png", ORIGIN), "https://cdn.example/x.png"); + assert.equal(resolveForumUrl("data:image/png;base64,AAAA", ORIGIN), undefined); + assert.equal(resolveForumUrl("#post-3", ORIGIN), undefined); + assert.deepEqual(embedTarget("https://rumble.com/embed/v1abc/?pub=4"), { + url: "https://rumble.com/embed/v1abc/", + provider: "rumble", + }); +}); + +test("pages in the way: a browser check, a captcha, a login, a missing thread, a refusal", () => { + const thread = threadPage({ page: 1, last: 1, posts: [post(1, 1)] }); + assert.equal(classifyForumPage(thread), null); + assert.equal(classifyForumPage(thread, 200), null); + + const challenge = classifyForumPage(CHALLENGE_PAGE, 403); + assert.equal(challenge?.kind, "challenge"); + assert.equal(challenge?.title, "Checking your browser"); + assert.equal(classifyForumPage(CAPTCHA_PAGE)?.kind, "captcha"); + assert.equal(classifyForumPage(LOGIN_PAGE)?.kind, "login"); + assert.equal(classifyForumPage(NOT_FOUND_PAGE)?.kind, "not-found"); + assert.equal(classifyForumPage("<html><body>nothing</body></html>", 429)?.kind, "blocked"); + assert.equal(classifyForumPage("<html><body>nothing</body></html>", 404)?.kind, "not-found"); + assert.equal(classifyForumPage("<html><body>You have been banned.</body></html>")?.kind, "blocked"); + assert.equal(classifyForumPage("<html><body>nothing</body></html>")?.kind, "unknown"); + // A check page that names a captcha in passing is still a check (waited + // out), not a captcha (stopped at once). + assert.equal( + classifyForumPage(CHALLENGE_PAGE.replace("</body>", "<p>No captcha needed.</p></body>"))?.kind, + "challenge", + ); +}); + +test("an article thread's first post, atop a later page with no #N, is post 1 of page 1", () => { + const html = threadPage({ page: 3, last: 4, article: post(1, 1, "The article."), posts: [post(41, 41), post(42, 42)] }); + const page = parseXenforoThreadPage(html, { channelSlug: "teapots" }); + assert.deepEqual(page.posts.map((p) => [p.id, p.forum?.position, p.forum?.page]), [ + ["1", 1, 1], + ["41", 41, 3], + ["42", 42, 3], + ]); + assert.equal(page.posts[0].isReply, false); +}); + +test("a message with no id or no date is not a post", () => { + const html = threadPage({ page: 1, last: 1, posts: [post(1, 1)] }) + .replace('data-timestamp="', 'data-x="') + .replace(/datetime="[^"]*"/, ""); + assert.equal(parseXenforoThreadPage(html, { channelSlug: "teapots" }).posts.length, 0); +}); diff --git a/common/social/xenforoParse.ts b/common/social/xenforoParse.ts @@ -0,0 +1,840 @@ +// XenForo 2 thread pages → `Post`s. The ONE parser: the live fetcher hands it +// the page a headless browser loaded (xenforoFetcher.ts), the import hands it +// a page the operator saved from their own browser (controller/importForumPages.ts), and +// both get the same posts. +// +// Pure: no browser, no network, no fs. The HTML is read with the small shared +// reader (htmlReader.ts), so every marker below runs in tests against +// synthetic pages. +// +// THE MARKUP (XenForo 2, as served): +// article.message[data-author][data-content="post-N"]#js-post-N +// (.message--article: an article thread's first post, repeated atop +// every page with no "#N") +// .message-user … a.username[data-user-id] the author +// header.message-attribution +// .message-attribution-main time.u-dt[data-timestamp] posted +// .message-attribution-opposite a "#N" position +// .bbWrapper the body +// blockquote.bbCodeBlock--quote[data-quote][data-source="post: N"] +// .bbCodeSpoiler, .bbCodeBlock--code, .bbImageWrapper, iframe, video… +// .message-attachments li.file a[href] attachments +// .message-lastEdit time.u-dt last edited +// h1.p-title-value thread title +// .pageNav-page(--current) page / last page +// +// A page SAVED from a browser has rewritten asset URLs (`./Thread_files/…`); +// the original absolute URL is kept wherever the markup still carries it +// (data-url, data-src, an anchor's href), else what is there is kept. + +import { + postPermalink, + uploadDateFromCreatedAt, + type ForumPostInfo, + type Post, + type PostMedia, + type PostRef, +} from "../lib/posts"; +import { XENFORO_THREAD_PATH_RE } from "../lib/detectPlatform.mjs"; +import { parseHtml, type HtmlElement, type HtmlNode } from "./htmlReader"; + +// --- thread URLs --------------------------------------------------------------- + +export type XenforoThreadUrl = { + origin: string; // "https://forum.example" + host: string; + // The thread's URL with no page: "https://forum.example/threads/a-title.123/" + base: string; + threadId: string; + // The page the URL names, when it names one (/page-N). + page?: number; +}; + +// A XenForo thread URL, read; null when the URL is not one. Accepts the +// friendly form (/threads/<slug>.<id>/, under any path prefix) with an +// optional /page-N and /post-N, and the non-friendly index.php?threads/… form. +export function parseXenforoThreadUrl(url: string): XenforoThreadUrl | null { + let u: URL; + try { + u = new URL(url.trim()); + } catch { + return null; + } + if (u.protocol !== "http:" && u.protocol !== "https:") return null; + const where = u.pathname + u.search; + const m = XENFORO_THREAD_PATH_RE.exec(where); + if (!m) return null; + const threadId = m[1]; + // Everything up to and including the thread segment (and its slash). + const end = m.index + m[0].length; + let basePath = where.slice(0, end); + if (!basePath.endsWith("/")) basePath += "/"; + const pageM = /(?:^|\/)page-(\d+)(?:\/|$|[?#])/.exec(where.slice(end)); + const out: XenforoThreadUrl = { + origin: u.origin, + host: u.hostname.toLowerCase(), + base: `${u.origin}${basePath}`, + threadId, + }; + if (pageM) out.page = Number(pageM[1]); + return out; +} + +// The URL of page `page` of a thread (page 1 is the bare thread URL). +export function xenforoPageUrl(t: Pick<XenforoThreadUrl, "base">, page: number): string { + return page <= 1 ? t.base : `${t.base}page-${page}`; +} + +// The thread key a channel uses as its handle: "<slug>.<id>" (or the id). +export function xenforoThreadHandle(url: string): string | null { + const t = parseXenforoThreadUrl(url); + if (!t) return null; + const seg = /threads\/([^/?#]+)\/?$/.exec(t.base); + return seg ? decodeURIComponent(seg[1]) : t.threadId; +} + +// --- the page's state: a thread, or something in its way ------------------------- + +export type ForumBlockKind = + // A JavaScript check (KiwiFlare's proof of work, a "just a moment" page) + // that a real browser normally clears by itself. + | "challenge" + // A captcha: a person has to answer it. + | "captcha" + // Refused outright: 403 / 429 / a ban or access-denied page. + | "blocked" + // The thread is only shown to a logged-in member. + | "login" + // The forum says the thread (or post) does not exist. + | "not-found" + // Not a thread page, and nothing above recognised. + | "unknown"; + +export type ForumBlock = { + kind: ForumBlockKind; + // What was recognised, for the log ("KiwiFlare marker", "HTTP 429"). + detail: string; + title?: string; +}; + +const POST_MARKER_RE = /data-content="post-\d+"|id="js-post-\d+"/; +const THREAD_TEMPLATE_RE = /data-template="thread_view"/; + +const CAPTCHA_MARKERS: [RegExp, string][] = [ + [/h-captcha|hcaptcha\.com/i, "hCaptcha"], + [/g-recaptcha|google\.com\/recaptcha|recaptcha\/api/i, "reCAPTCHA"], + [/cf-turnstile|challenges\.cloudflare\.com\/turnstile/i, "Turnstile"], +]; +// The bare word is weaker than a widget: a browser check's own page may name +// it, so it is read only after the check markers. +const CAPTCHA_WORD = /\bcaptcha\b/i; + +const CHALLENGE_MARKERS: [RegExp, string][] = [ + [/kiwiflare/i, "KiwiFlare"], + [/\/\.sssg\/|\bsssg[_-]/i, "KiwiFlare (sssg)"], + [/proof[- ]of[- ]work/i, "a proof-of-work check"], + [/checking your browser/i, "a browser check"], + [/just a moment\.\.\.|cf-chl|challenge-platform|cf_chl_/i, "Cloudflare's browser check"], + [/ddos-guard/i, "DDoS-Guard"], + [/please wait while (your request|we) .{0,40}verif/i, "a verification page"], +]; + +const BLOCK_TEXT: [RegExp, string][] = [ + [/you have been banned|your (ip|access) (has been|is) (banned|blocked)/i, "a ban page"], + [/access denied|error 1020|403 forbidden/i, "an access-denied page"], + [/too many requests|rate limit/i, "a rate-limit page"], +]; + +const LOGIN_TEXT = + /you must be logged[- ]in to do that|you do not have permission to view this page|data-template="login"/i; +const NOT_FOUND_TEXT = + /the requested (thread|post|page) could not be found|data-template="error"[^>]*>[\s\S]{0,4000}could not be found/i; + +function titleOf(html: string): string | undefined { + const m = /<title[^>]*>([\s\S]*?)<\/title>/i.exec(html); + const t = m?.[1].replace(/\s+/g, " ").trim(); + return t || undefined; +} + +// What stands between this page and the thread, or null when it IS a thread +// page (it carries posts, or XenForo's thread template). `status` is the HTTP +// status the page came with, when known. +export function classifyForumPage(html: string, status?: number): ForumBlock | null { + if (POST_MARKER_RE.test(html) || THREAD_TEMPLATE_RE.test(html)) return null; + const title = titleOf(html); + const withTitle = (b: Omit<ForumBlock, "title">): ForumBlock => (title ? { ...b, title } : b); + for (const [re, what] of CAPTCHA_MARKERS) { + if (re.test(html)) return withTitle({ kind: "captcha", detail: what }); + } + for (const [re, what] of CHALLENGE_MARKERS) { + if (re.test(html)) return withTitle({ kind: "challenge", detail: what }); + } + if (CAPTCHA_WORD.test(html)) return withTitle({ kind: "captcha", detail: "a captcha" }); + if (LOGIN_TEXT.test(html)) return withTitle({ kind: "login", detail: "the forum asks for a login" }); + if (NOT_FOUND_TEXT.test(html) || status === 404) { + return withTitle({ kind: "not-found", detail: status === 404 ? "HTTP 404" : "the forum says it could not be found" }); + } + if (status === 401 || status === 403 || status === 429 || status === 503) { + return withTitle({ kind: "blocked", detail: `HTTP ${status}` }); + } + for (const [re, what] of BLOCK_TEXT) { + if (re.test(html)) return withTitle({ kind: "blocked", detail: what }); + } + return withTitle({ kind: "unknown", detail: "no forum posts on the page" }); +} + +// --- tree helpers ------------------------------------------------------------------ + +const isEl = (n: HtmlNode): n is HtmlElement => typeof n !== "string"; +const classList = (el: HtmlElement): string[] => (el.attrs.class ?? "").split(/\s+/).filter(Boolean); +const hasClass = (el: HtmlElement, c: string) => classList(el).includes(c); +const hasClassPrefix = (el: HtmlElement, p: string) => classList(el).some((c) => c.startsWith(p)); + +function findAll( + el: HtmlElement, + pred: (e: HtmlElement) => boolean, + out: HtmlElement[] = [], + skip?: (e: HtmlElement) => boolean, +): HtmlElement[] { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) out.push(c); + if (skip?.(c)) continue; + findAll(c, pred, out, skip); + } + return out; +} + +function findFirst( + el: HtmlElement, + pred: (e: HtmlElement) => boolean, + skip?: (e: HtmlElement) => boolean, +): HtmlElement | null { + for (const c of el.children) { + if (!isEl(c)) continue; + if (pred(c)) return c; + if (skip?.(c)) continue; + const hit = findFirst(c, pred, skip); + if (hit) return hit; + } + return null; +} + +// Plain text of an element: whitespace collapsed, scripts dropped. +function plainText(node: HtmlNode): string { + if (!isEl(node)) return node; + if (node.tag === "script" || node.tag === "style" || node.tag === "template") return ""; + return node.children.map(plainText).join(""); +} +const squash = (s: string) => s.replace(/[\s\u00a0]+/g, " ").trim(); + +// --- URLs ------------------------------------------------------------------------ + +// A URL from the markup, absolute against the forum's origin. A path a browser +// wrote when it SAVED the page (`./Thread_files/x.jpg`, a `file:` URL) is kept +// as it is: it names nothing on the forum. +export function resolveForumUrl(raw: string | undefined, origin: string | undefined): string | undefined { + const v = raw?.trim(); + if (!v || v.startsWith("data:") || v.startsWith("javascript:") || v.startsWith("#")) return undefined; + if (/^[a-z][a-z0-9+.-]*:/i.test(v)) return v; + if (v.startsWith("//")) return `https:${v}`; + if (v.startsWith("./") || v.startsWith("../") || /_files\//.test(v)) return v; + if (!origin) return v; + try { + return new URL(v, `${origin}/`).href; + } catch { + return v; + } +} + +// The canonical page URL an embed's iframe stands for, and its provider. +export function embedTarget(src: string, provider?: string): { url: string; provider?: string } { + const yt = /youtube(?:-nocookie)?\.com\/embed\/([A-Za-z0-9_-]{6,})/.exec(src); + if (yt) return { url: `https://www.youtube.com/watch?v=${yt[1]}`, provider: "youtube" }; + // s9e's media embeds: an iframe page with the item id in the fragment. + const s9e = /s9e\.github\.io\/iframe\/\d+\/([a-z0-9]+)(?:\.min)?\.html#([^&?]+)/i.exec(src); + if (s9e) { + const kind = s9e[1].toLowerCase(); + const id = decodeURIComponent(s9e[2]); + if (kind === "twitter" && /^\d+$/.test(id)) { + return { url: `https://x.com/i/status/${id}`, provider: "twitter" }; + } + if (kind === "youtube") return { url: `https://www.youtube.com/watch?v=${id}`, provider: "youtube" }; + return { url: src, provider: provider ?? kind }; + } + const tw = /platform\.twitter\.com\/embed\/.*[?&]id=(\d+)/.exec(src); + if (tw) return { url: `https://x.com/i/status/${tw[1]}`, provider: "twitter" }; + const rumble = /rumble\.com\/embed\/([A-Za-z0-9]+)/.exec(src); + if (rumble) return { url: `https://rumble.com/embed/${rumble[1]}/`, provider: "rumble" }; + return provider ? { url: src, provider } : { url: src }; +} + +// --- the body → text ------------------------------------------------------------------ + +type BodyCtx = { + origin?: string; + links: string[]; + media: PostMedia[]; + quotes: NonNullable<ForumPostInfo["quotes"]>; +}; + +const BLOCK_TAGS = new Set([ + "address", "article", "aside", "blockquote", "dd", "details", "div", "dl", + "dt", "figcaption", "figure", "footer", "form", "h1", "h2", "h3", "h4", "h5", + "h6", "header", "hr", "li", "main", "nav", "ol", "p", "pre", "section", + "summary", "table", "tbody", "td", "tfoot", "th", "thead", "tr", "ul", +]); +const SILENT_TAGS = new Set(["script", "style", "noscript", "template", "svg", "button", "input", "select", "canvas"]); + +function addLink(ctx: BodyCtx, url: string | undefined) { + if (url && /^https?:\/\//i.test(url) && !ctx.links.includes(url)) ctx.links.push(url); +} +function addMedia(ctx: BodyCtx, m: PostMedia) { + if (!ctx.media.some((x) => x.url === m.url && x.kind === m.kind)) ctx.media.push(m); +} + +const isQuote = (el: HtmlElement) => + el.tag === "blockquote" && (hasClass(el, "bbCodeBlock--quote") || "data-quote" in el.attrs); +const isSpoiler = (el: HtmlElement) => hasClass(el, "bbCodeSpoiler") || hasClass(el, "bbCodeInlineSpoiler"); +const isUnfurl = (el: HtmlElement) => + hasClass(el, "bbCodeBlock--unfurl") || (el.attrs["data-unfurl"] === "true" && !!el.attrs["data-url"]); +const isImageWrapper = (el: HtmlElement) => hasClass(el, "bbImageWrapper"); +const isSmilie = (el: HtmlElement) => + el.tag === "img" && (hasClass(el, "smilie") || hasClassPrefix(el, "smilie--") || "data-shortname" in el.attrs); + +function imageUrl(el: HtmlElement, ctx: BodyCtx, wrapper?: HtmlElement): string | undefined { + return resolveForumUrl( + el.attrs["data-url"] || wrapper?.attrs["data-src"] || el.attrs["data-src"] || el.attrs.src, + ctx.origin, + ); +} + +// A quote's source post id: data-source="post: 123", else the jump link. +function quoteSource(el: HtmlElement): { postId?: string; author?: string; authorId?: string } { + const out: { postId?: string; author?: string; authorId?: string } = {}; + const src = /post:\s*(\d+)/.exec(el.attrs["data-source"] ?? ""); + if (src) out.postId = src[1]; + if (!out.postId) { + const jump = findFirst(el, (e) => e.tag === "a" && hasClass(e, "bbCodeBlock-sourceJump")); + const m = + /[?&]id=(\d+)/.exec(jump?.attrs.href ?? "") ?? + /post-(\d+)/.exec(jump?.attrs["data-content-selector"] ?? "") ?? + /\/posts\/(\d+)/.exec(jump?.attrs.href ?? ""); + if (m) out.postId = m[1]; + } + const author = squash(el.attrs["data-quote"] ?? ""); + if (author) out.author = author; + const member = /member:\s*(\d+)/.exec(el.attrs["data-attributes"] ?? ""); + if (member) out.authorId = member[1]; + return out; +} + +// A block boundary: a line break that merges with any break beside it, where +// "\n" (a <br>) is a break of its own — so <br><br> is a blank line and a +// list item is not. +const SOFT = "\u0000"; + +// Render one body subtree. Text is emitted with "\n" for <br> and SOFT around +// blocks; `finish` turns the run into lines. +function render(node: HtmlNode, ctx: BodyCtx, out: string[]): void { + if (!isEl(node)) { + out.push(node.replace(/[\s\u00a0]+/g, " ")); + return; + } + const el = node; + if (SILENT_TAGS.has(el.tag) && !(el.tag === "button" && isSpoilerButton(el))) return; + if (hasClass(el, "bbCodeBlock-expandLink") || hasClass(el, "bbCodeBlock-title")) return; + if (el.tag === "br") { + out.push("\n"); + return; + } + + if (isQuote(el)) { + const src = quoteSource(el); + ctx.quotes.push(src); + const inner: string[] = []; + const content = findFirst(el, (e) => hasClass(e, "bbCodeBlock-content")) ?? el; + for (const c of content.children) render(c, ctx, inner); + const body = finish(inner); + const head = + `Quoting ${src.author ?? "an earlier post"}` + (src.postId ? ` (post ${src.postId})` : "") + ":"; + const lines = [head, ...(body ? body.split("\n") : [])]; + out.push(SOFT, lines.map((l) => (l ? `> ${l}` : ">")).join("\n"), SOFT); + return; + } + + if (isSpoiler(el)) { + const titleEl = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-button-title")); + const title = titleEl ? squash(plainText(titleEl)) : ""; + const content = findFirst(el, (e) => hasClass(e, "bbCodeSpoiler-content") || hasClass(e, "bbCodeBlock-content")) ?? el; + const inner: string[] = []; + for (const c of content.children) render(c, ctx, inner); + const inline = hasClass(el, "bbCodeInlineSpoiler"); + if (inline) { + out.push(`[spoiler: ${finish(inner).replace(/\n+/g, " ")}]`); + return; + } + out.push(SOFT, `[Spoiler${title && !/^spoiler$/i.test(title) ? `: ${title}` : ""}]`, "\n", finish(inner), "\n", "[/Spoiler]", SOFT); + return; + } + + if (isUnfurl(el)) { + const url = resolveForumUrl(el.attrs["data-url"], ctx.origin); + if (url) { + addLink(ctx, url); + addMedia(ctx, { kind: "link-card", url }); + out.push(SOFT, url, SOFT); + } + return; + } + + if (isImageWrapper(el)) { + const img = findFirst(el, (e) => e.tag === "img"); + const url = img ? imageUrl(img, ctx, el) : resolveForumUrl(el.attrs["data-src"], ctx.origin); + if (url) { + const name = el.attrs.title || img?.attrs.alt; + addMedia(ctx, { kind: "image", url, ...(name ? { name } : {}) }); + } + out.push("[image]"); + return; + } + + if (el.tag === "img") { + if (isSmilie(el)) { + out.push(el.attrs.alt ?? ""); + return; + } + const url = imageUrl(el, ctx); + if (url) addMedia(ctx, { kind: "image", url, ...(el.attrs.alt ? { name: el.attrs.alt } : {}) }); + out.push("[image]"); + return; + } + + if (el.tag === "video" || el.tag === "audio") { + const src = + el.attrs.src ?? + findFirst(el, (e) => e.tag === "source" && !!e.attrs.src)?.attrs.src; + const url = resolveForumUrl(src, ctx.origin); + if (url) addMedia(ctx, { kind: "video", url }); + out.push(el.tag === "video" ? "[video]" : "[audio]"); + return; + } + + if (el.tag === "iframe" || "data-s9e-mediaembed" in el.attrs) { + const frame = el.tag === "iframe" ? el : findFirst(el, (e) => e.tag === "iframe"); + const provider = el.attrs["data-s9e-mediaembed"] || frame?.attrs["data-s9e-mediaembed"] || undefined; + const raw = + frame?.attrs["data-s9e-mediaembed-src"] || frame?.attrs.src || frame?.attrs["data-src"] || + el.attrs["data-s9e-mediaembed-src"]; + const src = resolveForumUrl(raw, ctx.origin); + if (src && /^https?:/i.test(src)) { + const target = embedTarget(src, provider); + addMedia(ctx, { kind: "embed", url: target.url, ...(target.provider ? { provider: target.provider } : {}) }); + addLink(ctx, target.url); + out.push(SOFT, `[embed: ${target.url}]`, SOFT); + } else if (frame || provider) { + out.push("[embed]"); + } + return; + } + + if (el.tag === "blockquote" && hasClass(el, "twitter-tweet")) { + const link = findAll(el, (e) => e.tag === "a" && /\/status\/\d+/.test(e.attrs.href ?? "")).pop(); + const url = link?.attrs.href; + if (url) { + addMedia(ctx, { kind: "embed", url, provider: "twitter" }); + addLink(ctx, url); + out.push(SOFT, `[embed: ${url}]`, SOFT); + } + return; + } + + if (el.tag === "a") { + renderLink(el, ctx, out); + return; + } + + if (el.tag === "pre" || el.tag === "code") { + if (el.tag === "pre" || hasClass(el, "bbCodeCode")) { + out.push(SOFT, plainText(el).replace(/\u00a0/g, " ").replace(/^\n+|\n+$/g, ""), SOFT); + return; + } + } + + const block = BLOCK_TAGS.has(el.tag); + if (block) out.push(SOFT); + if (el.tag === "li") out.push("- "); + for (const c of el.children) render(c, ctx, out); + if (block) out.push(SOFT); +} + +function isSpoilerButton(el: HtmlElement): boolean { + return hasClass(el, "bbCodeSpoiler-button"); +} + +function renderLink(el: HtmlElement, ctx: BodyCtx, out: string[]): void { + const href = el.attrs.href; + // A member mention or an in-forum jump: its text is the content. + const mention = hasClass(el, "username") || "data-user-id" in el.attrs; + const internalJump = !href || href.startsWith("#") || /\/goto\/post/.test(href); + const hasImg = !!findFirst(el, (e) => e.tag === "img" && !isSmilie(e)); + if (mention || internalJump) { + for (const c of el.children) render(c, ctx, out); + return; + } + const url = resolveForumUrl(el.attrs["data-url"] || href, ctx.origin); + if (hasImg) { + // An image that links somewhere: an attachment's full-size page, or a + // clickable image. The attachment link is the medium; the image inside is + // rendered (and recorded) as usual. + if (url && /\/attachments\//.test(url)) addMedia(ctx, { kind: "attachment", url }); + else addLink(ctx, url); + for (const c of el.children) render(c, ctx, out); + return; + } + const inner: string[] = []; + for (const c of el.children) render(c, ctx, inner); + const text = squash(inner.join("")); + if (!url) { + out.push(text); + return; + } + if (/\/attachments\//.test(url)) addMedia(ctx, { kind: "attachment", url, ...(text ? { name: text } : {}) }); + else addLink(ctx, url); + const bare = text.replace(/(…|\.\.\.)$/, ""); + if (!text || url === text || (bare.length > 8 && url.includes(bare)) || text === href) { + out.push(url); + } else { + out.push(`${text} (${url})`); + } +} + +// Emitted pieces → lines: each line trimmed, at most one blank line in a row, +// none at the ends. +function finish(pieces: string[]): string { + // Every run of breaks (soft or hard, with the spaces between them) becomes + // one line break, or a blank line when it holds two or more hard ones. + const joined = pieces + .join("") + .replace(/[ \t]*[\u0000\n][\u0000\n \t]*/g, (run) => + (run.match(/\n/g)?.length ?? 0) >= 2 ? "\n\n" : "\n", + ); + const lines = joined.split("\n").map((l) => l.replace(/ {2,}/g, " ").trim()); + const out: string[] = []; + for (const l of lines) { + if (!l && (out.length === 0 || out[out.length - 1] === "")) continue; + out.push(l); + } + while (out.length && out[out.length - 1] === "") out.pop(); + return out.join("\n"); +} + +// A post body element → text, links, media and quotes. +export function forumBodyToText( + body: HtmlElement, + origin?: string, +): { text: string; links: string[]; media: PostMedia[]; quotes: NonNullable<ForumPostInfo["quotes"]> } { + const ctx: BodyCtx = { origin, links: [], media: [], quotes: [] }; + const pieces: string[] = []; + for (const c of body.children) render(c, ctx, pieces); + return { text: finish(pieces), links: ctx.links, media: ctx.media, quotes: ctx.quotes }; +} + +// --- times -------------------------------------------------------------------------- + +// A XenForo <time>: data-timestamp (epoch seconds) is exact; datetime +// ("2024-01-02T03:04:05-0500") is the fallback. +export function forumTimeIso(el: HtmlElement | null | undefined): string | undefined { + if (!el) return undefined; + const ts = Number(el.attrs["data-timestamp"] ?? el.attrs["data-time"]); + if (Number.isFinite(ts) && ts > 0) return new Date(ts * 1000).toISOString(); + const dt = el.attrs.datetime; + if (dt) { + // "-0500" → "-05:00", which every Date parser takes. + const norm = dt.replace(/([+-]\d{2})(\d{2})$/, "$1:$2"); + const ms = Date.parse(norm); + if (Number.isFinite(ms)) return new Date(ms).toISOString(); + } + return undefined; +} + +const isTime = (e: HtmlElement) => e.tag === "time" && (hasClass(e, "u-dt") || !!e.attrs["data-timestamp"] || !!e.attrs.datetime); + +// --- a whole page --------------------------------------------------------------------- + +export type XenforoThreadPage = { + origin?: string; + host?: string; + threadId?: string; + threadTitle?: string; + // The thread's URL without a page. + threadUrl?: string; + page: number; + lastPage: number; + // In page order (oldest first). + posts: Post[]; +}; + +// A post: message--post, or an article thread's starter (message--article). +const isMessage = (e: HtmlElement) => + e.tag === "article" && + hasClass(e, "message") && + (/^post-\d+$/.test(e.attrs["data-content"] ?? "") || /^js-post-\d+$/.test(e.attrs.id ?? "")); + +function postIdOf(el: HtmlElement): string | undefined { + return ( + /^post-(\d+)$/.exec(el.attrs["data-content"] ?? "")?.[1] ?? + /^js-post-(\d+)$/.exec(el.attrs.id ?? "")?.[1] ?? + /\/posts\/(\d+)/.exec(el.attrs.itemid ?? "")?.[1] + ); +} + +function metaContent(root: HtmlElement, pred: (e: HtmlElement) => boolean): string | undefined { + return findFirst(root, (e) => e.tag === "meta" && pred(e))?.attrs.content; +} + +// Where the page came from: its canonical link, og:url, the comment a browser +// writes into a saved page, else the caller's URL. +function pageUrlOf(root: HtmlElement, html: string, fallback?: string): string | undefined { + const canonical = findFirst(root, (e) => e.tag === "link" && (e.attrs.rel ?? "").split(/\s+/).includes("canonical"))?.attrs.href; + if (canonical && /^https?:\/\//.test(canonical)) return canonical; + const og = metaContent(root, (e) => e.attrs.property === "og:url"); + if (og && /^https?:\/\//.test(og)) return og; + const saved = /<!--\s*saved from url=\(\d+\)(https?:\/\/[^\s>]+)\s*-->/i.exec(html.slice(0, 4096)); + if (saved) return saved[1]; + return fallback; +} + +function threadTitleOf(root: HtmlElement): string | undefined { + const h1 = findFirst(root, (e) => e.tag === "h1" && hasClass(e, "p-title-value")); + if (h1) { + const text = squash( + h1.children + .map((c) => (isEl(c) && (hasClass(c, "label") || hasClass(c, "label-append") || hasClass(c, "labelLink")) ? "" : plainText(c))) + .join(""), + ); + if (text) return text; + } + const og = metaContent(root, (e) => e.attrs.property === "og:title"); + if (og) return squash(og); + const title = findFirst(root, (e) => e.tag === "title"); + const t = title ? squash(plainText(title)) : ""; + return t ? t.replace(/\s+\|\s+[^|]+$/, "") : undefined; +} + +function pageNumbers(root: HtmlElement): { current?: number; last?: number } { + const nav = findAll(root, (e) => hasClass(e, "pageNav-page")); + let current: number | undefined; + let last: number | undefined; + for (const li of nav) { + const n = Number(squash(plainText(li)).replace(/[^\d]/g, "")); + if (!Number.isFinite(n) || n < 1) continue; + if (hasClass(li, "pageNav-page--current")) current ??= n; + last = Math.max(last ?? 0, n); + } + // The page-jump box carries the last page as its max. + for (const input of findAll(root, (e) => e.tag === "input" && hasClass(e, "js-pageJumpPage"))) { + const max = Number(input.attrs.max); + if (Number.isFinite(max) && max >= 1) last = Math.max(last ?? 0, max); + } + // The compact nav ("3 of 120"). + if (last === undefined) { + const simple = findFirst(root, (e) => hasClass(e, "pageNavSimple-el--current")); + const m = simple ? /(\d+)\s+of\s+(\d+)/i.exec(squash(plainText(simple))) : null; + if (m) { + current ??= Number(m[1]); + last = Number(m[2]); + } + } + return { current, last }; +} + +export type ParseXenforoOptions = { + channelSlug: string; + // The URL the page was loaded from (the live fetcher) — a fallback when the + // page names none itself. + pageUrl?: string; +}; + +export function parseXenforoThreadPage(html: string, opts: ParseXenforoOptions): XenforoThreadPage { + const root = parseHtml(html); + const pageUrl = pageUrlOf(root, html, opts.pageUrl); + const thread = pageUrl ? parseXenforoThreadUrl(pageUrl) : null; + const fallbackThread = opts.pageUrl ? parseXenforoThreadUrl(opts.pageUrl) : null; + const t = thread ?? fallbackThread; + let origin = t?.origin; + if (!origin && pageUrl) { + try { + origin = new URL(pageUrl).origin; + } catch { + /* no origin */ + } + } + const host = origin ? new URL(origin).hostname.toLowerCase() : undefined; + const htmlEl = findFirst(root, (e) => e.tag === "html"); + const contentKey = /^thread-(\d+)$/.exec(htmlEl?.attrs["data-content-key"] ?? "")?.[1]; + const threadId = t?.threadId ?? contentKey; + const threadTitle = threadTitleOf(root); + const nums = pageNumbers(root); + const page = nums.current ?? t?.page ?? 1; + const lastPage = Math.max(page, nums.last ?? page); + + const posts: Post[] = []; + const seen = new Set<string>(); + for (const msg of findAll(root, isMessage, [], isMessage)) { + const post = messageToPost(msg, { + channelSlug: opts.channelSlug, + origin, + host, + threadId, + threadTitle, + threadUrl: t?.base, + page, + }); + if (post && !seen.has(post.id)) { + seen.add(post.id); + posts.push(post); + } + } + return { + ...(origin ? { origin } : {}), + ...(host ? { host } : {}), + ...(threadId ? { threadId } : {}), + ...(threadTitle ? { threadTitle } : {}), + ...(t?.base ? { threadUrl: t.base } : {}), + page, + lastPage, + posts, + }; +} + +type MessageCtx = { + channelSlug: string; + origin?: string; + host?: string; + threadId?: string; + threadTitle?: string; + threadUrl?: string; + page: number; +}; + +// One article.message → a Post, or null when it carries no id or no date (a +// deleted-post placeholder, an ad slot dressed as a message). +export function messageToPost(msg: HtmlElement, ctx: MessageCtx): Post | null { + const id = postIdOf(msg); + if (!id) return null; + const inBody = (e: HtmlElement) => hasClass(e, "bbWrapper") || hasClass(e, "message-body"); + const isLastEdit = (e: HtmlElement) => hasClass(e, "message-lastEdit"); + const attribution = + findFirst(msg, (e) => hasClass(e, "message-attribution-main")) ?? + findFirst(msg, (e) => hasClass(e, "message-attribution")); + const time = + (attribution ? findFirst(attribution, isTime) : null) ?? + findFirst(msg, isTime, (e) => inBody(e) || isLastEdit(e)); + const createdAt = forumTimeIso(time); + if (!createdAt) return null; + + const userLink = + findFirst(msg, (e) => hasClass(e, "username") && !!e.attrs["data-user-id"], inBody) ?? + findFirst(msg, (e) => !!e.attrs["data-user-id"], inBody); + const nameEl = findFirst(msg, (e) => hasClass(e, "message-name"), inBody); + const author = squash(msg.attrs["data-author"] ?? "") || (nameEl ? squash(plainText(nameEl)) : "") || (userLink ? squash(plainText(userLink)) : ""); + const authorId = userLink?.attrs["data-user-id"]; + + let position: number | undefined; + const opposite = findFirst(msg, (e) => hasClass(e, "message-attribution-opposite")); + for (const a of findAll(opposite ?? msg, (e) => e.tag === "a", [], inBody)) { + const m = /^#\s*([\d,]+)$/.exec(squash(plainText(a))); + if (m) { + position = Number(m[1].replace(/,/g, "")); + break; + } + } + + // An ARTICLE thread shows its first post (message--article) at the top of + // every page, with no "#N": it is post 1, of page 1, wherever it is read. + const articleStarter = hasClass(msg, "message--article"); + if (articleStarter && position === undefined) position = 1; + + const bodyEl = findFirst(msg, (e) => hasClass(e, "bbWrapper")); + const body = bodyEl + ? forumBodyToText(bodyEl, ctx.origin) + : { text: "", links: [], media: [], quotes: [] }; + + // Attachments listed under the post (not inline in the body). + const attachments = findFirst(msg, (e) => hasClass(e, "message-attachments")); + if (attachments) { + for (const li of findAll(attachments, (e) => e.tag === "li" && hasClass(e, "file"))) { + const a = + findFirst(li, (e) => e.tag === "a" && /\/attachments\//.test(e.attrs.href ?? "")) ?? + findFirst(li, (e) => e.tag === "a" && !!e.attrs.href && !hasClass(e, "u-anchorTarget")); + const url = resolveForumUrl(a?.attrs.href, ctx.origin); + if (!url) continue; + const nameEl2 = findFirst(li, (e) => hasClass(e, "file-name")); + const name = nameEl2 ? nameEl2.attrs.title || squash(plainText(nameEl2)) : undefined; + if (!body.media.some((m) => m.url === url)) { + body.media.push({ kind: "attachment", url, ...(name ? { name } : {}) }); + } + } + } + + const editedAt = forumTimeIso(findFirst(findFirst(msg, isLastEdit) ?? { tag: "#", attrs: {}, children: [] }, isTime)); + + const itemid = msg.attrs.itemid; + const url = + itemid && /^https?:\/\//.test(itemid) + ? itemid + : ctx.origin + ? postPermalink("xenforo", author, id, ctx.origin) + : `/posts/${id}/`; + + const forum: ForumPostInfo = { + host: ctx.host ?? "", + threadId: ctx.threadId ?? "", + page: articleStarter && position === 1 ? 1 : ctx.page, + }; + if (ctx.threadTitle) forum.threadTitle = ctx.threadTitle; + if (ctx.threadUrl) forum.threadUrl = ctx.threadUrl; + if (position) forum.position = position; + if (authorId && authorId !== "0") forum.authorId = authorId; + if (editedAt) forum.editedAt = editedAt; + if (body.quotes.length > 0) forum.quotes = body.quotes; + + const firstQuoted = body.quotes.find((q) => q.postId); + const quotedRef: PostRef | undefined = firstQuoted?.postId + ? { + platform: "xenforo", + id: firstQuoted.postId, + ...(ctx.origin ? { url: postPermalink("xenforo", "", firstQuoted.postId, ctx.origin) } : {}), + ...(firstQuoted.author ? { author: firstQuoted.author } : {}), + } + : undefined; + + const post: Post = { + id, + slug: `${ctx.channelSlug}/${id}`, + channelSlug: ctx.channelSlug, + author, + createdAt, + uploadDate: uploadDateFromCreatedAt(createdAt), + text: body.text, + url, + platform: "xenforo", + // Every post after the thread's first answers the thread. + isReply: position !== 1, + isRepost: false, + links: body.links, + }; + if (quotedRef) { + post.quoted = quotedRef; + post.replyTo = quotedRef; + } + if (body.media.length > 0) { + post.media = body.media; + post.mediaCount = body.media.length; + } + // `forum` is meaningful only with a host and a thread; a page that names + // neither (a bare fragment) still yields the post. + if (forum.host && forum.threadId) post.forum = forum; + return post; +}