commit 78e5b101e8570e4156cb51435536a06fbf30dd43
parent 18582324729636b2e40cef5a77a2c7e713e0b12a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 14:59:31 -0400
editor, mcp, docs: forum-thread channels — create, fetch latest pages, Connect, import
The channel form takes a XenForo thread URL (slug from the thread title,
fetcher xenforo-thread). The posts panel of a forum-thread channel gets a
"Latest N pages" cap on Fetch, a Forum session section with Connect (a headed
window on the host's profile, at the thread) and Import saved pages (a path on
the editor's machine, run as an import-forum-pages job on the platform queue).
fetch-posts carries `pages` through its spec and replay. MCP get_post shows a
forum post's thread, position, edit time, quotes and media; get_thread gives
its conversation. CHANNEL.md (postPagePauseSeconds) and ENVIRONMENT.md
regenerated; README section on posts sources; changelogs.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
17 files changed, 375 insertions(+), 20 deletions(-)
diff --git a/CHANNEL.md b/CHANNEL.md
@@ -18,6 +18,7 @@ Regenerate this file with `pnpm --filter yt-dlp-transcript-common exec tsx bin/f
| `sourceKind` | config | What KIND of source this is: `"video"` (default — yt-dlp + transcription) or `"social"` (an account fetched into the posts corpus, skipped by the video scan). A separate axis from `handling`, so every binary handling branch stays binary. |
| `postFetcher` | config | Social channels only: which social fetcher drives ingest (e.g. `"bluesky-atproto"`, `"x-gallery-dl"`). Absent = resolve by URL detection. Trimmed. |
| `socialHandle` | config | Social channels only: the bare account handle (a leading "@" is stripped). Derived from `url` at creation but stored, so a later URL-format change upstream cannot silently re-point ingest at a different account. |
+| `postPagePauseSeconds` | config | Social channels that are read page by page (a forum thread) only: the pause between two page loads, in seconds; each pause is jittered to 0.85–1.65× of it. Absent = the fetcher's own (12 s, so 10–20 s); floored at 5, capped at 600. |
| `platform` | config | The source platform (youtube, rumble, …). An unknown value is dropped. |
| `name` | config | Display name. |
| `url` | config | The channel / playlist / account URL syncs enumerate. Absent = the channel is never auto-synced. |
diff --git a/ENVIRONMENT.md b/ENVIRONMENT.md
@@ -71,7 +71,7 @@ Tokens, credentials and knobs a running process reads. Most configuration is not
| `OLLAMA_DIGEST_MODEL` | `qwen2.5:7b` | The ollama model the local digest lane asks for when settings name none. | common/lib/digestApps.ts |
| `CLAUDE_DIGEST_MODEL` | the CLI's default | The model the metered digest lane asks `claude` for when settings name none. | common/lib/digestApps.ts |
| `NITTER_INSTANCES` | a built-in list | Comma-separated Nitter instances for the X fallback fetcher, in order of preference. | common/social/xNitterFetcher.ts |
-| `ARCHILYZER_X_BROWSER` | the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium | The Chromium-family browser /settings' "Connect X account" opens (a path, or a name looked up on PATH). It is launched without the automation signals, in the X session profile; a value that is not an executable refuses the connect rather than opening another browser. | common/social/xBrowser.ts |
+| `ARCHILYZER_X_BROWSER` | the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium | The Chromium-family browser /settings' "Connect X account" opens (a path, or a name looked up on PATH), and a forum-thread channel's "Connect forum session" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser. | common/social/xBrowser.ts |
| `UMTOOL_URL` | unset (no link) | umtool's front door; when set, the video page links to it. | editor/app/channels/[slug]/videos/[id]/page.tsx |
| `TRANSCRIPT_SITE_URL` | — | MCP server: one published archive to read over HTTP. | mcp/src/sources.ts |
| `TRANSCRIPT_HUB_URL` | — | MCP server: a hub, federating every archive it lists. | mcp/src/sources.ts |
diff --git a/README.md b/README.md
@@ -452,6 +452,22 @@ Three programs share one library and one pile of data.
Two more pieces round out the workspace: the **project site** (`homepage/`) and the
**MCP server** (`mcp/`). Full layout in [CONTRIBUTING.md](CONTRIBUTING.md).
+## Posts: X, Bluesky and forum threads
+
+Beside video channels, a channel can be a **posts source**: an X or Bluesky account, or
+a **forum thread** (XenForo — Kiwi Farms is the first host, any XenForo 2 forum works the
+same way). Paste the account's or the thread's URL into the new-channel form; the posts
+are fetched into the corpus and searched alongside the transcripts.
+
+A forum thread is read in a headless browser, newest page first, one page at a time with
+a 10–20 s pause, on a browser profile kept per forum host. A browser check (Kiwi Farms'
+KiwiFlare) normally clears by itself and stays cleared; when it does not, or the thread
+needs a login, the run stops and says so, and **Connect forum session** on the channel
+page opens the profile in a window for you to clear it. Pages you saved from your own
+browser can be imported instead: `pnpm archilyzer posts import-html <slug> <file-or-dir>…`
+(or **Import saved pages** on the channel page). `pnpm archilyzer posts fetch --slug
+<slug> --pages N` reads just the latest N pages.
+
## Where your data lives
Transcripts and per-channel state live at `<repo>/transcripts/` — **its own git repo**,
diff --git a/common/lib/channelConfigSchema.test.ts b/common/lib/channelConfigSchema.test.ts
@@ -39,7 +39,7 @@ test("one key list: docs = coercions = schema shape, sync-state keys inside it",
assert.deepEqual(Object.keys(CHANNEL_CONFIG_COERCIONS), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(channelConfigObjectSchema.shape), [...CHANNEL_CONFIG_KEYS]);
assert.deepEqual(Object.keys(CHANNEL_CONFIG_FIELD_DOCS), [...CHANNEL_CONFIG_KEYS]);
- assert.equal(CHANNEL_CONFIG_KEYS.length, 31);
+ assert.equal(CHANNEL_CONFIG_KEYS.length, 32);
assert.equal(sameKeys, true);
assert.equal(fits, true);
for (const k of CHANNEL_SYNC_STATE_KEYS) assert.ok(CHANNEL_CONFIG_KEYS.includes(k), k);
diff --git a/common/lib/envVars.ts b/common/lib/envVars.ts
@@ -107,7 +107,7 @@ const DECLARED: EnvVarDecl[] = [
{ name: "OLLAMA_DIGEST_MODEL", audience: "runtime", default: "`qwen2.5:7b`", readBy: "common/lib/digestApps.ts", doc: "The ollama model the local digest lane asks for when settings name none." },
{ name: "CLAUDE_DIGEST_MODEL", audience: "runtime", default: "the CLI's default", readBy: "common/lib/digestApps.ts", doc: "The model the metered digest lane asks `claude` for when settings name none." },
{ name: "NITTER_INSTANCES", audience: "runtime", default: "a built-in list", readBy: "common/social/xNitterFetcher.ts", doc: "Comma-separated Nitter instances for the X fallback fetcher, in order of preference." },
- { name: "ARCHILYZER_X_BROWSER", audience: "runtime", default: "the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium", readBy: "common/social/xBrowser.ts", doc: "The Chromium-family browser /settings' \"Connect X account\" opens (a path, or a name looked up on PATH). It is launched without the automation signals, in the X session profile; a value that is not an executable refuses the connect rather than opening another browser." },
+ { name: "ARCHILYZER_X_BROWSER", audience: "runtime", default: "the first of `chromium`, `google-chrome`, `google-chrome-stable`, `chrome` on PATH, else Playwright's bundled Chromium", readBy: "common/social/xBrowser.ts", doc: "The Chromium-family browser /settings' \"Connect X account\" opens (a path, or a name looked up on PATH), and a forum-thread channel's \"Connect forum session\" too. It is launched without the automation signals, in the X session profile (or the forum host's profile); a value that is not an executable refuses the connect rather than opening another browser." },
{ name: "UMTOOL_URL", audience: "runtime", default: "unset (no link)", readBy: "editor/app/channels/[slug]/videos/[id]/page.tsx", doc: "umtool's front door; when set, the video page links to it." },
{ name: "TRANSCRIPT_SITE_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: one published archive to read over HTTP." },
{ name: "TRANSCRIPT_HUB_URL", audience: "runtime", default: "—", readBy: "mcp/src/sources.ts", doc: "MCP server: a hub, federating every archive it lists." },
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines.
- **A report video's cue lookup names a site that publishes only its reports.** Pointing a report-to-video manifest at such a site (`corpus.json` `site.scope: "cited"`) used to fail with "channel … is not in corpus.json"; it now says the site publishes no transcripts and to use a full archive or a local corpus. The site form's **Publish** hint says a cited-only site is never listed on the homepage or the hub.
- **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose now writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with `publish: "cited"` removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site set to cited whose last build was a full one. The hub's compose removes a report site's files too.
- **A site has a Reports tab.** `/sites/<site>/reports` lists every report under the site's `reports/` directory — the published ones in their order, then the drafts — with its kind, dates, sections, claims, citations by kind and, for a fact-check, how many claims carry each verdict. Each report's problems, from the same checker the prepare step and the build use, open under it. A draft with no problems can be published, and a published report moved up or down or unpublished; each writes only the site's `reports` list, applied to the list as it is on disk at that moment, so it never overwrites another change to the site. "Prepare evidence media" queues the `reports-prepare` job, and beside it the tab shows the last prepared media (moments by kind, total size, problems by kind) and links the last prepare job. What the site publishes (full or cited) is shown with a link to Settings, where it is changed.
diff --git a/editor/app/channels/[slug]/components/SocialChannelPanel.tsx b/editor/app/channels/[slug]/components/SocialChannelPanel.tsx
@@ -9,9 +9,12 @@
import { useState, useTransition } from "react";
import {
checkPostAvailabilityAction,
+ connectForumSessionAction,
fetchPostsAction,
+ importForumPagesAction,
setPostFetcherAction,
} from "../socialActions";
+import type { ForumSessionStatus } from "yt-dlp-transcript-common/social/forumSession";
export type SocialChannelState = {
postCount: number;
@@ -34,6 +37,10 @@ export type SocialChannelState = {
// has one, and where it stands, as one line.
canFetchOlder: boolean;
olderStatus: string;
+ // The channel's platform; "xenforo" is a forum thread.
+ platform?: string;
+ // A forum thread's browser session (its host's profile), when it is one.
+ forumSession?: ForumSessionStatus;
};
export function SocialChannelPanel({
@@ -67,14 +74,55 @@ export function SocialChannelPanel({
if (res.ok) void res.stream.cancel();
});
+ const forum = state.platform === "xenforo";
+ // A forum thread is read page by page, newest first: the cap on pages one
+ // run reads (blank = walk back until already-archived posts, or page 1).
+ const [pages, setPages] = useState("");
+ const [forumNote, setForumNote] = useState<string | null>(null);
+ const [forumErr, setForumErr] = useState<string | null>(null);
+ const [importPath, setImportPath] = useState("");
+
+ const connect = () =>
+ startTransition(async () => {
+ setForumNote(null);
+ setForumErr(null);
+ const res = await connectForumSessionAction(slug);
+ if (res.ok) {
+ setForumNote(
+ res.record.cleared
+ ? "Connected: the thread showed before the window closed. Fetch again."
+ : "The window closed before a thread page showed; the profile kept what it had.",
+ );
+ } else {
+ setForumErr(res.error);
+ }
+ });
+
+ const importPages = () =>
+ startTransition(async () => {
+ setForumNote(null);
+ setForumErr(null);
+ const res = await importForumPagesAction(slug, importPath);
+ if (res.ok) {
+ void res.stream.cancel();
+ setForumNote("Import queued — its log is on the jobs page.");
+ } else {
+ setForumErr(res.error);
+ }
+ });
+
const run = (full: boolean, older = false) =>
startTransition(async () => {
+ const n = Number(pages);
const res = await fetchPostsAction(
slug,
undefined,
full,
undefined,
older || undefined,
+ undefined,
+ undefined,
+ forum && pages.trim() && Number.isInteger(n) && n > 0 ? n : undefined,
);
// The job streams to the jobs page; nothing to consume here, but the
// stream must be released or its buffered chunks leak.
@@ -90,7 +138,7 @@ export function SocialChannelPanel({
<div className="flex flex-wrap items-baseline gap-x-3 gap-y-1">
<h2 className="text-sm font-medium">Posts</h2>
<span className="text-xs text-muted-foreground">
- @{state.handle} · {state.fetcherLabel}
+ {forum ? state.handle : `@${state.handle}`} · {state.fetcherLabel}
</span>
</div>
@@ -160,7 +208,23 @@ export function SocialChannelPanel({
</p>
)}
- <div className="mt-4 flex flex-wrap gap-2">
+ <div className="mt-4 flex flex-wrap items-center gap-2">
+ {forum && (
+ <label className="flex items-center gap-1 text-sm">
+ <span>Latest</span>
+ <input
+ type="number"
+ min={1}
+ inputMode="numeric"
+ value={pages}
+ onChange={(e) => setPages(e.target.value)}
+ placeholder="all"
+ aria-label="pages to fetch"
+ className="w-20 rounded border border-border bg-card px-2 py-1 text-sm"
+ />
+ <span>pages</span>
+ </label>
+ )}
<button
type="button"
onClick={() => run(false)}
@@ -214,8 +278,9 @@ export function SocialChannelPanel({
role="status"
className="mt-3 rounded border border-warning/40 bg-warning-soft px-3 py-2 text-xs"
>
- Needs credentials: the fetcher could not authenticate. Configure
- cookies-from-browser for this channel (or globally) and re-run.
+ {forum
+ ? "Needs the forum session: the forum showed a check (or a login) the headless browser could not get past. Use Connect forum session below, clear it in the window, close it, and fetch again."
+ : "Needs credentials: the fetcher could not authenticate. Configure cookies-from-browser for this channel (or globally) and re-run."}
</p>
)}
{state.lastError && (
@@ -242,6 +307,86 @@ export function SocialChannelPanel({
)}
</section>
+ {forum && (
+ <section
+ data-forum-session=""
+ className="flex flex-col gap-3 rounded-lg border border-border bg-card p-4"
+ >
+ <div className="flex flex-wrap items-baseline gap-x-3">
+ <h2 className="text-sm font-medium">Forum session</h2>
+ <span aria-label="forum session state" className="text-xs text-muted-foreground">
+ {state.forumSession?.lastConnect
+ ? `${state.forumSession.host}: connected ${new Date(state.forumSession.lastConnect.connectedAt).toLocaleString()}` +
+ (state.forumSession.lastConnect.cleared ? "" : " (the thread did not show)")
+ : state.forumSession?.hasProfile
+ ? `${state.forumSession.host}: profile in use, never connected by hand`
+ : `${state.forumSession?.host ?? "this forum"}: no profile yet`}
+ </span>
+ </div>
+ <p className="text-xs text-muted-foreground">
+ The fetcher reads the thread in a headless browser that keeps one
+ profile per forum host, so a browser check it clears once (Kiwi
+ Farms' KiwiFlare, say) stays cleared. When a check will not
+ clear by itself, or the thread needs a login, Connect opens that
+ profile in a window <strong>on the machine running the
+ editor</strong>, at the thread: clear it (or log in), wait for the
+ thread to show, then close the window.
+ </p>
+ <div className="flex flex-wrap gap-2">
+ <button
+ type="button"
+ onClick={connect}
+ disabled={pending}
+ aria-label="connect forum session"
+ className="rounded-md border border-border px-3 py-1 text-sm hover:bg-muted disabled:opacity-50"
+ >
+ {pending ? "Working…" : "Connect forum session"}
+ </button>
+ </div>
+
+ <div className="border-t border-border pt-3">
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Import saved pages</span>
+ <span className="text-xs text-muted-foreground">
+ Thread pages saved from your browser (“Save page as”,
+ complete or HTML only): a .html file, or a folder of them, on
+ the editor's machine. New posts are added, edited ones
+ updated.
+ </span>
+ <div className="flex flex-wrap gap-2">
+ <input
+ type="text"
+ value={importPath}
+ onChange={(e) => setImportPath(e.target.value)}
+ placeholder="/path/to/saved/pages"
+ aria-label="saved pages path"
+ className="min-w-0 flex-1 rounded border border-border bg-card px-2 py-1 font-mono text-sm"
+ />
+ <button
+ type="button"
+ onClick={importPages}
+ disabled={pending || !importPath.trim()}
+ aria-label="import saved pages"
+ className="rounded-md border border-border px-3 py-1 text-sm hover:bg-muted disabled:opacity-50"
+ >
+ Import
+ </button>
+ </div>
+ </label>
+ </div>
+ {forumNote && (
+ <p role="status" className="text-xs text-success">
+ {forumNote}
+ </p>
+ )}
+ {forumErr && (
+ <p role="alert" className="text-xs text-destructive">
+ {forumErr}
+ </p>
+ )}
+ </section>
+ )}
+
<p className="text-xs text-muted-foreground">
Posts are indexed into the search corpus by the normal build — there is
no per-post download or transcription stage.
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -16,6 +16,9 @@ import "yt-dlp-transcript-common/social/blueskyFetcher";
import "yt-dlp-transcript-common/social/xGalleryDlFetcher";
import "yt-dlp-transcript-common/social/xPlaywrightFetcher";
import "yt-dlp-transcript-common/social/xNitterFetcher";
+import "yt-dlp-transcript-common/social/xenforoFetcher";
+import { readForumSessionStatus } from "yt-dlp-transcript-common/social/forumSession";
+import { parseXenforoThreadUrl } from "yt-dlp-transcript-common/social/xenforoParse";
import { SocialChannelPanel } from "./components/SocialChannelPanel";
import { listPostFetchersFor } from "./socialActions";
import { NoReportYet } from "./components/NoReportYet";
@@ -166,6 +169,12 @@ export default async function ChannelDetailPage({
// does (it used to drop `progress`, `tasks`, `drainable` and the reorder
// bounds).
const socialRunningJobs = await liveJobRows((j) => j.channelSlug === slug);
+ // A forum thread's browser session (one profile per forum host).
+ const forumThread =
+ config.platform === "xenforo" ? parseXenforoThreadUrl(config.url ?? "") : null;
+ const forumSession = forumThread
+ ? await readForumSessionStatus(paths, forumThread.host)
+ : undefined;
return (
<div className="flex flex-col gap-6">
<RunningJobsList jobs={socialRunningJobs} hideChannelSlug />
@@ -191,6 +200,8 @@ export default async function ChannelDetailPage({
olderStatus: describeOlderBackfill(fetchState?.older),
handle: config.socialHandle ?? slug,
accountUrl: config.url,
+ platform: config.platform,
+ ...(forumSession ? { forumSession } : {}),
}}
/>
</div>
diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts
@@ -52,6 +52,14 @@ import {
runManagedFunction,
type StreamActionResult,
} from "yt-dlp-transcript-common/jobs/streamCommand";
+import { importForumPages } from "yt-dlp-transcript-common/controller/importForumPages";
+import {
+ connectForumSession,
+ readForumSessionStatus,
+ type ForumSessionRecord,
+ type ForumSessionStatus,
+} from "yt-dlp-transcript-common/social/forumSession";
+import { parseXenforoThreadUrl } from "yt-dlp-transcript-common/social/xenforoParse";
// Switch which SocialFetcher drives this channel. The fetchers differ in what
// they can actually reach — for X, gallery-dl needs credentials for depth,
@@ -100,6 +108,7 @@ async function registerBuiltinSocialFetchers(): Promise<void> {
await import("yt-dlp-transcript-common/social/xGalleryDlFetcher");
await import("yt-dlp-transcript-common/social/xPlaywrightFetcher");
await import("yt-dlp-transcript-common/social/xNitterFetcher");
+ await import("yt-dlp-transcript-common/social/xenforoFetcher");
}
// The posts analogue of the video availability check.
@@ -155,8 +164,13 @@ export async function fetchPostsAction(
older?: boolean,
floor?: string,
force?: boolean,
+ // A page walker's cap (a forum thread: its latest N pages this run).
+ pages?: number,
): Promise<StreamActionResult> {
if (full && older) return { ok: false, error: FULL_AND_OLDER_REFUSAL };
+ if (pages !== undefined && !(Number.isInteger(pages) && pages > 0)) {
+ return { ok: false, error: "Pages must be a whole number above 0." };
+ }
if (floor !== undefined && !older) {
return { ok: false, error: "A floor date applies only to an older-posts fetch." };
}
@@ -201,7 +215,7 @@ export async function fetchPostsAction(
spec: {
kind: "fetch-posts",
slug,
- params: { queueKey, full, limit, older, floor, force },
+ params: { queueKey, full, limit, older, floor, force, ...(pages ? { pages } : {}) },
},
fn: async (onLog, signal, _progress, ctx) => {
const result = await fetchPosts({
@@ -213,6 +227,7 @@ export async function fetchPostsAction(
floor,
force,
limit,
+ pages,
onLog,
signal,
// A drained fetch keeps its resume point and returns ok: the job
@@ -294,3 +309,80 @@ export async function capturePostsAction(
},
});
}
+
+// --- forum threads (platform "xenforo") ------------------------------------------
+
+async function forumChannel(
+ slug: string,
+): Promise<{ ok: true; url: string; host: string } | { ok: false; error: string }> {
+ const config = await readChannelConfig(getPaths(), slug);
+ if (!config) return { ok: false, error: `No such channel: ${slug}` };
+ if (!isSocialChannel(config) || config.platform !== "xenforo") {
+ return { ok: false, error: `${slug} is not a forum-thread channel.` };
+ }
+ const thread = parseXenforoThreadUrl(config.url ?? "");
+ if (!thread) return { ok: false, error: `${slug}'s URL is not a XenForo thread URL.` };
+ return { ok: true, url: thread.base, host: thread.host };
+}
+
+export async function forumSessionStatusAction(
+ slug: string,
+): Promise<{ ok: true; status: ForumSessionStatus } | { ok: false; error: string }> {
+ const ch = await forumChannel(slug);
+ if (!ch.ok) return ch;
+ return { ok: true, status: await readForumSessionStatus(getPaths(), ch.host) };
+}
+
+// CONNECT: a HEADED browser window on the forum host's profile, opened at the
+// thread on the machine running the editor, for the operator to clear a check,
+// answer a captcha or log in — then close. The headless fetcher and capture
+// reuse the profile. Only ever this explicit action opens a window; a fetch
+// that meets a check it cannot clear stops and says to come here.
+export async function connectForumSessionAction(
+ slug: string,
+): Promise<{ ok: true; record: ForumSessionRecord } | { ok: false; error: string }> {
+ const ch = await forumChannel(slug);
+ if (!ch.ok) return ch;
+ try {
+ const record = await connectForumSession(getPaths(), ch.url, {
+ onLog: (line) => console.log(`[forum-session] ${line}`),
+ });
+ revalidatePath(`/channels/${slug}`);
+ return { ok: true, record };
+ } catch (e) {
+ return { ok: false, error: (e as Error).message };
+ }
+}
+
+// IMPORT: thread pages saved from a browser ("Save page as", .html), read from
+// a path on the editor's machine — a file or a directory of them. New posts
+// are appended, edited ones updated. A job on the channel's platform queue, so
+// it never writes beside a fetch.
+export async function importForumPagesAction(
+ slug: string,
+ inputPath: string,
+ queueKey?: string,
+): Promise<StreamActionResult> {
+ const wanted = (inputPath ?? "").trim();
+ if (!wanted) return { ok: false, error: "Name a saved page or a directory of them." };
+ const ch = await forumChannel(slug);
+ if (!ch.ok) return ch;
+ const paths = getPaths();
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `No such channel: ${slug}` };
+ const key = resolveQueueKey(downloadQueueKey(config), queueKey);
+ return runManagedFunction({
+ kind: "import-forum-pages",
+ queueKey: key,
+ paths,
+ channelSlug: slug,
+ fn: async (onLog) => {
+ const result = await importForumPages({ paths, slug, inputs: [wanted], onLog });
+ for (const p of result.pages) {
+ if (p.skipped) onLog(`skipped ${p.file}: ${p.skipped}`);
+ }
+ safeRevalidate([`/channels/${slug}`]);
+ if (!result.ok) throw new Error(result.error ?? "Import failed");
+ },
+ });
+}
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -93,6 +93,7 @@ async function registerBuiltinSocialFetchers(): Promise<void> {
await import("yt-dlp-transcript-common/social/xGalleryDlFetcher");
await import("yt-dlp-transcript-common/social/xPlaywrightFetcher");
await import("yt-dlp-transcript-common/social/xNitterFetcher");
+ await import("yt-dlp-transcript-common/social/xenforoFetcher");
}
export type ProbeChannelResult =
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -73,6 +73,14 @@ const SLUG_NOISE_SEGMENTS = new Set([
"live",
]);
+// A long thread title cut at a word (hyphen) boundary, so its slug stays short.
+function shortSlugBase(title: string, max = 48): string {
+ if (title.length <= max) return title;
+ const cut = title.slice(0, max);
+ const at = cut.lastIndexOf("-");
+ return at > 12 ? cut.slice(0, at) : cut;
+}
+
// Offline slug candidate from a channel/playlist URL, no network. Prefers an
// @handle (@Veritasium -> veritasium); otherwise the last non-noise path segment
// (/c/Some Name -> some-name). Returns "" when nothing usable is found, so the
@@ -93,8 +101,15 @@ function slugCandidateFromUrl(url: string): string {
})
.filter(Boolean);
const handle = segs.find((s) => s.startsWith("@"));
+ // A forum thread (…/threads/<title>.<id>/page-N): the title.
+ const threadAt = segs.indexOf("threads");
+ const threadTitle =
+ threadAt >= 0 && /\.\d+$/.test(segs[threadAt + 1] ?? "")
+ ? shortSlugBase(segs[threadAt + 1].replace(/\.\d+$/, ""))
+ : undefined;
const base =
handle ??
+ threadTitle ??
[...segs].reverse().find((s) => !SLUG_NOISE_SEGMENTS.has(s.toLowerCase())) ??
segs[segs.length - 1] ??
"";
@@ -186,9 +201,7 @@ export function ChannelForm({
const editSourceKind = c?.sourceKind ?? "video";
const isSocial = isEdit
? editSourceKind === "social"
- : isSocialPlatform(platform as Platform) ||
- detectPlatform(url) === "twitter" ||
- detectPlatform(url) === "bluesky";
+ : isSocialPlatform(platform as Platform) || isSocialPlatform(detectPlatform(url));
const [probe, setProbe] = useState<{
state: "idle" | "loading" | "done" | "error";
message?: string;
@@ -217,6 +230,8 @@ export function ChannelForm({
}
if (detected === "twitter") setPostFetcher("x-gallery-dl");
else if (detected === "bluesky") setPostFetcher("bluesky-atproto");
+ // A forum thread URL (…/threads/<title>.<id>/): the thread walker.
+ else if (detected === "xenforo") setPostFetcher("xenforo-thread");
else setPostFetcher("");
}, [url, isEdit, platformTouched, handlingTouched, slugTouched, handleTouched]);
@@ -371,6 +386,7 @@ export function ChannelForm({
<option value="kick">Kick</option>
<option value="twitter">X / Twitter (posts)</option>
<option value="bluesky">Bluesky (posts)</option>
+ <option value="xenforo">Forum thread — XenForo (posts)</option>
</SeededSelect>
<span className="text-xs text-muted-foreground">
Used as the default job queue, so all channels on the same
@@ -490,6 +506,7 @@ export function ChannelForm({
<option value="kick">Kick</option>
<option value="twitter">X / Twitter (posts)</option>
<option value="bluesky">Bluesky (posts)</option>
+ <option value="xenforo">Forum thread — XenForo (posts)</option>
</ControlledSelect>
<span className="text-xs text-muted-foreground">
Used as the default job queue, so all channels on the same
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -269,6 +269,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
bool(p.older),
str(p.floor),
bool(p.force),
+ num(p.pages),
);
},
// The ids are the spec's own (a capture is OF specific posts, unlike a
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Forum posts read like the other posts.** A post from a forum-thread channel shows its place in the thread (#N), an "edited" mark, the thread's title and its media as links, and opening its thread shows its conversation — the posts it quotes and the posts quoting it — rather than the whole forum thread.
- **A report-only site with one report opens on that report.** Its home page is the report itself, its header links nothing, and `/reports/` forwards home: there is no index of one. With more reports the home page is the list, without repeating the site's title under the header; a list entry is the report's name, subtitle and dates (its counts and tally are on its page). Pages a report-only site does not have link home.
- **A report can belong to a series.** `report.json` `series` leads the report's title in the accent colour, in place of the kind's label ("Fact-check"); a page title, a cited-in link, `llms.txt` and the MCP name it `<series>: <title>`.
- **`pnpm start:export` serves a built site's moment pages.** It used `serve`, which listed a video or audio moment's directory (`3126.00-3151.00`) instead of serving its page; it now runs `export/scripts/serve-out.mjs`, which serves directories as Cloudflare Pages does, on `EXPORT_DEV_PORT` (3000).
diff --git a/export/app/lib/askConversation.ts b/export/app/lib/askConversation.ts
@@ -421,7 +421,7 @@ export function gatherSystemPrompt(
export function answerSystemPrompt(usedAliases: SearchAlias[]): string {
const base =
"You are answering a question about an archive of video transcripts AND " +
- "social posts (X/Twitter, Bluesky) from the same commentators — one corpus, " +
+ "social posts (X/Twitter, Bluesky, forum threads) from the same commentators — one corpus, " +
"two kinds of source. Base your " +
"answer on the excerpts provided in this conversation. Excerpts " +
"may appear in earlier turns, and a follow-up request (reformatting, " +
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -1175,6 +1175,41 @@ test("posts: findPost resolves by id and getThread returns the whole thread", as
assert.deepEqual(thread.map((p) => p.id), ["p1", "p2"]);
});
+test("posts: a forum post's thread is its conversation, not the whole forum thread", async () => {
+ // SYNTHETIC forum posts: 3 quotes 2, 4 quotes 3; 5 is unrelated.
+ const mk = (id: string, position: number, quotes: string[] = []): Post => ({
+ id,
+ slug: `forum/${id}`,
+ channelSlug: "forum",
+ author: `Member${position}`,
+ createdAt: new Date(Date.UTC(2026, 0, 1, 0, position)).toISOString(),
+ uploadDate: "20260101",
+ text: `Post ${position}.`,
+ url: `https://forum.example/posts/${id}/`,
+ platform: "xenforo",
+ isReply: position > 1,
+ isRepost: false,
+ links: [],
+ forum: {
+ host: "forum.example",
+ threadId: "1",
+ position,
+ ...(quotes.length ? { quotes: quotes.map((postId) => ({ postId })) } : {}),
+ },
+ });
+ const posts = [mk("5", 5), mk("4", 4, ["3"]), mk("3", 3, ["2"]), mk("2", 2), mk("1", 1)];
+ const source = {
+ async postsManifest() {
+ return { version: 1, channelSlug: "forum", pageCount: 1, maxPageBytes: 0, generatedAt: "", slugToPage: {} };
+ },
+ async postsPage() {
+ return posts;
+ },
+ } as unknown as Parameters<typeof getThread>[0];
+ const thread = await getThread(source, { slug: "forum" } as Parameters<typeof getThread>[1], posts[2]);
+ assert.deepEqual(thread.map((p) => p.id), ["2", "3", "4"]);
+});
+
test("server: get_post returns the post with no timestamps", async () => {
const client = await connectClient(new StubSource());
const res = await client.callTool({
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -8,7 +8,7 @@
// CONSTRUCTION: this file cannot quietly pick a different ceiling than the one
// the policy names.
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
-import type { Post } from "yt-dlp-transcript-common/lib/posts";
+import { postConversation, type Post } from "yt-dlp-transcript-common/lib/posts";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import {
@@ -868,6 +868,9 @@ export async function findPost(
// Every archived post in the same thread as `post`, oldest first. A thread can
// straddle shard pages, so this walks the channel's whole (byte-capped) tree.
+// A FORUM post's "thread" is its conversation — the posts it quotes and the
+// posts quoting it (postConversation) — not the whole forum thread, which is
+// the channel itself and may run to thousands of posts.
export async function getThread(
source: ShardSource,
ch: ChannelRef,
@@ -876,7 +879,9 @@ export async function getThread(
const threadId = post.threadId || post.id;
const manifest = await source.postsManifest(ch);
if (!manifest) return [post];
+ const forum = post.platform === "xenforo";
const thread: Post[] = [];
+ const all: Post[] = [];
for (let page = 0; page < manifest.pageCount; page++) {
let posts: Post[];
try {
@@ -885,9 +890,11 @@ export async function getThread(
continue;
}
for (const p of posts) {
- if ((p.threadId || p.id) === threadId) thread.push(p);
+ if (forum) all.push(p);
+ else if ((p.threadId || p.id) === threadId) thread.push(p);
}
}
+ if (forum) return postConversation(post, all);
rankThread(thread);
return thread.length > 0 ? thread : [post];
}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -400,7 +400,7 @@ export const TOOLS: Tool[] = [
name: "search_transcripts",
description:
"Search the archive for a term or phrase. The corpus holds video " +
- "transcripts AND social posts (X/Twitter, Bluesky) from the same " +
+ "transcripts AND social posts (X/Twitter, Bluesky, forum threads) from the same " +
"commentators; by default BOTH are searched and returned in one result " +
"set — use content_types to narrow. A post hit carries no timestamps " +
"(cite it as a bare source, never with @ mm:ss). Returns matching " +
@@ -597,7 +597,7 @@ export const TOOLS: Tool[] = [
type: "object",
properties: {
...SOURCE_ARG,
- post_id: { type: "string", description: "The post id (tweet id / atproto rkey)." },
+ post_id: { type: "string", description: "The post id (tweet id / atproto rkey / forum post id)." },
channel: {
type: "string",
description: "Optional owning channel slug/name to skip the lookup.",
@@ -611,8 +611,10 @@ export const TOOLS: Tool[] = [
name: "get_thread",
description:
"Fetch the whole thread a post belongs to (its root and every archived " +
- "reply), oldest first. This is the post-corpus analogue of reading the " +
- "transcript around a cited moment.",
+ "reply), oldest first. For a forum post it is the post's conversation " +
+ "instead: the posts it quotes and the posts quoting it, in thread order. " +
+ "This is the post-corpus analogue of reading the transcript around a " +
+ "cited moment.",
inputSchema: {
type: "object",
properties: {
@@ -2256,7 +2258,9 @@ function postToMarkdown(post: Post, heading = true, archive: string | null = nul
const lines: string[] = [];
if (heading) lines.push(`# Post by ${post.authorName || post.author}`);
lines.push(
- `- author: ${post.authorName ? `${post.authorName} (@${post.author})` : `@${post.author}`}`,
+ post.platform === "xenforo"
+ ? `- author: ${post.author}${post.forum?.authorId ? ` (member ${post.forum.authorId})` : ""}`
+ : `- author: ${post.authorName ? `${post.authorName} (@${post.author})` : `@${post.author}`}`,
);
lines.push(`- posted: ${post.createdAt}`);
lines.push(`- platform: ${post.platform}`);
@@ -2270,7 +2274,30 @@ function postToMarkdown(post: Post, heading = true, archive: string | null = nul
if (post.threadId && post.threadId !== post.id) {
lines.push(`- thread_id: ${post.threadId}`);
}
- if (post.mediaCount) {
+ // A forum post: which thread, where in it, edits, and whom it quotes. Quoted
+ // text in the body is marked "> " — it is the quoted author's, not this
+ // post's author's.
+ if (post.forum) {
+ const f = post.forum;
+ lines.push(
+ `- forum thread: ${f.threadTitle ? `${f.threadTitle} ` : ""}(${f.host}, thread ${f.threadId}` +
+ `${f.page ? `, page ${f.page}` : ""}${f.position ? `, post #${f.position}` : ""})`,
+ );
+ if (f.editedAt) lines.push(`- last edited: ${f.editedAt}`);
+ if (f.quotes?.length) {
+ lines.push(
+ `- quotes: ${f.quotes
+ .map((q) => `${q.author ?? "unnamed"}${q.postId ? ` (post_id ${q.postId})` : ""}`)
+ .join("; ")} — quoted text is marked "> " in the body`,
+ );
+ }
+ }
+ if (post.media?.length) {
+ lines.push(`- media: ${post.media.length} item(s), linked not archived:`);
+ for (const m of post.media.slice(0, 20)) {
+ lines.push(` - ${m.kind}${m.provider ? ` (${m.provider})` : ""}: ${m.url}`);
+ }
+ } else if (post.mediaCount) {
lines.push(`- media: ${post.mediaCount} attachment(s) (not archived)`);
}
if (post.engagement) {