Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b63208b86560533079814136b1c5195159d13d43
parent 38bc728ec0cb330a8d70705264977b06f0e7af85
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 17:46:18 -0400

Merge posts/forum-mirror-hosts

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

# Conflicts:
#	editor/CHANGELOG.md

Diffstat:
Mcommon/controller/importForumPages.ts | 3++-
Mcommon/lib/detectPlatform.mjs | 17+++++++++++++++++
Acommon/lib/platform.test.ts | 20++++++++++++++++++++
Mcommon/lib/platform.ts | 4++--
Meditor/CHANGELOG.md | 2+-
5 files changed, 42 insertions(+), 4 deletions(-)

diff --git a/common/controller/importForumPages.ts b/common/controller/importForumPages.ts @@ -8,6 +8,7 @@ // Server-only (node:fs). Nothing here touches the network: a saved page's // assets are read from the page as it is, never fetched. +import { sameXenforoForum } from "../lib/platform"; import { readdir, readFile, stat } from "node:fs/promises"; import path from "node:path"; import { readChannelConfig } from "./channels"; @@ -123,7 +124,7 @@ export async function importForumPages(opts: ImportForumPagesOptions): Promise<I }); continue; } - if (parsed.host && parsed.host !== thread.host) { + if (parsed.host && !sameXenforoForum(parsed.host, thread.host)) { pages.push({ file: name, page: parsed.page, posts: 0, skipped: `a page of another forum (${parsed.host})` }); continue; } diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs @@ -13,6 +13,23 @@ // not the rule. export const XENFORO_HOSTS = ["kiwifarms.st", "kiwifarms.net"]; +// Hosts that serve ONE forum under several domains (a page saved from one is a +// page of the same thread on the other). Compared by `sameXenforoForum`. +export const XENFORO_MIRRORS = [["kiwifarms.st", "kiwifarms.net"]]; + +/** + * Whether two hosts are the same forum: equal, or mirrors of one another. + * @param {string} a + * @param {string} b + * @returns {boolean} + */ +export function sameXenforoForum(a, b) { + const x = a.toLowerCase().replace(/^www\./, ""); + const y = b.toLowerCase().replace(/^www\./, ""); + if (x === y) return true; + return XENFORO_MIRRORS.some((group) => group.includes(x) && group.includes(y)); +} + // A XenForo thread path: /threads/<slug>.<id>/ or /threads/<id>/, optionally // under a prefix (/community/threads/…) or behind index.php? (non-friendly // URLs). diff --git a/common/lib/platform.test.ts b/common/lib/platform.test.ts @@ -0,0 +1,20 @@ +// Host helpers re-exported by lib/platform (the table lives in detectPlatform.mjs). +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/platform.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { detectPlatform, sameXenforoForum } from "./platform"; + +test("sameXenforoForum: equal hosts and a forum's mirror domains are one forum", () => { + assert.equal(sameXenforoForum("forum.example", "forum.example"), true); + assert.equal(sameXenforoForum("www.forum.example", "forum.example"), true); + assert.equal(sameXenforoForum("kiwifarms.net", "kiwifarms.st"), true); + assert.equal(sameXenforoForum("KiwiFarms.ST", "kiwifarms.net"), true); + assert.equal(sameXenforoForum("kiwifarms.st", "forum.example"), false); +}); + +test("both mirror domains are detected as xenforo", () => { + assert.equal(detectPlatform("https://kiwifarms.net/threads/x.1/"), "xenforo"); + assert.equal(detectPlatform("https://kiwifarms.st/threads/x.1/"), "xenforo"); +}); diff --git a/common/lib/platform.ts b/common/lib/platform.ts @@ -42,8 +42,8 @@ export function isSocialPlatform(platform: Platform | null | undefined): boolean // Host → platform. Lives in `detectPlatform.mjs` (plain JS so umtool's `.mjs` // scripts can import the same copy); every TS caller imports it from here. -import { detectPlatform } from "./detectPlatform.mjs"; -export { detectPlatform }; +import { detectPlatform, sameXenforoForum } from "./detectPlatform.mjs"; +export { detectPlatform, sameXenforoForum }; // Best-effort canonical webpage URL for a video given its platform and // canonical id. Used both as a metadata fallback in summarize() and to diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -3,7 +3,7 @@ ## [Unreleased] - **Exporting a changed report records a new revision of it.** `reports export` (and **Export reports** on a site's Reports tab, and the end of a prepare) commits a revision to the report's own git history, `sites/<site>/reports/<id>/history-git/`, whenever its `report.json` changed since the last one: the `report.json`, its Markdown export and the checksums of every export file, with a message of `Revision N` and a summary of the change. A re-export of an unchanged report records nothing. The commits carry the site's name and a `noreply@<site>.invalid` address with dates in UTC, never your git name, email or time zone. The Reports tab shows each report's revision, its commit and the last change under **Exports**, and the site's next build publishes the history. Add `history-git/` to the corpus repository's `.gitignore`. - **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images. -- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines. +- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated; a page saved from one of a forum's mirror domains (kiwifarms.net for a kiwifarms.st channel) is a page of the same thread. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines. - **A site's reports can be exported as files a reader saves and hosts again.** `archilyzer reports export <site> [--report <id>] [--formats html,pdf,md,zip]`, the `reports-export` job (`POST /api/ops/reports-export`, `pnpm ops reports-export`, and **Export reports** on a site's Reports tab) write each published report, checked as the build checks it, into `.export-index/sites/<site>/report-exports/<report>/`: `report.html`, one self-contained page (its own style, no script, stills and post screenshots inlined and recompressed, clips linked on the site); `report.pdf`, that page printed by headless Chromium, skipped with a note where there is none; `report.md`, plain Markdown with numbered references; and `evidence-pack.zip`, the page with its clips, stills and screenshots as files plus the Markdown and the citations, packed by the system `zip` (a host without it fails that format, naming it). An `export.json` names each file's size and checksum and the checksum of the report.json it was made from; every export ends with the report's date and the start of that checksum. Preparing the evidence media exports at its end when nothing is missing, on the same queue. The build publishes an export beside the report only when it was made from the report as it is now and is at most 24 MiB — a larger evidence pack stays local — and the Reports tab lists each report's exports, their sizes and which the next build publishes. The 24 MiB limit is one number, shared with the source mirror and the evidence clips. - **archive.org items are a source (`platform: "archiveorg"`).** A channel can hold recordings imported from archive.org and transcribe them like any transcribe channel. **Import video** takes an item page (`https://archive.org/details/<identifier>`) when the item holds one media file, or ONE file of a multi-file item (`…/details/<identifier>/<file>`); an item with several media files is refused with the way to choose files. `pnpm ops import-archive-org --json '{"slug":…,"item":…,"files":[…]}'` (or `"match": "<regex>"`, `"dryRun": true`) imports chosen files of one item as one drainable job. A whole item's id is its identifier; a file's is `<identifier>__<slug>-<hash>`, stable and unique per file. Each record keeps an `archiveorg.json` sidecar — the item's title, date, creator and collections, its torrent, and for a mirror of a YouTube upload the original's id, URL, title and upload date read from the info.json uploaded beside it — and its metadata takes the file's own page and title (and a mirror's original title and date), recorded in the metadata history as `archiveorg-provenance`. The video page says "Archived on archive.org: <item> · torrent" and, for a mirror, "Originally on YouTube: <url> (uploaded <date>)". The channel form offers archive.org in both platform lists. - **Polite to archive.org.** archive.org runs on its own queue (`platform:archiveorg`), one download at a time, with a jittered pause of at least 8 s between files (the channel's or the global `sleepBetweenDownloadsSeconds` when longer). Its metadata API is asked once per item (cached for 6 h), with an identifying User-Agent, at most one request at a time and 2 s apart, honouring `Retry-After` and backing off exponentially on 429/503, stopping after four attempts. A file already downloaded is never fetched again, and a bulk import stops on a rate limit or after three failures in a row — re-running it resumes.