commit ae86a6637b32d18b1585db4d28dc3da6c50ccb33
parent b60db16f328ad7639678e03415549ed5a677baf7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 02:22:36 -0400
archive.org: a file record is dated by its name, else its folder (one folder per upload)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
5 files changed, 36 insertions(+), 5 deletions(-)
diff --git a/common/lib/archiveOrg.test.ts b/common/lib/archiveOrg.test.ts
@@ -428,3 +428,17 @@ test("one file of many with no title of its own is titled and dated from its nam
assert.equal(withFileNameFields(infoNoDate).mirror?.uploadDate, "20210102");
assert.equal(withFileNameFields(infoNoDate).mirror?.title, "Original");
});
+
+test("archiveOrgMetadataPatch: a file that is no mirror is dated by its name or folder", () => {
+ const prov = {
+ version: 1,
+ identifier: "example-item",
+ file: "Show/20190303_The Example Show - A Guest/The Example Show - A Guest.mkv",
+ itemUrl: "https://archive.org/details/example-item",
+ fileUrl: "https://archive.org/details/example-item/Show/x.mkv",
+ item: { collections: [], publicDate: "2019-06-07" },
+ } as unknown as Parameters<typeof archiveOrgMetadataPatch>[0];
+ const patch = archiveOrgMetadataPatch(prov, { upload_date: "20190607", timestamp: 1559900000 });
+ assert.equal(patch.upload_date, "20190303");
+ assert.equal(patch.timestamp, null);
+});
diff --git a/common/lib/archiveOrg.ts b/common/lib/archiveOrg.ts
@@ -376,7 +376,9 @@ export function coerceArchiveOrgProvenance(value: unknown): ArchiveOrgProvenance
// and extension off; titleFromMirrorFileName), else the name —
// never the item's for one file of many.
// upload_date the original's date (a mirror: its info.json's, else the
-// date its file name starts with), with its timestamp dropped
+// date its file name starts with); for any other file of an
+// item, the date its name or folder starts with — with the
+// timestamp dropped either way
// description the original's (a mirror), when it had one
// uploader the original's uploader, else the item's public credit — never
// the uploading account's e-mail address
@@ -394,7 +396,9 @@ export function archiveOrgMetadataPatch(
mirror?.title ??
(prov.file ? (prov.fileTitle ?? titleFromMirrorFileName(prov.file) ?? fileName) : undefined);
if (title) want.title = title;
+ const fileDate = prov.file ? dateFromMirrorFileName(prov.file) : null;
if (mirror?.uploadDate) want.upload_date = mirror.uploadDate;
+ else if (fileDate) want.upload_date = fileDate;
if (mirror?.description) want.description = mirror.description;
const uploader = mirror?.uploader ?? prov.item.creator;
const current = info.uploader;
@@ -404,7 +408,7 @@ export function archiveOrgMetadataPatch(
for (const [k, v] of Object.entries(want)) {
if (JSON.stringify(info[k]) !== JSON.stringify(v)) patch[k] = v;
}
- // A mirror's date replaces archive.org's; the timestamp yt-dlp derived from
+ // A mirror's or a file's own date replaces archive.org's; the timestamp yt-dlp derived from
// the item's publicdate would contradict it.
if (patch.upload_date !== undefined && typeof info.timestamp === "number") patch.timestamp = null;
return patch;
diff --git a/common/lib/archiveOrgId.test.ts b/common/lib/archiveOrgId.test.ts
@@ -148,3 +148,10 @@ test("a mirrored file's name gives its upload day: the leading [word_]YYYYMMDD,
];
for (const [name, want] of cases) assert.equal(dateFromMirrorFileName(name), want, name);
});
+
+test("dateFromMirrorFileName: a file with no date of its own takes its folder's", () => {
+ assert.equal(dateFromMirrorFileName("Show/20190303_The Example Show - A Guest/The Example Show - A Guest.mkv"), "20190303");
+ assert.equal(dateFromMirrorFileName("Show/20190303_Folder/20180101 Own date.mkv"), "20180101", "the file's own date wins");
+ assert.equal(dateFromMirrorFileName("Show/Folder without date/Plain name.mkv"), null);
+ assert.equal(dateFromMirrorFileName("Show/20191340_Bad date/Plain.mkv"), null, "not a calendar day");
+});
diff --git a/common/lib/archiveOrgId.ts b/common/lib/archiveOrgId.ts
@@ -206,9 +206,15 @@ function mirrorFileStem(file: string): string {
// THE DATE A MIRRORED FILE'S NAME CARRIES: the leading `[<word>_]YYYYMMDD`
// archiving tools write (the original's upload day), as YYYYMMDD, only for a
-// real calendar day. Null when the name starts with no such date.
+// real calendar day. A file whose own name has none takes its folder's (an
+// archive that keeps one folder per upload: `<YYYYMMDD>_<title>/<title>.mkv`).
+// Null when neither starts with such a date.
export function dateFromMirrorFileName(file: string): string | null {
- return mirrorDatePrefix(mirrorFileStem(file))?.date ?? null;
+ const own = mirrorDatePrefix(mirrorFileStem(file))?.date;
+ if (own) return own;
+ const parts = file.split("/");
+ const folder = parts.length > 1 ? parts[parts.length - 2] : "";
+ return folder ? (mirrorDatePrefix(folder)?.date ?? null) : null;
}
export function titleFromMirrorFileName(file: string): string | null {
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -6,7 +6,7 @@
- **archive.org files come over BitTorrent when possible, else straight from archive.org — never through yt-dlp.** The chosen file of an archive.org import is fetched from the item's own torrent (`<identifier>_archive.torrent`, which lists archive.org as a web seed, so other peers take load off archive.org) with aria2c, only that file of the item, and seeded afterwards for 10 minutes or to a ratio of 1, whichever comes first; the log shows "torrent: <file> (n of m pieces, peers p, web seed yes)" and "seeding 10 min…". With no aria2c, a torrent that does not carry the file, or no progress for 5 minutes, it is downloaded directly from `archive.org/download/…` instead (resumable, backing off on 429/503), and the log says "fell back to direct download: <reason>". Every file is checked against archive.org's sha1/md5: a mismatch is downloaded once more directly, a second one fails the record. The record is written from the item's metadata: `metadata.info.json` with the file's page, the canonical id, the duration ffprobe measures and archive.org's playable copies of the file, the `archiveorg.json` provenance (a mirror's original title, date and uploader), and `audio.<fmt>` — an audio file already in the channel's format is used as is, anything else goes through the app's audio extraction, a video kept in the saved-video store when the channel keeps sources. An .avi/.mpeg/.flac/.wav original is fetched as archive.org's mp4 or mp3 of it. aria2c runs in its own process group: cancelling the job stops it and everything it started, and it stops itself if the editor exits. New settings block `archiveOrg` (`torrent`, `seedMinutes`, `seedRatio`, `stallMinutes`, `maxPeers`, `maxDownloadKiBps`, `maxUploadKiBps`), `ARIA2C_BIN`, an aria2c row in `archilyzer doctor`, and `aria2` in the runtime Docker images.
- **A forum thread can be archived as a posts source.** A XenForo thread URL (`…/threads/<title>.<id>/`; Kiwi Farms is recognised by host) makes a forum-thread channel — platform "xenforo", one channel per thread, each forum post a post — searchable and readable like X and Bluesky posts, in the editor, the export and the MCP (`get_thread` gives a forum post's conversation: the posts it quotes and the posts quoting it). **Fetch posts** reads the thread in a headless browser, newest page first, one page at a time with a 10–20 s pause (the channel key `postPagePauseSeconds` sets it), and stops at already-archived posts; a **Latest N pages** box (`archilyzer posts fetch --pages N`) caps a run, and the next run continues where it stopped. The browser keeps one profile per forum host, so a browser check it clears once (KiwiFlare's proof of work, say) stays cleared; a check that does not clear within a minute, a captcha, a login wall or a refusal stops the run with the reason and keeps its place — never retried at once. **Connect forum session** on the channel page opens that profile in a window on the editor's machine, at the thread, for the operator to clear it or log in. **Import saved pages** (`archilyzer posts import-html <slug> <file-or-dir>…`) reads thread pages saved from a browser ("Save page as", complete or HTML only) through the same parser: new posts are added and a post saved again after an edit is updated; a page saved from one of a forum's mirror domains (kiwifarms.net for a kiwifarms.st channel) is a page of the same thread. **Capture posts** works on forum posts: a screenshot of the post and its attached files, through the same profile. A forum post keeps its thread title, page, position, author id, last-edit time, quoted posts and its media links; quoted text is marked with "> " lines.
- **A site's reports can be exported as files a reader saves and hosts again.** `archilyzer reports export <site> [--report <id>] [--formats html,pdf,md,zip]`, the `reports-export` job (`POST /api/ops/reports-export`, `pnpm ops reports-export`, and **Export reports** on a site's Reports tab) write each published report, checked as the build checks it, into `.export-index/sites/<site>/report-exports/<report>/`: `report.html`, one self-contained page (its own style, no script, stills and post screenshots inlined and recompressed, clips linked on the site); `report.pdf`, that page printed by headless Chromium, skipped with a note where there is none; `report.md`, plain Markdown with numbered references; and `evidence-pack.zip`, the page with its clips, stills and screenshots as files plus the Markdown and the citations, packed by the system `zip` (a host without it fails that format, naming it). An `export.json` names each file's size and checksum and the checksum of the report.json it was made from; every export ends with the report's date and the start of that checksum. Preparing the evidence media exports at its end when nothing is missing, on the same queue. The build publishes an export beside the report only when it was made from the report as it is now and is at most 24 MiB — a larger evidence pack stays local — and the Reports tab lists each report's exports, their sizes and which the next build publishes. The 24 MiB limit is one number, shared with the source mirror and the evidence clips.
-- **archive.org items are a source (`platform: "archiveorg"`).** A channel can hold recordings imported from archive.org and transcribe them like any transcribe channel. **Import video** takes an item page (`https://archive.org/details/<identifier>`) when the item holds one media file, or ONE file of a multi-file item (`…/details/<identifier>/<file>`); an item with several media files is refused with the way to choose files. `pnpm ops import-archive-org --json '{"slug":…,"item":…,"files":[…]}'` (or `"match": "<regex>"`, `"dryRun": true`) imports chosen files of one item as one drainable job. A whole item's id is its identifier; a file's is `<identifier>__<slug>-<hash>`, stable and unique per file. Each record keeps an `archiveorg.json` sidecar — the item's title, date, creator and collections, its torrent, and for a mirror of a YouTube upload the original's id, URL, title and upload date read from the info.json uploaded beside it — and its metadata takes the file's own page and title (and a mirror's original title and date), recorded in the metadata history as `archiveorg-provenance`. The video page says "Archived on archive.org: <item> · torrent" and, for a mirror, "Originally on YouTube: <url> (uploaded <date>)". The channel form offers archive.org in both platform lists. One file of a multi-file item with no uploaded info.json is dated by the `YYYYMMDD` its file name starts with (after any `<word>_`, a real calendar day only) and, when the item gives it no title of its own, titled from that name with the date, the `[<n> views]` count, the YouTube id and the extension taken off and ` _ ` read as ` | ` — the raw name stays in `archiveorg.json` as `file` — and `archilyzer archive-org refresh <slug> [--dry-run]` brings a channel's existing file records to the same title and date, offline, as `archiveorg-provenance` entries in the metadata history, printing old → new.
+- **archive.org items are a source (`platform: "archiveorg"`).** A channel can hold recordings imported from archive.org and transcribe them like any transcribe channel. **Import video** takes an item page (`https://archive.org/details/<identifier>`) when the item holds one media file, or ONE file of a multi-file item (`…/details/<identifier>/<file>`); an item with several media files is refused with the way to choose files. `pnpm ops import-archive-org --json '{"slug":…,"item":…,"files":[…]}'` (or `"match": "<regex>"`, `"dryRun": true`) imports chosen files of one item as one drainable job. A whole item's id is its identifier; a file's is `<identifier>__<slug>-<hash>`, stable and unique per file. Each record keeps an `archiveorg.json` sidecar — the item's title, date, creator and collections, its torrent, and for a mirror of a YouTube upload the original's id, URL, title and upload date read from the info.json uploaded beside it — and its metadata takes the file's own page and title (and a mirror's original title and date), recorded in the metadata history as `archiveorg-provenance`. The video page says "Archived on archive.org: <item> · torrent" and, for a mirror, "Originally on YouTube: <url> (uploaded <date>)". The channel form offers archive.org in both platform lists. One file of a multi-file item with no uploaded info.json is dated by the `YYYYMMDD` its file name starts with (after any `<word>_`, a real calendar day only) and, when the item gives it no title of its own, titled from that name with the date, the `[<n> views]` count, the YouTube id and the extension taken off and ` _ ` read as ` | ` — the raw name stays in `archiveorg.json` as `file` — and `archilyzer archive-org refresh <slug> [--dry-run]` brings a channel's existing file records to the same title and date, offline, as `archiveorg-provenance` entries in the metadata history, printing old → new. A file with no date in its own name takes the one its folder starts with (`<YYYYMMDD>_<title>/…`), and any file of an item, not only a mirror, is dated that way.
- **Polite to archive.org.** archive.org runs on its own queue (`platform:archiveorg`), one download at a time, with a jittered pause of at least 8 s between files (the channel's or the global `sleepBetweenDownloadsSeconds` when longer). Its metadata API is asked once per item (cached for 6 h), with an identifying User-Agent, at most one request at a time and 2 s apart, honouring `Retry-After` and backing off exponentially on 429/503, stopping after four attempts. A file already downloaded is never fetched again, and a bulk import stops on a rate limit or after three failures in a row — re-running it resumes.
- **A report video's cue lookup names a site that publishes only its reports.** Pointed at such a site (`corpus.json` `site.scope: "cited"`), a report-to-video manifest's cue lookup says the site publishes no transcripts and to use a full archive or a local corpus.
- **A site's build composes its reports, and a site that publishes only its reports ships nothing else.** Every site's compose writes the reports its `site.json` publishes: each report's page and its citations as `citations.json` and `citations.csv` under `/reports/<id>/`, its cited stills, a page per cited moment with the record, the transcript lines around the span and every report that cites it, and the clips and post captures `archilyzer reports prepare` made for it, only the cited ones. Each quote is checked against the record as it is composed (a span's against its cues within 5 s either side, read from `en-orig` when the `en` track has no cues; a post's against its text) and the score, time and method are written into the citation, replacing any typed by hand. The build stops with the list of every problem before anything is written: an invalid report, a citation of a channel outside the site or of a post the site may not carry, a missing record, still or post, a quote that matches less than 60 % of what the record says, and a citation without prepared media or with media cut for another span (`--allow-missing-media` on `archilyzer compose site` and `build site` lets those two through, without a clip). A site with search off (`search: false`) removes everything corpus-shaped from `export/public` before it writes its reports, and its built `out/` is checked against what a cited site may hold: anything else, a file over 25 MiB or more than 20,000 files fails the build, and every deploy path (the Publish tab, `deploy site`, Build & deploy, Build & deploy all, the container build) refuses it, as it refuses a site with search off whose last build was a full one. The hub's compose removes a report site's files too.