commit 8716c71e27bd55393ce4dd72e3d558bd98277eaa
parent 5e314ea46a132ac9413641ae755d8c4f169251fd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 16:02:13 -0400
sources: archive.org import comments name the torrent path, not yt-dlp
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 13 insertions(+), 8 deletions(-)
diff --git a/common/controller/archiveOrgImport.ts b/common/controller/archiveOrgImport.ts
@@ -2,19 +2,22 @@
// of files of one item, into an existing channel.
//
// Every download is `downloadOneManaged` (the same managed path every other
-// import takes), fetched by the CANONICAL page of what is imported
+// import takes), by the CANONICAL page of what is imported
// (lib/archiveOrgId.ts): `https://archive.org/details/<identifier>` for an item
// holding one media file, `…/details/<identifier>/<file>` for one file of
-// many. yt-dlp's ArchiveOrg extractor resolves either to exactly one record,
-// and the provenance step (lib/archiveOrg-server.ts) runs inside it.
+// many. That routes it to controller/archiveOrgDownload.ts — no yt-dlp: the
+// file comes over BitTorrent from the item's torrent when it can (archive.org
+// is the torrent's web seed, and the file is seeded for a while after), else
+// straight from archive.org, verified against the item's checksums, and the
+// record is built from the item's metadata with its provenance sidecar.
//
-// AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE. yt-dlp would treat
-// it as a playlist and write every file into the one pinned `data/<id>/`; the
-// import refuses it and names the way to choose files instead.
+// AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE: one record is one
+// file. The import refuses it and names the way to choose files instead.
//
// POLITE (the operator: "be polite to archive.org"):
// - the item's metadata is asked for once (lib/archiveOrgClient.ts caches it)
// and before any download, so a typo'd identifier costs one request;
+// - over BitTorrent when possible (settings.archiveOrg), seeding after;
// - files go one at a time, on the `platform:archiveorg` queue, with a
// jittered gap between them — the channel's (else the global)
// `sleepBetweenDownloadsSeconds`, never under ARCHIVE_ORG_MIN_GAP_SECONDS,
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -515,8 +515,10 @@ export async function importVideoAction(
// metadata): an item holding several media files is refused — import a file
// of it, or several with import-archive-org — and the URL becomes the
// canonical page of what is imported (controller/archiveOrgImport.ts). It
- // runs on archive.org's own queue whatever the channel's platform, with
- // archive.org's yt-dlp args, and a file already on disk is not fetched again.
+ // runs on archive.org's own queue whatever the channel's platform — over
+ // BitTorrent when it can, else straight from archive.org, never yt-dlp
+ // (controller/archiveOrgDownload.ts) — and a file already on disk is not
+ // fetched again.
const archiveOrg = detectPlatform(videoUrl) === "archiveorg";
let downloadConfig = channelConfig;
if (archiveOrg) {