// IMPORTING FROM archive.org — one item, one file of an item, or a chosen set // of files of one item, into an existing channel. // // Every download is `downloadOneManaged` (the same managed path every other // import takes), by the CANONICAL page of what is imported // (lib/archiveOrgId.ts): `https://archive.org/details/` for an item // holding one media file, `…/details//` for one file of // many. That routes it to controller/archiveOrgDownload.ts — no yt-dlp: the // file comes over BitTorrent from the item's torrent when it can (archive.org // is the torrent's web seed, and the file is seeded for a while after), else // straight from archive.org, verified against the item's checksums, and the // record is built from the item's metadata with its provenance sidecar. // // AN ITEM WITH SEVERAL MEDIA FILES IS NEVER IMPORTED WHOLE: one record is one // file. The import refuses it and names the way to choose files instead. // // POLITE (the operator: "be polite to archive.org"): // - the item's metadata is asked for once (lib/archiveOrgClient.ts caches it) // and before any download, so a typo'd identifier costs one request; // - over BitTorrent when possible (settings.archiveOrg), seeding after; // - files go one at a time, on the `platform:archiveorg` queue, with a // jittered gap between them — the channel's (else the global) // `sleepBetweenDownloadsSeconds`, never under ARCHIVE_ORG_MIN_GAP_SECONDS, // plus up to half again at random; // - a file already downloaded is never fetched again; // - a rate-limited download stops the batch at once (the platform's own // backoff takes over), and ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES failures in // a row stop it too. Re-running the same command resumes: what landed is // skipped. // // MANY ITEMS (release 19 A6): `{items: [...]}` or `{query}` (archive.org's // advanced search, one request) is one job over many items, the same polite // rules shared across the batch plus a jittered pause between items. A file // archive.org marks as not for download (`access-restricted-item` on the item, // `private` on the file) is RESTRICTED: a dry run lists it, an import skips it; // a download answered 401/403 marks the rest of its item restricted, and three // such items in a row stop the batch (the platform backs off). A record already // HELD — on disk, or its media in the saved-video store — is never fetched. import path from "node:path"; import type { ChannelConfig } from "../lib/channelConfig"; import type { Paths } from "../lib/paths"; import type { DownloadOutcomeRecord } from "../lib/downloadOutcome"; import { listArchiveOrgMediaFiles, pickArchiveOrgFiles, type ArchiveOrgFileSelection, type ArchiveOrgItemMetadata, } from "../lib/archiveOrg"; import { archiveOrgDetailsUrl, archiveOrgVideoId, parseArchiveOrgUrl, } from "../lib/archiveOrgId"; import { ArchiveOrgRequestError, archiveOrgClient, type ArchiveOrgClient, } from "../lib/archiveOrgClient"; import { isSavedVideo } from "../lib/savedVideo-server"; import { getSettings } from "../lib/settings"; import { diskGate } from "../lib/diskSpace"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; import { downloadOneManaged, type ManagedDownloadOpts } from "../ytdlp/downloadOneManaged"; import { destinationExists } from "../ytdlp/runYtdlp"; import { mergeRosterFile } from "./rosterStore"; export const ARCHIVE_ORG_MIN_GAP_SECONDS = 8; export const ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES = 3; // ─── One URL ─── export type ResolvedArchiveOrgImport = | { ok: true; url: string; id: string; identifier: string; file?: string } | { ok: false; error: string }; // What an archive.org URL imports as. An item URL for an item with ONE media // file is the item (id = identifier); a file URL is that file — unless the // item holds only that one file, when it is the item too, so the same // recording never gets two ids. An item URL for an item with several media // files is refused, naming them. export async function resolveArchiveOrgImportUrl( url: string, opts: { client?: ArchiveOrgClient; signal?: AbortSignal } = {}, ): Promise { const ref = parseArchiveOrgUrl(url); if (!ref) { return { ok: false, error: `Not an archive.org item URL: ${url} (expected https://archive.org/details/[/])`, }; } const client = opts.client ?? archiveOrgClient; let item: ArchiveOrgItemMetadata; try { item = await client.itemMetadata(ref.identifier, opts.signal); } catch (err) { return { ok: false, error: (err as Error).message }; } const identifier = item.metadata.identifier || ref.identifier; const media = listArchiveOrgMediaFiles(item).map((f) => f.name); if (ref.file) { if (!item.files.some((f) => f.name === ref.file)) { return { ok: false, error: `archive.org item "${identifier}" has no file "${ref.file}"` }; } if (media.length === 1 && media[0] === ref.file) { return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier }; } const file = ref.file; return { ok: true, url: archiveOrgDetailsUrl({ identifier, file }), id: archiveOrgVideoId({ identifier, file }), identifier, file, }; } if (media.length === 0) { return { ok: false, error: `archive.org item "${identifier}" has no media files to import` }; } if (media.length > 1) { const shown = media.slice(0, 5).map((n) => `"${n}"`).join(", "); return { ok: false, error: `archive.org item "${identifier}" holds ${media.length} media files (${shown}${media.length > 5 ? ", …" : ""}). ` + `Import one by its file URL (https://archive.org/details/${identifier}/), or several with ` + `pnpm ops import-archive-org --json '{"slug":"","item":"${identifier}","files":[…]}' (or "match": "").`, }; } return { ok: true, url: archiveOrgDetailsUrl({ identifier }), id: identifier, identifier }; } // ─── Many files of one item ─── export type ArchiveOrgImportPlanEntry = { file: string; url: string; id: string; // HELD: already downloaded, or its media kept in the saved-video store (the // saved-container tier) — skipped, never fetched again. onDisk: boolean; // archive.org marks the file (or the whole item) as not for download: a // fetch answers 401/403. Listed by a dry run, skipped by an import. restricted: boolean; }; export type ArchiveOrgImportPlan = { identifier: string; title?: string; mediaFiles: number; entries: ArchiveOrgImportPlanEntry[]; // Named in `files` but not a media original of the item. unknown: string[]; // The whole item is access-restricted (archive.org's lending flag). restrictedItem: boolean; }; const truthy = (v: unknown): boolean => v === true || v === "true" || v === "1"; // What archive.org will not hand out, read off the item's own metadata: an // item carrying `access-restricted-item` (every file), else each file marked // `private`. Either answers a download with 401/403; the metadata says so // before anything is asked for. export function archiveOrgRestriction(item: ArchiveOrgItemMetadata): { item: boolean; files: Set; } { const meta = item.metadata as Record; const whole = truthy(meta["access-restricted-item"]); const files = new Set(); for (const f of item.files) { if (whole || truthy((f as Record).private)) files.add(f.name); } return { item: whole, files }; } // HELD: the record has its transcript or (on a transcribe channel) its audio — // `destinationExists`, what every download path asks — or its media lives in // the saved-video store, which a download must never bypass. export async function isHeldArchiveOrgRecord( dataDir: string, id: string, handling: ChannelConfig["handling"], ): Promise { if (await destinationExists(dataDir, id, handling)) return true; return isSavedVideo(path.join(dataDir, id)); } export async function planArchiveOrgImport(opts: { identifier: string; selection: ArchiveOrgFileSelection; dataDir: string; handling: ChannelConfig["handling"]; client?: ArchiveOrgClient; signal?: AbortSignal; }): Promise { const client = opts.client ?? archiveOrgClient; const item = await client.itemMetadata(opts.identifier, opts.signal); const identifier = item.metadata.identifier || opts.identifier; const media = listArchiveOrgMediaFiles(item); const restriction = archiveOrgRestriction(item); const { picked, unknown } = pickArchiveOrgFiles(item, opts.selection); const entries: ArchiveOrgImportPlanEntry[] = []; for (const file of picked) { // An item of one media file imports as the item (resolveArchiveOrgImportUrl). const whole = media.length === 1; const id = whole ? identifier : archiveOrgVideoId({ identifier, file }); const url = whole ? archiveOrgDetailsUrl({ identifier }) : archiveOrgDetailsUrl({ identifier, file }); entries.push({ file, url, id, onDisk: await isHeldArchiveOrgRecord(opts.dataDir, id, opts.handling), restricted: restriction.files.has(file), }); } const title = item.metadata.title; return { identifier, ...(typeof title === "string" ? { title } : {}), mediaFiles: media.length, entries, unknown, restrictedItem: restriction.item, }; } // ─── A search ─── export const ARCHIVE_ORG_SEARCH_DEFAULT_ROWS = 100; export const ARCHIVE_ORG_SEARCH_MAX_ROWS = 500; // archive.org's advanced search, ONE request through the polite client (its // chain, its gap, its backoff): the identifiers of the first `rows` items the // query matches, in identifier order so a re-run walks the same list. export function archiveOrgSearchUrl(query: string, rows: number): string { const q = new URLSearchParams(); q.set("q", query); q.append("fl[]", "identifier"); q.append("fl[]", "title"); q.append("sort[]", "identifier asc"); q.set("rows", String(rows)); q.set("page", "1"); q.set("output", "json"); return `https://archive.org/advancedsearch.php?${q.toString()}`; } export async function searchArchiveOrgItems( query: string, opts: { rows?: number; client?: ArchiveOrgClient; signal?: AbortSignal } = {}, ): Promise<{ found: number; items: { identifier: string; title?: string }[] }> { const rows = Math.min(ARCHIVE_ORG_SEARCH_MAX_ROWS, Math.max(1, opts.rows ?? ARCHIVE_ORG_SEARCH_DEFAULT_ROWS)); const client = opts.client ?? archiveOrgClient; const raw = (await client.getJson(archiveOrgSearchUrl(query, rows), opts.signal)) as { response?: { numFound?: unknown; docs?: unknown }; } | null; const docs = Array.isArray(raw?.response?.docs) ? (raw!.response!.docs as unknown[]) : []; const items: { identifier: string; title?: string }[] = []; for (const d of docs) { const r = d as { identifier?: unknown; title?: unknown }; if (typeof r?.identifier !== "string" || !r.identifier) continue; const title = Array.isArray(r.title) ? r.title.find((t) => typeof t === "string") : r.title; items.push({ identifier: r.identifier, ...(typeof title === "string" ? { title } : {}) }); } const found = typeof raw?.response?.numFound === "number" ? raw.response.numFound : items.length; return { found, items }; } // The gap before the next file: the configured pause, floored, plus up to // half again at random so a batch never settles into a fixed beat. export function archiveOrgGapMs(sleepBetweenDownloadsSeconds: number, random: number): number { const base = Math.max(ARCHIVE_ORG_MIN_GAP_SECONDS, sleepBetweenDownloadsSeconds || 0); return Math.round(base * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000); } // THE INTER-ITEM GAP: before the next item of a batch is asked about, a // jittered pause on top of the client's own 2 s between requests, so a dry run // over a search of hundreds of items reads as a person paging, not a crawl. // (A download after the first is paced by archiveOrgGapMs as well.) export const ARCHIVE_ORG_ITEM_GAP_SECONDS = 3; export function archiveOrgItemGapMs(random: number): number { return Math.round(ARCHIVE_ORG_ITEM_GAP_SECONDS * (1 + 0.5 * Math.min(1, Math.max(0, random))) * 1000); } // Items in a row whose downloads archive.org refused with 401/403 before the // batch stops: one restricted item is that item; three in a row is archive.org // refusing us, and the platform backs off. export const ARCHIVE_ORG_MAX_REFUSED_ITEMS = 3; function isOk(rec: DownloadOutcomeRecord): boolean { return rec.status.startsWith("ok"); } // A download error that is archive.org saying "not for you" (lib/archiveOrgClient // words every non-retryable answer "archive.org answered HTTP for "). export function isRefusalError(error: string): boolean { return /\bHTTP 40[13]\b/.test(error); } export type ArchiveOrgImportResult = { identifier: string; planned: number; imported: string[]; // Held: on disk or in the saved-video store. skipped: string[]; // Not for download (archive.org's metadata says so, or a fetch answered // 401/403): never fetched. restricted: string[]; failed: { file: string; error: string }[]; unknown: string[]; // Why the batch ended before its last file, when it did. stopped?: string; // archive.org refused a download as a rate limit: the caller backs the // platform off (jobs/downloadBackoff.ts). rateLimited?: boolean; }; export type ArchiveOrgImportDeps = { client?: ArchiveOrgClient; downloadOne?: (opts: ManagedDownloadOpts) => Promise; sleep?: (ms: number, signal: AbortSignal) => Promise; random?: () => number; // After each imported file (the editor revalidates its pages). onImported?: (id: string) => void; }; function abortableSleep(ms: number, signal: AbortSignal): Promise { return new Promise((resolve) => { if (signal.aborted) return resolve(); const t = setTimeout(done, ms); function done() { clearTimeout(t); signal.removeEventListener("abort", done); resolve(); } signal.addEventListener("abort", done, { once: true }); }); } export type ArchiveOrgItemRequest = { identifier: string; selection: ArchiveOrgFileSelection; }; export type ArchiveOrgBatchResult = { // One per item reached, in order. items: ArchiveOrgImportResult[]; // Items archive.org has no record of (or whose metadata it would not give). missing: { identifier: string; error: string }[]; // Items never reached: the batch stopped first. notReached: string[]; stopped?: string; rateLimited?: boolean; // Three items in a row refused with 401/403: the caller backs the platform off. refusedStorm?: boolean; dryRun: boolean; }; type BatchOpts = { paths: Paths; slug: string; channelConfig: ChannelConfig; onLog: (line: string) => void; signal: AbortSignal; drainSignal?: AbortSignal; dryRun?: boolean; deps?: ArchiveOrgImportDeps; }; // MANY ITEMS, ONE JOB (`import-archive-org {items | query}`). Each item is // planned (one cached metadata request) and imported as a single-item import // is, with the state that makes it polite shared across the batch: the gap // before every download after the first, the consecutive-failure count, the // rate-limit stop. Between items, the inter-item gap. An item archive.org has // no record of is listed and passed over; a rate limit stops everything (the // rest are `notReached`, and re-running the same body resumes — held records // are skipped); three items in a row refused with 401/403 stop it too. export async function runArchiveOrgBatchImport( opts: BatchOpts & { items: ArchiveOrgItemRequest[] }, ): Promise { const deps = opts.deps ?? {}; const downloadOne = deps.downloadOne ?? downloadOneManaged; const sleep = deps.sleep ?? abortableSleep; const random = deps.random ?? Math.random; const settings = getSettings(); const dataDir = path.join(opts.paths.channelsDir, opts.slug, "data"); const log = opts.onLog; const sleepSeconds = opts.channelConfig.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds; const batch: ArchiveOrgBatchResult = { items: [], missing: [], notReached: [], dryRun: opts.dryRun === true }; let fetched = 0; let consecutiveFailures = 0; let refusedItemsInARow = 0; const stop = (why: string) => { batch.stopped = why; }; for (let i = 0; i < opts.items.length; i++) { const req = opts.items[i]; if (batch.stopped) { batch.notReached.push(req.identifier); continue; } if (opts.signal.aborted) { stop("cancelled"); batch.notReached.push(req.identifier); continue; } if (opts.drainSignal?.aborted) { stop("drained"); batch.notReached.push(req.identifier); continue; } if (i > 0) { await sleep(archiveOrgItemGapMs(random()), opts.signal); if (opts.signal.aborted) { stop("cancelled"); batch.notReached.push(req.identifier); continue; } } let plan: ArchiveOrgImportPlan; try { plan = await planArchiveOrgImport({ identifier: req.identifier, selection: req.selection, dataDir, handling: opts.channelConfig.handling, client: deps.client, signal: opts.signal, }); } catch (err) { const e = err as ArchiveOrgRequestError; if (e instanceof ArchiveOrgRequestError && e.rateLimited) { batch.rateLimited = true; stop(`archive.org did not answer for item ${req.identifier}; stopping (re-run later — held records are skipped)`); log(`${batch.stopped}.\n`); batch.notReached.push(req.identifier); continue; } if (opts.signal.aborted) { stop("cancelled"); batch.notReached.push(req.identifier); continue; } batch.missing.push({ identifier: req.identifier, error: e.message }); log(`archive.org item ${req.identifier}: ${e.message} — passed over.\n`); continue; } const result: ArchiveOrgImportResult = { identifier: plan.identifier, planned: plan.entries.length, imported: [], skipped: [], restricted: [], failed: [], unknown: plan.unknown, }; batch.items.push(result); const held = plan.entries.filter((e) => e.onDisk).length; const restricted = plan.entries.filter((e) => e.restricted && !e.onDisk).length; log( `archive.org item ${plan.identifier}${plan.title ? ` ("${plan.title}")` : ""}: ` + `${plan.mediaFiles} media files, ${plan.entries.length} chosen, ` + `${held} already held` + `${restricted ? `, ${restricted} restricted${plan.restrictedItem ? " (the item is access-restricted)" : ""}` : ""}.\n`, ); if (plan.unknown.length > 0) { log(`Not media files of the item (ignored): ${plan.unknown.map((n) => JSON.stringify(n)).join(", ")}\n`); } if (opts.dryRun) { for (const e of plan.entries) { const state = e.onDisk ? "held " : e.restricted ? "RESTRICTED" : "would get"; log(` ${state} ${e.id} ${e.file}\n`); } result.skipped = plan.entries.filter((e) => e.onDisk).map((e) => e.file); result.restricted = plan.entries.filter((e) => !e.onDisk && e.restricted).map((e) => e.file); continue; } let refusedHere = false; for (const entry of plan.entries) { if (opts.signal.aborted) { stop("cancelled"); break; } if (opts.drainSignal?.aborted) { stop("drained"); break; } if (entry.onDisk || (await isHeldArchiveOrgRecord(dataDir, entry.id, opts.channelConfig.handling))) { result.skipped.push(entry.file); continue; } if (entry.restricted || refusedHere) { result.restricted.push(entry.file); continue; } const gate = await diskGate(opts.paths, settings, { dir: dataDir }); if (!gate.ok) { stop(gate.message); log(`Stopping: ${gate.message}.\n`); break; } if (fetched > 0) { const gap = archiveOrgGapMs(sleepSeconds, random()); log(`Waiting ${(gap / 1000).toFixed(1)}s before the next file (archive.org pacing)...\n`); await sleep(gap, opts.signal); if (opts.signal.aborted) { stop("cancelled"); break; } } fetched++; log(`[${fetched}] ${entry.file} → data/${entry.id}/\n`); let rec: DownloadOutcomeRecord | null = null; let error = ""; try { rec = await downloadOne({ channelSlug: opts.slug, channelConfig: opts.channelConfig, paths: opts.paths, videoUrl: entry.url, onLog: log, signal: opts.signal, cookiePolicy: resolveCookiePolicy(settings, opts.channelConfig), inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback, globalSkipLiveDownloads: settings.skipLiveDownloads, appendArchive: true, }); } catch (err) { error = (err as Error).message; } if (rec && isOk(rec)) { consecutiveFailures = 0; refusedItemsInARow = 0; result.imported.push(entry.file); await mergeRosterFile( opts.paths, opts.slug, [{ id: entry.id, url: entry.url }], new Date().toISOString(), "import", ).catch(() => { /* the download succeeded; a roster write failure must not fail it */ }); deps.onImported?.(entry.id); continue; } const why = error || rec?.attempts.at(-1)?.error || rec?.status || "failed"; if (isRefusalError(why)) { // archive.org will not hand this item's files to us: the rest of the // item is skipped as restricted, and it is not a failure streak. refusedHere = true; result.restricted.push(entry.file); log(` refused (${why}) — the rest of ${plan.identifier} is skipped as restricted\n`); continue; } consecutiveFailures++; result.failed.push({ file: entry.file, error: why }); log(` failed: ${why}\n`); if (rec?.failureClass === "rate_limit") { result.rateLimited = true; batch.rateLimited = true; stop("archive.org rate-limited the download; stopping (re-run later — held records are skipped)"); log(`${batch.stopped}.\n`); break; } if (consecutiveFailures >= ARCHIVE_ORG_MAX_CONSECUTIVE_FAILURES) { stop(`${consecutiveFailures} failures in a row; stopping`); log(`${batch.stopped}.\n`); break; } } if (batch.stopped) result.stopped = batch.stopped; if (refusedHere) { refusedItemsInARow++; if (refusedItemsInARow >= ARCHIVE_ORG_MAX_REFUSED_ITEMS && !batch.stopped) { batch.refusedStorm = true; stop(`archive.org refused ${refusedItemsInARow} items in a row (401/403); stopping`); result.stopped = batch.stopped; log(`${batch.stopped}.\n`); } } log( `archive.org import of ${plan.identifier}: ${result.imported.length} imported, ` + `${result.skipped.length} already held, ${result.restricted.length} restricted, ${result.failed.length} failed` + `${result.stopped ? ` — stopped: ${result.stopped}` : ""}.\n`, ); } return batch; } // The batch's totals, for its last log line (`summary: {…}`) and the action's // verdict. export function summarizeArchiveOrgBatch(b: ArchiveOrgBatchResult): { dryRun: boolean; items: number; planned: number; imported: number; held: number; restricted: { identifier: string; files: string[] }[]; failed: number; missing: string[]; notReached: string[]; stopped?: string; } { const sum = (f: (r: ArchiveOrgImportResult) => number) => b.items.reduce((n, r) => n + f(r), 0); return { dryRun: b.dryRun, items: b.items.length, planned: sum((r) => r.planned), imported: sum((r) => r.imported.length), held: sum((r) => r.skipped.length), restricted: b.items .filter((r) => r.restricted.length > 0) .map((r) => ({ identifier: r.identifier, files: r.restricted })), failed: sum((r) => r.failed.length), missing: b.missing.map((m) => m.identifier), notReached: b.notReached, ...(b.stopped ? { stopped: b.stopped } : {}), }; } // ONE ITEM — the original `import-archive-org {item}` — as a batch of one. export async function runArchiveOrgImport( opts: BatchOpts & { identifier: string; selection: ArchiveOrgFileSelection }, ): Promise { const { identifier, selection, ...rest } = opts; const batch = await runArchiveOrgBatchImport({ ...rest, items: [{ identifier, selection }] }); // A single item's metadata failure is the import's error, as it always was. if (batch.missing.length > 0) throw new Error(batch.missing[0].error); const result = batch.items[0]; if (!result) { if (batch.stopped && batch.stopped !== "cancelled" && batch.stopped !== "drained") { throw new Error(batch.stopped); } return { identifier, planned: 0, imported: [], skipped: [], restricted: [], failed: [], unknown: [], ...(batch.stopped ? { stopped: batch.stopped } : {}), ...(batch.rateLimited ? { rateLimited: true } : {}), }; } return { ...result, ...(batch.rateLimited ? { rateLimited: true } : {}) }; }