// ONE archive.org RECORD DOWNLOADED — without yt-dlp. // // Every managed download of an archive.org URL lands here // (ytdlp/downloadOneManaged.ts routes it before any yt-dlp spawn): the // single-URL import, each file of a bulk import (controller/archiveOrgImport.ts), // a re-download, "Persist source video". yt-dlp's ArchiveOrg extractor scrapes // the item's page, and fails on items whose page it does not understand ("opening // play-av tag not found" on an audio item); the item's metadata API, which the // import has already asked for (lib/archiveOrgClient.ts caches it), names every // file, its size and its checksums. So the record is built from that. // // THE LADDER, for the one media file the record is (lib/archiveOrg.ts // archiveOrgRecordFile → archiveOrgFetchFile): // // 1. BITTORRENT (the operator: "use torrents when possible to be extra polite // to archive.org") — when settings.archiveOrg.torrent is on, aria2c is // installed, the item lists `_archive.torrent`, and the // torrent carries the file. aria2c fetches only that file from the swarm // and archive.org's web seed, then seeds it (lib/archiveOrgTorrent-server.ts). // A stall, an aria2c failure, or a torrent that cannot be used falls back: // "fell back to direct download: ". // 2. DIRECT — https://archive.org/download//, one stream, // resumed with a Range request, backing off on 429/503 // (ArchiveOrgClient.downloadFile). // // Whichever fetched it, the file is VERIFIED against the item's sha1 (else // md5, else size). A mismatch deletes it and falls back once to a direct // download; a second mismatch fails the record. // // THEN THE RECORD, in data// as any managed download leaves it: // - archiveorg.json, the provenance (kept when already there; the mirror's // uploaded info.json read for the original's title and date); // - metadata.info.json, synthesised (lib/archiveOrg.ts archiveOrgInfoJson) // with the duration ffprobe measures, inside the metadata history; // - audio.: an audio file already in the channel's format IS the audio; // anything else is source-media., extracted by the app's own // extraction (ytdlp/finalizeAppExtraction.ts) — a video kept in the // saved-video store when the persistence plan says so, else removed; an // audio original is never kept as a "source video"; // - the media tier's hook, the archive line `archiveorg `, and // download-outcome.json with the attempts ("archiveorg-torrent", // "archiveorg-direct"). // // A CANCEL kills aria2c by its process group and leaves the partial in the // staging dir (`.archiveorg-fetch/` under data//), so the next run resumes. import path from "node:path"; import { createHash } from "node:crypto"; import { createReadStream } from "node:fs"; import { appendFile, mkdir, readdir, rename, rm, stat } from "node:fs/promises"; import type { ManagedDownloadOpts } from "../ytdlp/downloadOneManaged"; import type { AudioFormat } from "../lib/channelConfig"; import type { DownloadAttempt, DownloadAttemptKind, DownloadOutcomeRecord, DownloadOutcomeStatus, } from "../lib/downloadOutcome"; import type { DownloadFailureClass } from "../lib/availability"; import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { archiveOrgFetchFile, archiveOrgInfoJson, archiveOrgRecordFile, buildArchiveOrgProvenance, isArchiveOrgVideoFileName, listArchiveOrgMediaFiles, type ArchiveOrgItemFile, type ArchiveOrgItemMetadata, type ArchiveOrgProvenance, } from "../lib/archiveOrg"; import { archiveOrgVideoId, parseArchiveOrgUrl, type ArchiveOrgRef } from "../lib/archiveOrgId"; import { ARCHIVE_ORG_USER_AGENT, ArchiveOrgRequestError, archiveOrgClient, type ArchiveOrgClient, } from "../lib/archiveOrgClient"; import { ensureArchiveOrgProvenanceSidecar } from "../lib/archiveOrg-server"; import { DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS, findTorrentFile, parseTorrent, type ArchiveOrgFetchSettings, } from "../lib/archiveOrgTorrent"; import { aria2cAvailable as defaultAria2cAvailable, fetchFileByTorrent, type TorrentFetchOpts, type TorrentFetchResult, } from "../lib/archiveOrgTorrent-server"; import { AUDIO_EXTS, MEDIA_EXTS } from "../lib/mediaFiles"; import { tierMediaFile, tierVideoDir } from "../lib/mediaTier-server"; import { withMetadataHistory } from "../lib/metadataHistory-server"; import { writeJsonAtomic } from "../lib/jsonFile-server"; import { getSettings } from "../lib/settings"; import { formatBytes } from "../lib/format"; import { isDoNotClean } from "../lib/doNotClean-server"; import { isInKeepWindow, uploadKeyFor } from "./keptVideos"; import { resolvePersistenceDecision } from "../ytdlp/persistencePlan"; import { finalizeAppExtraction } from "../ytdlp/finalizeAppExtraction"; import { probeMediaDurationSec } from "../ytdlp/ffprobeDuration"; import { transcribeWithWorker } from "./transcribeOne"; export const ARCHIVE_ORG_STAGING_DIR = ".archiveorg-fetch"; export type ArchiveOrgVerifyResult = | { ok: true; by: "sha1" | "md5" | "size" | "nothing" } | { ok: false; reason: string }; export type ArchiveOrgDownloadDeps = { client?: ArchiveOrgClient; settings?: ArchiveOrgFetchSettings; aria2cAvailable?: (bin: string) => Promise; fetchByTorrent?: (o: TorrentFetchOpts) => Promise; // The plain download: the client's, by default. fetchDirect?: (o: { identifier: string; file: string; dest: string; expectedSize?: number; signal: AbortSignal; onLog: (s: string) => void; }) => Promise; verify?: (file: string, entry: ArchiveOrgItemFile) => Promise; probeDuration?: (file: string) => Promise; finalize?: typeof finalizeAppExtraction; transcribe?: typeof transcribeWithWorker; }; // ─── Verification ─── // The file against archive.org's checksum: sha1 when the item lists one, else // md5, else the size alone, else nothing to check against. One read. export async function verifyArchiveOrgFile( file: string, entry: ArchiveOrgItemFile, ): Promise { const size = Number(entry.size); const st = await stat(file).catch(() => null); if (!st) return { ok: false, reason: "the file is missing" }; if (Number.isFinite(size) && size > 0 && st.size !== size) { return { ok: false, reason: `size ${st.size} bytes, archive.org lists ${size}` }; } const want = entry.sha1 ? { algo: "sha1" as const, hex: entry.sha1 } : entry.md5 ? { algo: "md5" as const, hex: entry.md5 } : null; if (!want) return { ok: true, by: Number.isFinite(size) && size > 0 ? "size" : "nothing" }; const hash = createHash(want.algo); await new Promise((resolve, reject) => { createReadStream(file) .on("data", (c) => hash.update(c)) .on("end", () => resolve()) .on("error", reject); }); const got = hash.digest("hex"); if (got.toLowerCase() !== want.hex.trim().toLowerCase()) { return { ok: false, reason: `${want.algo} ${got}, archive.org lists ${want.hex}` }; } return { ok: true, by: want.algo }; } // ─── Small helpers ─── function extOf(name: string): string { const m = /\.([A-Za-z0-9]{1,8})$/.exec(name); return m ? m[1].toLowerCase() : ""; } async function hasTranscriptOnDisk(videoDir: string): Promise { const entries = await readdir(videoDir).catch(() => [] as string[]); return entries.some((e) => { if (e === "transcript.json") return true; const m = /^transcript\.([^.]+)\.(?:vtt|json|json3|srv1|srv2|srv3)$/.exec(e); return !!m && m[1] !== "live_chat"; }); } function failureClassOf(err: unknown): DownloadFailureClass { if (err instanceof ArchiveOrgRequestError) { if (err.rateLimited) return "rate_limit"; if (err.status === 403 || err.status === 404 || err.status === 410) return "per_video"; return err.status === null ? "network" : "unknown"; } return "network"; } function attempt(n: number, kind: DownloadAttemptKind, handling: DownloadAttempt["handling"], error?: string): DownloadAttempt { return { n, kind, handling, usedCookies: false, // No yt-dlp ran: 0 for a fetch that delivered a verified file, 1 for one // that did not, so every reader of the field keeps meaning "succeeded". ytdlpExitCode: error ? 1 : 0, ...(error ? { error } : {}), }; } // ─── The download ─── export async function downloadArchiveOrgManaged( opts: ManagedDownloadOpts, deps: ArchiveOrgDownloadDeps = {}, ): Promise { const startedAt = new Date().toISOString(); const log = (s: string) => opts.onLog(s.endsWith("\n") ? s : `${s}\n`); const client = deps.client ?? archiveOrgClient; const handling = opts.channelConfig.handling; const channelDir = path.join(opts.paths.channelsDir, opts.channelSlug); const attempts: DownloadAttempt[] = []; const ref: ArchiveOrgRef | null = parseArchiveOrgUrl(opts.videoUrl); if (!ref) { log(`Not an archive.org item URL: ${opts.videoUrl}`); return { videoId: "unknown", webpageUrl: opts.videoUrl, status: "failed", startedAt, finishedAt: new Date().toISOString(), attempts: [attempt(1, "archiveorg-direct", handling, "not an archive.org item URL")], failureClass: "per_video", }; } const videoId = archiveOrgVideoId(ref); const videoDir = path.join(channelDir, "data", videoId); const staging = path.join(videoDir, ARCHIVE_ORG_STAGING_DIR); await mkdir(videoDir, { recursive: true }); const finish = async (o: { status: DownloadOutcomeStatus; failureClass?: DownloadFailureClass; fellBackToTranscribe?: boolean; }): Promise => { const record: DownloadOutcomeRecord = { videoId, webpageUrl: opts.videoUrl, status: o.status, startedAt, finishedAt: new Date().toISOString(), attempts, ...(o.failureClass ? { failureClass: o.failureClass } : {}), ...(o.fellBackToTranscribe ? { fellBackToTranscribe: true } : {}), }; try { // THE MEDIA TIER'S HOOK (release 17): what this run finalised moves into // channels//media when the channel has one. Never throws. await tierVideoDir(videoDir, { onLog: opts.onLog }); await writeDownloadOutcome(videoDir, record); } catch (err) { log(`Failed to write download-outcome.json: ${(err as Error).message}`); } return record; }; const fail = (kind: DownloadAttemptKind, error: string, failureClass: DownloadFailureClass) => { attempts.push(attempt(attempts.length + 1, kind, handling, error)); return finish({ status: "failed", failureClass }); }; // ---------- The item ---------- let item: ArchiveOrgItemMetadata; try { item = await client.itemMetadata(ref.identifier, opts.signal); } catch (err) { log(`archive.org metadata for ${ref.identifier} not fetched: ${(err as Error).message}`); return fail("archiveorg-direct", (err as Error).message, failureClassOf(err)); } const identifier = item.metadata.identifier || ref.identifier; const file = archiveOrgRecordFile(item, ref.file); if (!file) { const media = listArchiveOrgMediaFiles(item).length; const why = ref.file ? `archive.org item "${identifier}" has no file "${ref.file}"` : media === 0 ? `archive.org item "${identifier}" has no media files` : `archive.org item "${identifier}" holds ${media} media files; import one by its file URL`; log(why); return fail("archiveorg-direct", why, "per_video"); } const fetchEntry = archiveOrgFetchFile(item, file)!; const fetchSize = Number(fetchEntry.size); log( `archive.org: ${identifier} / ${file}` + (fetchEntry.name !== file ? ` — fetching archive.org's ${fetchEntry.format ?? extOf(fetchEntry.name)} of it, ${fetchEntry.name}` : "") + (Number.isFinite(fetchSize) && fetchSize > 0 ? ` (${formatBytes(fetchSize)})` : "") + ".", ); // ---------- Provenance (before the bytes: it is what the record says) ---------- let prov: ArchiveOrgProvenance | null; try { prov = await ensureArchiveOrgProvenanceSidecar( videoDir, { identifier, ...(ref.file ? { file: ref.file } : {}) }, { client, signal: opts.signal, onLog: opts.onLog }, ); } catch { return fail("archiveorg-direct", "cancelled", "unknown"); } if (!prov) { // archive.org answered the metadata a moment ago but not this: the record // is built from the item alone (the next download fills the sidecar in). prov = buildArchiveOrgProvenance({ ref: { identifier, ...(ref.file ? { file: ref.file } : {}) }, item, fetchedAt: new Date().toISOString(), }); } // ---------- The bytes ---------- const settings = deps.settings ?? (getSettings().archiveOrg ?? DEFAULT_ARCHIVE_ORG_FETCH_SETTINGS); const verify = deps.verify ?? verifyArchiveOrgFile; const aria2cBin = opts.paths.aria2cBin ?? "aria2c"; let got: string | null = null; let mismatches = 0; if (settings.torrent) { const why = await (async (): Promise => { if (!(await (deps.aria2cAvailable ?? defaultAria2cAvailable)(aria2cBin))) { return "torrents need aria2c (not found on PATH; ARIA2C_BIN names another binary)"; } const torrentName = `${identifier}_archive.torrent`; if (!item.files.some((f) => f.name === torrentName)) return "the item lists no torrent"; let buf: Buffer; try { buf = await client.itemTorrent(identifier, opts.signal); } catch (err) { if (opts.signal.aborted) return "cancelled"; return `the torrent was not fetched (${(err as Error).message})`; } const parsed = parseTorrent(buf); if (!parsed) return "the torrent could not be read"; const entry = findTorrentFile(parsed, fetchEntry.name); if (!entry) return `the item's torrent does not carry ${fetchEntry.name} (older than the file?)`; const res = await (deps.fetchByTorrent ?? fetchFileByTorrent)({ aria2cBin, torrent: buf, parsed, entry, stagingDir: path.join(staging, "torrent"), settings, userAgent: ARCHIVE_ORG_USER_AGENT, onLog: opts.onLog, signal: opts.signal, }); if (!res.ok) { attempts.push(attempt(attempts.length + 1, "archiveorg-torrent", handling, res.reason)); return res.cancelled ? "cancelled" : res.reason; } const v = await verify(res.file, fetchEntry); if (!v.ok) { mismatches++; attempts.push(attempt(attempts.length + 1, "archiveorg-torrent", handling, `checksum mismatch: ${v.reason}`)); return `checksum mismatch (${v.reason})`; } attempts.push(attempt(attempts.length + 1, "archiveorg-torrent", handling)); log(`Verified ${fetchEntry.name} (${v.by}).`); got = res.file; return null; })(); if (why === "cancelled" || opts.signal.aborted) { log("Cancelled; the partial download is kept for the next run."); if (attempts.at(-1)?.error !== "cancelled") attempts.push(attempt(attempts.length + 1, "archiveorg-torrent", handling, "cancelled")); return finish({ status: "failed" }); } if (why) { log(`fell back to direct download: ${why}`); await rm(path.join(staging, "torrent"), { recursive: true, force: true }); } } else { log("archiveOrg.torrent is off; downloading directly."); } while (!got) { const dest = path.join(staging, "direct", fetchEntry.name.split("/").pop() || "file"); await mkdir(path.dirname(dest), { recursive: true }); try { await (deps.fetchDirect ?? defaultFetchDirect(client))({ identifier, file: fetchEntry.name, dest, ...(Number.isFinite(fetchSize) && fetchSize > 0 ? { expectedSize: fetchSize } : {}), signal: opts.signal, onLog: opts.onLog, }); } catch (err) { if (opts.signal.aborted) { attempts.push(attempt(attempts.length + 1, "archiveorg-direct", handling, "cancelled")); log("Cancelled; the partial download is kept for the next run."); return finish({ status: "failed" }); } log(`Direct download failed: ${(err as Error).message}`); return fail("archiveorg-direct", (err as Error).message, failureClassOf(err)); } const v = await verify(dest, fetchEntry); if (v.ok) { attempts.push(attempt(attempts.length + 1, "archiveorg-direct", handling)); log(`Verified ${fetchEntry.name} (${v.by}).`); got = dest; break; } mismatches++; await rm(dest, { force: true }); if (mismatches >= 2) { log(`Checksum mismatch again (${v.reason}); the record fails.`); return fail("archiveorg-direct", `checksum mismatch: ${v.reason}`, "per_video"); } attempts.push(attempt(attempts.length + 1, "archiveorg-direct", handling, `checksum mismatch: ${v.reason}`)); log(`fell back to direct download: checksum mismatch (${v.reason}); downloading once more.`); } // ---------- The record ---------- const fetchedPath: string = got; const durationSec = (await (deps.probeDuration ?? ((f: string) => probeMediaDurationSec({ ffprobeBin: opts.paths.ffprobeBin, file: f, signal: opts.signal, onLog: () => {} })))( fetchedPath, )) ?? undefined; const info = archiveOrgInfoJson({ prov, item, file, fetched: fetchEntry, durationSec }); try { await withMetadataHistory(videoDir, { by: "archiveorg-import", onLog: opts.onLog }, () => writeJsonAtomic(path.join(videoDir, "metadata.info.json"), info, { indent: 0, newline: false }), ); } catch (err) { return fail("archiveorg-direct", `metadata.info.json not written: ${(err as Error).message}`, "unknown"); } const fmt: AudioFormat = opts.audioFormatOverride ?? opts.channelConfig.audioFormat ?? "mp3"; const ext = extOf(fetchEntry.name); const isVideo = isArchiveOrgVideoFileName(fetchEntry.name); const isAudio = !isVideo && AUDIO_EXTS.includes(ext); const hasTranscript = await hasTranscriptOnDisk(videoDir); // "Persist source video" over a record that already has a transcript keeps // the container and extracts nothing (the managed download's keepTranscript). const keepTranscript = opts.forceMedia === true && hasTranscript; const plan = resolvePersistenceDecision({ inWindow: isInKeepWindow(uploadKeyFor(info.upload_date as string | undefined, videoId), opts.keepWindow), pinned: await isDoNotClean(videoDir), channelExtractionMode: opts.channelConfig.extractionMode, overrides: { keepSourceVideoOverride: opts.keepSourceVideoOverride, extractImmediately: opts.extractImmediately, }, }); const persist = isVideo && plan.persist; if (isVideo) { log(`Persistence: ${persist ? "keep source video" : "audio-only"} (${plan.reason}).`); } if (isAudio && ext === fmt && !keepTranscript) { // Already the channel's audio format: it IS the audio. const name = `audio.${fmt}`; await rename(fetchedPath, path.join(videoDir, name)); await tierMediaFile(videoDir, name, { onLog: opts.onLog }); log(`Wrote ${name} (${fetchEntry.name}, no transcode).`); } else if (keepTranscript && !persist) { log( `forceMedia: a transcript is on disk and ${isVideo ? `this download keeps no source video (${plan.reason})` : "an audio original is not kept as a source video"}; nothing kept.`, ); await rm(fetchedPath, { force: true }); } else { const sourceName = `source-media.${MEDIA_EXTS.includes(ext) ? ext : ext || "bin"}`; await rename(fetchedPath, path.join(videoDir, sourceName)); await (deps.finalize ?? finalizeAppExtraction)({ paths: opts.paths, channelSlug: opts.channelSlug, channelConfig: opts.channelConfig, videoDir, videoId, fmt, persist, category: plan.category, origin: opts.persistOrigin, extractAudio: !keepTranscript, sourceFilename: sourceName, onLog: opts.onLog, signal: opts.signal, }); // An audio original whose extraction failed is still the only copy of the // recording: finalize keeps it, and the record fails below for want of // audio. A video that was not persisted has been removed by finalize. } await rm(staging, { recursive: true, force: true }); const entries = await readdir(videoDir).catch(() => [] as string[]); const haveAudio = entries.includes(`audio.${fmt}`); if (!keepTranscript && !haveAudio) { // Whatever finalize kept (the container) stays for a retry to use. return fail("archiveorg-direct", `no audio.${fmt} was produced from ${fetchEntry.name}`, "unknown"); } // ---------- Transcription hand-off ---------- // archive.org has no captions: a youtube-handling channel's record is the // no-subs fallback's, a transcribe-handling channel's is a plain download. let status: DownloadOutcomeStatus = "ok"; const fellBackToTranscribe = handling === "youtube" && !keepTranscript; if (fellBackToTranscribe && opts.inlineTranscribeOnFallback && !opts.signal.aborted) { try { await (deps.transcribe ?? transcribeWithWorker)({ paths: opts.paths, videoDir, videoId, audioFilename: `audio.${fmt}`, onLog: opts.onLog, signal: opts.signal, }); status = "ok-auto-transcribed"; } catch (err) { log(`Whisper failed after the archive.org download: ${(err as Error).message}`); status = "failed"; } } else if (!keepTranscript) { log(`Audio downloaded for ${videoId}; it is transcribed by the next "Transcribe missing" pass.`); } if (status !== "failed" && opts.appendArchive !== false) { try { await mkdir(channelDir, { recursive: true }); await appendFile(path.join(channelDir, "archive"), `archiveorg ${videoId}\n`); } catch (err) { log(`Archive append failed (archiveorg ${videoId}): ${(err as Error).message}`); } } return finish({ status, fellBackToTranscribe }); } function defaultFetchDirect(client: ArchiveOrgClient): NonNullable { return async ({ identifier, file, dest, expectedSize, signal, onLog }) => { let lastLogAt = 0; onLog(`direct: https://archive.org/download/${identifier}/${file}\n`); await client.downloadFile(identifier, file, dest, { signal, ...(expectedSize ? { expectedSize } : {}), onProgress: ({ bytes, total }) => { const t = Date.now(); if (t - lastLogAt < 15_000) return; lastLogAt = t; onLog( `direct: ${file} ${formatBytes(bytes)}${total ? ` of ${formatBytes(total)} (${Math.floor((bytes / total) * 100)}%)` : ""}\n`, ); }, }); }; }