// archive.org PROVENANCE ON DISK — the `archiveorg.json` sidecar. // // `ensureArchiveOrgProvenanceSidecar` is what every archive.org download runs // (controller/archiveOrgDownload.ts), before it writes the record's // metadata.info.json from the item. `ensureArchiveOrgProvenance` is the older, // whole step for a record yt-dlp wrote (archive.org records were fetched by // yt-dlp before the torrent slice): the sidecar, then that record corrected. // // THE WHOLE STEP, `ensureArchiveOrgProvenance`: // // 1. the sidecar: kept when it is already there for this item and file; // otherwise built from the item's metadata API (one cached request per // item, lib/archiveOrgClient.ts) and, when the item carries one, the // `.info.json` a mirroring tool uploaded beside the file (one request). // 2. the record: metadata.info.json corrected from it // (lib/archiveOrg.ts archiveOrgMetadataPatch) through `patchMetadataInfo`, // so the change is in metadata.history.json as `archiveorg-provenance`. // // Step 2 needs no network and always runs: a sidecar that could not be fetched // (archive.org refusing, the operator offline) still leaves the record's // `webpage_url` the file's page, which is what keeps its directory its own. // A failure is logged and does not fail the download — the next download of // the record, or a re-import, fills it in. import path from "node:path"; import { readFile } from "node:fs/promises"; import { ARCHIVE_ORG_PROVENANCE_FILENAME, archiveOrgMetadataPatch, buildArchiveOrgProvenance, coerceArchiveOrgProvenance, findArchiveOrgInfoJson, type ArchiveOrgProvenance, } from "./archiveOrg"; import { archiveOrgDetailsUrl, parseArchiveOrgUrl, type ArchiveOrgRef, } from "./archiveOrgId"; import { archiveOrgClient, type ArchiveOrgClient } from "./archiveOrgClient"; import { patchMetadataInfo } from "./metadataHistory-server"; import { sidecar, sidecarField } from "./sidecar-server"; export const archiveOrgProvenanceSidecar = sidecar( ARCHIVE_ORG_PROVENANCE_FILENAME, sidecarField(coerceArchiveOrgProvenance), ); export const { load: loadArchiveOrgProvenance, write: writeArchiveOrgProvenance, } = archiveOrgProvenanceSidecar; // Fetch what the sidecar records: the item's metadata (cached) and, for a // mirror, its uploaded info.json. export async function fetchArchiveOrgProvenance( ref: ArchiveOrgRef, opts: { client?: ArchiveOrgClient; signal?: AbortSignal; now?: () => Date } = {}, ): Promise { const client = opts.client ?? archiveOrgClient; const item = await client.itemMetadata(ref.identifier, opts.signal); const infoName = findArchiveOrgInfoJson(item, ref.file); let infoJson: unknown; if (infoName) { try { infoJson = await client.itemJsonFile(item.metadata.identifier, infoName, opts.signal); } catch { // The names still say whether it is a mirror; the info.json only adds // the original's title and date. infoJson = undefined; } } return buildArchiveOrgProvenance({ ref, item, infoJson, fetchedAt: (opts.now?.() ?? new Date()).toISOString(), }); } async function readInfo(videoDir: string): Promise | null> { try { const v = JSON.parse(await readFile(path.join(videoDir, "metadata.info.json"), "utf8")); return v && typeof v === "object" && !Array.isArray(v) ? (v as Record) : null; } catch { return null; } } export type EnsureArchiveOrgProvenanceOpts = { videoDir: string; // The URL the record was fetched by (a details/embed/download URL). videoUrl: string; onLog?: (line: string) => void; signal?: AbortSignal; client?: ArchiveOrgClient; now?: () => Date; }; // The sidecar for `ref`: the one on disk when it is for this item and file, // else fetched (the item's metadata is cached; a mirror's info.json is one more // request) and written. Null when archive.org could not be asked — logged, // never thrown, unless the job was cancelled. export async function ensureArchiveOrgProvenanceSidecar( videoDir: string, ref: ArchiveOrgRef, opts: { client?: ArchiveOrgClient; signal?: AbortSignal; now?: () => Date; onLog?: (line: string) => void } = {}, ): Promise { const log = opts.onLog ?? (() => {}); let prov = await loadArchiveOrgProvenance(videoDir); if (prov && (prov.identifier !== ref.identifier || (prov.file ?? "") !== (ref.file ?? ""))) { prov = null; } if (prov) return prov; try { prov = await fetchArchiveOrgProvenance(ref, opts); await writeArchiveOrgProvenance(videoDir, prov); log( `archive.org provenance: ${prov.identifier}${prov.file ? ` / ${prov.file}` : ""}` + (prov.mirror ? ` — mirror of YouTube ${prov.mirror.id} (from ${prov.mirror.from})` : "") + `.\n`, ); return prov; } catch (err) { if (opts.signal?.aborted) throw err; log(`archive.org provenance not fetched (${(err as Error).message}).\n`); return null; } } export async function ensureArchiveOrgProvenance( opts: EnsureArchiveOrgProvenanceOpts, ): Promise { const log = opts.onLog ?? (() => {}); const ref = parseArchiveOrgUrl(opts.videoUrl); if (!ref) return null; const prov = await ensureArchiveOrgProvenanceSidecar(opts.videoDir, ref, opts); const info = await readInfo(opts.videoDir); if (!info) return prov; // Without a sidecar only the page is corrected — it is the one field the // directory's name depends on. const patch = prov ? archiveOrgMetadataPatch(prov, info) : info.webpage_url === archiveOrgDetailsUrl(ref) ? {} : { webpage_url: archiveOrgDetailsUrl(ref) }; if (Object.keys(patch).length > 0) { try { await patchMetadataInfo(opts.videoDir, patch, { by: "archiveorg-provenance", onLog: log }); } catch (err) { log(`Could not correct metadata.info.json from the archive.org provenance: ${(err as Error).message}\n`); } } return prov; }