// archive.org AS A SOURCE — the pure half: what an item's metadata says, the // provenance a record keeps, the links a citation derives from it. // // Pure (no node:, no fetch) so the browser, the export build and the report // compose can all use it. The network half — the polite metadata client — is // lib/archiveOrgClient.ts; the per-video sidecar and the metadata patch are // lib/archiveOrg-server.ts. // // THE RECORD. A video imported from archive.org keeps yt-dlp's // metadata.info.json like every other (extractor_key "ArchiveOrg"), plus one // sidecar, `archiveorg.json` (ArchiveOrgProvenance below): the identifier and // file, the item's own title/date/creator/collections, the torrent, and — when // the item is a mirror of a YouTube upload — the ORIGINAL's id, URL, title, // date and uploader, read from the info.json the mirroring tool uploaded beside // the media. // // WEB-ARCHIVE (WARC / Wayback) RECORDS ARE NOT THIS. A captured web page is a // different kind of record with its own reader; it would get its own platform // and module beside this one, never a branch in it. import { archiveOrgDetailsUrl, archiveOrgDownloadUrl, archiveOrgTorrentUrl, archiveOrgVideoId, parseArchiveOrgUrl, dateFromMirrorFileName, titleFromMirrorFileName, youtubeIdFromFileName, youtubeIdFromIdentifier, type ArchiveOrgRef, } from "./archiveOrgId"; export const ARCHIVE_ORG_PROVENANCE_FILENAME = "archiveorg.json"; export const ARCHIVE_ORG_PROVENANCE_VERSION = 1; // ─── The item, as the metadata API returns it ─── // https://archive.org/metadata/ — only the fields read here. export type ArchiveOrgItemFile = { name: string; // "original" | "derivative" | "metadata" source?: string; // archive.org's format name ("h.264", "MPEG4", "VBR MP3", "JSON", …). format?: string; size?: string; // Seconds, as a string ("123.45") or a clock ("02:03"). length?: string; title?: string; // On a derivative: the original it was made from. original?: string; // archive.org's checksums of the file, hex. What a fetched file is verified // against (controller/archiveOrgDownload.ts). md5?: string; sha1?: string; }; export type ArchiveOrgItemMetadata = { metadata: { identifier: string; title?: string | string[]; date?: string; publicdate?: string; creator?: string | string[]; uploader?: string; collection?: string | string[]; mediatype?: string; description?: string | string[]; subject?: string | string[]; }; files: ArchiveOrgItemFile[]; }; function firstString(v: unknown): string | undefined { if (typeof v === "string") return v.trim() || undefined; if (Array.isArray(v)) { for (const x of v) if (typeof x === "string" && x.trim()) return x.trim(); } return undefined; } function stringList(v: unknown): string[] { if (typeof v === "string") return v.trim() ? [v.trim()] : []; if (Array.isArray(v)) return v.filter((x): x is string => typeof x === "string" && !!x.trim()); return []; } // The metadata API answers `{}` (HTTP 200) for an identifier with no item, and // a "dark" item has metadata but no files. Null for anything without both. export function parseArchiveOrgItemMetadata(raw: unknown): ArchiveOrgItemMetadata | null { if (!raw || typeof raw !== "object") return null; const r = raw as { metadata?: unknown; files?: unknown }; const m = r.metadata as Record | undefined; if (!m || typeof m !== "object" || typeof m.identifier !== "string") return null; const files = Array.isArray(r.files) ? r.files.filter( (f): f is ArchiveOrgItemFile => !!f && typeof f === "object" && typeof (f as { name?: unknown }).name === "string", ) : []; return { metadata: m as ArchiveOrgItemMetadata["metadata"], files }; } // The extensions archive.org serves media under that yt-dlp can download. const MEDIA_EXTS = new Set([ "mp4", "m4v", "mkv", "webm", "mov", "avi", "mpeg", "mpg", "ogv", "flv", "wmv", "3gp", "ts", "mp3", "m4a", "ogg", "oga", "opus", "flac", "wav", "aac", "wma", ]); function extOf(name: string): string { const m = /\.([A-Za-z0-9]{1,8})$/.exec(name); return m ? m[1].toLowerCase() : ""; } export function isMediaFileName(name: string): boolean { return MEDIA_EXTS.has(extOf(name)); } // The item's media files: ORIGINALS only (a derivative is archive.org's // transcode of one, the same recording), in the item's own order. export function listArchiveOrgMediaFiles(item: ArchiveOrgItemMetadata): ArchiveOrgItemFile[] { return item.files.filter((f) => f.source === "original" && isMediaFileName(f.name)); } // ─── Picking files for a bulk import ─── export type ArchiveOrgFileSelection = | { files: string[] } | { match: string }; export type ArchiveOrgFilePick = { // The chosen media files, in the item's order. picked: string[]; // Named in `files` but not a media original of the item. unknown: string[]; }; // `files` names exact paths; `match` is a case-insensitive regex over each // media file's path. Either way only media originals can be picked. export function pickArchiveOrgFiles( item: ArchiveOrgItemMetadata, sel: ArchiveOrgFileSelection, ): ArchiveOrgFilePick { const media = listArchiveOrgMediaFiles(item).map((f) => f.name); if ("files" in sel) { const want = new Set(sel.files); const known = new Set(media); return { picked: media.filter((n) => want.has(n)), unknown: sel.files.filter((n) => !known.has(n)), }; } const re = new RegExp(sel.match, "i"); return { picked: media.filter((n) => re.test(n)), unknown: [] }; } // ─── The mirror's original ─── export type ArchiveOrgMirror = { platform: "youtube"; id: string; url: string; title?: string; // YYYYMMDD uploadDate?: string; uploader?: string; channelUrl?: string; description?: string; // Where the original's identity was read from: the uploaded info.json (and // then every field above is the original's own), or only a name. from: "info-json" | "identifier" | "file-name"; }; // The info.json a mirroring tool uploaded beside a media file: the file's own // stem + `.info.json`. For a single-media item, the item's only info.json. export function findArchiveOrgInfoJson( item: ArchiveOrgItemMetadata, file: string | undefined, ): string | null { const names = item.files.map((f) => f.name); const infos = names.filter((n) => n.endsWith(".info.json")); if (file) { const stem = file.replace(/\.[A-Za-z0-9]{1,8}$/, ""); if (infos.includes(`${stem}.info.json`)) return `${stem}.info.json`; return null; } return infos.length === 1 ? infos[0] : null; } function youtubeWatchUrl(id: string): string { return `https://www.youtube.com/watch?v=${id}`; } // The original, from an uploaded info.json when there is one (authoritative: // the record yt-dlp wrote when it downloaded the upload), else from the names. export function archiveOrgMirrorOf(opts: { identifier: string; file?: string; infoJson?: unknown; }): ArchiveOrgMirror | null { const info = opts.infoJson as Record | null | undefined; if (info && typeof info === "object") { const key = String(info.extractor_key ?? info.extractor ?? ""); const id = typeof info.id === "string" ? info.id : ""; if (/^youtube$/i.test(key) && /^[A-Za-z0-9_-]{11}$/.test(id)) { const str = (k: string) => (typeof info[k] === "string" && (info[k] as string).trim() ? (info[k] as string) : undefined); const date = str("upload_date"); return { platform: "youtube", id, url: youtubeWatchUrl(id), title: str("title"), uploadDate: date && /^\d{8}$/.test(date) ? date : undefined, uploader: str("uploader") ?? str("channel"), channelUrl: str("channel_url") ?? str("uploader_url"), description: str("description"), from: "info-json", }; } } const fromIdent = youtubeIdFromIdentifier(opts.identifier); if (fromIdent) return { platform: "youtube", id: fromIdent, url: youtubeWatchUrl(fromIdent), from: "identifier" }; const fromName = opts.file ? youtubeIdFromFileName(opts.file) : null; if (fromName) return { platform: "youtube", id: fromName, url: youtubeWatchUrl(fromName), from: "file-name" }; return null; } // ─── The provenance sidecar ─── export type ArchiveOrgProvenance = { version: number; identifier: string; // The file's path inside the item; absent for a whole item. file?: string; // The item's page, and the file's own page inside it. itemUrl: string; fileUrl?: string; // The media file's bytes, when the record is one file. downloadUrl?: string; // The item's BitTorrent file, when the item lists one. torrentUrl?: string; item: { title?: string; // archive.org's `date` (the content date the uploader gave), and // `publicdate` (when the item went up). date?: string; publicDate?: string; creator?: string; collections: string[]; mediatype?: string; }; // The file's own title in the item, when it has one. fileTitle?: string; mirror?: ArchiveOrgMirror; fetchedAt: string; }; export function buildArchiveOrgProvenance(opts: { ref: ArchiveOrgRef; item: ArchiveOrgItemMetadata; infoJson?: unknown; fetchedAt: string; }): ArchiveOrgProvenance { const { ref, item } = opts; const m = item.metadata; const identifier = m.identifier || ref.identifier; const fileEntry = ref.file ? item.files.find((f) => f.name === ref.file) : undefined; const torrentName = `${identifier}_archive.torrent`; const hasTorrent = item.files.some((f) => f.name === torrentName); const mirror = archiveOrgMirrorOf({ identifier, file: ref.file, infoJson: opts.infoJson }); // The uploader field is the uploading account's e-mail address; it is never // kept. `creator` is the public credit. const prov: ArchiveOrgProvenance = { version: ARCHIVE_ORG_PROVENANCE_VERSION, identifier, ...(ref.file ? { file: ref.file } : {}), itemUrl: archiveOrgDetailsUrl({ identifier }), ...(ref.file ? { fileUrl: archiveOrgDetailsUrl({ identifier, file: ref.file }), downloadUrl: archiveOrgDownloadUrl(identifier, ref.file), } : {}), ...(hasTorrent ? { torrentUrl: archiveOrgTorrentUrl(identifier) } : {}), item: { title: firstString(m.title), date: firstString(m.date), publicDate: firstString(m.publicdate), creator: firstString(m.creator), collections: stringList(m.collection), mediatype: firstString(m.mediatype), }, ...(fileEntry?.title?.trim() ? { fileTitle: fileEntry.title.trim() } : {}), ...(mirror ? { mirror } : {}), fetchedAt: opts.fetchedAt, }; return stripUndefinedDeep(withFileNameFields(prov)); } // A mirror's original known only by name (no info.json, or one without these // fields) for one file of a multi-file item takes them from the file's name // (archiveOrgId.ts): // // title titleFromMirrorFileName, unless the item gives the file a // title of its own (`fileTitle`) // uploadDate dateFromMirrorFileName, the leading `[_]YYYYMMDD` // // What a name gives is re-derived every time, so a refresh follows the rule; // an info.json's title and date are the original's own and always win. // Returns `prov` itself when nothing changes. export function withFileNameFields(prov: ArchiveOrgProvenance): ArchiveOrgProvenance { const mirror = prov.mirror; if (!prov.file || !mirror) return prov; const byName = mirror.from !== "info-json"; const next: ArchiveOrgMirror = { ...mirror }; if (!prov.fileTitle && (byName || !mirror.title)) { const title = titleFromMirrorFileName(prov.file); if (title) next.title = title; else delete next.title; } if (byName || !mirror.uploadDate) { const date = dateFromMirrorFileName(prov.file); if (date) next.uploadDate = date; else delete next.uploadDate; } if (JSON.stringify(next) === JSON.stringify(mirror)) return prov; return { ...prov, mirror: next }; } function stripUndefinedDeep(v: T): T { if (Array.isArray(v)) return v.map(stripUndefinedDeep) as T; if (v && typeof v === "object") { const out: Record = {}; for (const [k, x] of Object.entries(v)) if (x !== undefined) out[k] = stripUndefinedDeep(x); return out as T; } return v; } // The sidecar's shape check: a record or null. export function coerceArchiveOrgProvenance(value: unknown): ArchiveOrgProvenance | null { const v = value as Partial | null; if ( !v || typeof v !== "object" || typeof v.identifier !== "string" || typeof v.itemUrl !== "string" || typeof v.fetchedAt !== "string" || !v.item || typeof v.item !== "object" ) { return null; } const item = v.item as ArchiveOrgProvenance["item"]; const mirror = v.mirror; return { ...(v as ArchiveOrgProvenance), item: { ...item, collections: Array.isArray(item.collections) ? item.collections : [] }, ...(mirror && (typeof mirror.id !== "string" || typeof mirror.url !== "string") ? { mirror: undefined } : {}), }; } // ─── What the record's metadata.info.json is corrected to ─── // yt-dlp describes an entry of a multi-file item with the ITEM's title and // page, and a mirror with archive.org's upload date. The record says: // // webpage_url the page of what was imported (the file's page inside the // item). NOT cosmetic: the snapshot renames every video dir to // `extractVideoId(webpage_url)` (reconcileVideoDirs.ts), so an // entry left with the item's page would be merged into the item. // title the original's title (a mirror), else the file's own title, // else the title its file name carries (date, view count, id // and extension off; titleFromMirrorFileName), else the name — // never the item's for one file of many. // upload_date the original's date (a mirror: its info.json's, else the // date its file name starts with); for any other file of an // item, the date its name or folder starts with — with the // timestamp dropped either way // description the original's (a mirror), when it had one // uploader the original's uploader, else the item's public credit — never // the uploading account's e-mail address // // Returns only the keys that differ from `info`; empty when nothing does. export function archiveOrgMetadataPatch( prov: ArchiveOrgProvenance, info: Record, ): Record { const want: Record = {}; want.webpage_url = prov.fileUrl ?? prov.itemUrl; const mirror = prov.mirror; const fileName = prov.file ? (prov.file.split("/").pop() ?? prov.file).replace(/\.[A-Za-z0-9]{1,8}$/, "") : undefined; const title = mirror?.title ?? (prov.file ? (prov.fileTitle ?? titleFromMirrorFileName(prov.file) ?? fileName) : undefined); if (title) want.title = title; const fileDate = prov.file ? dateFromMirrorFileName(prov.file) : null; if (mirror?.uploadDate) want.upload_date = mirror.uploadDate; else if (fileDate) want.upload_date = fileDate; if (mirror?.description) want.description = mirror.description; const uploader = mirror?.uploader ?? prov.item.creator; const current = info.uploader; if (uploader) want.uploader = uploader; else if (typeof current === "string" && current.includes("@")) want.uploader = null; const patch: Record = {}; for (const [k, v] of Object.entries(want)) { if (JSON.stringify(info[k]) !== JSON.stringify(v)) patch[k] = v; } // A mirror's or a file's own date replaces archive.org's; the timestamp yt-dlp derived from // the item's publicdate would contradict it. if (patch.upload_date !== undefined && typeof info.timestamp === "number") patch.timestamp = null; return patch; } // ─── The links a record derives ─── export type ExternalLink = { label: string; url: string }; // What a citation of an archive.org record links, beside its moment page: // // original where the recording was first published — the YouTube upload // at the cited second for a mirror, else the archive.org page // downloads where a reader can fetch the file to check it: the archive.org // page (for a mirror, since `original` is YouTube there) and the // item's torrent // // Derived from the record's `webpage_url` alone when there is no provenance // (the torrent name is archive.org's fixed `_archive.torrent`); // the provenance adds the mirror. export function archiveOrgCitationLinks(opts: { webpageUrl: string | null | undefined; provenance?: ArchiveOrgProvenance | null; seconds?: number; }): { original?: ExternalLink; downloads: ExternalLink[] } { const prov = opts.provenance ?? null; const ref = opts.webpageUrl ? parseArchiveOrgUrl(opts.webpageUrl) : null; const identifier = prov?.identifier ?? ref?.identifier; if (!identifier) return { downloads: [] }; const page = prov?.fileUrl ?? prov?.itemUrl ?? archiveOrgDetailsUrl(ref!); const torrent: ExternalLink = { label: "torrent", url: prov ? (prov.torrentUrl ?? archiveOrgTorrentUrl(identifier)) : archiveOrgTorrentUrl(identifier), }; const archive: ExternalLink = { label: "archive.org", url: page }; const mirror = prov?.mirror; if (mirror) { const secs = Math.max(0, Math.floor(opts.seconds ?? 0)); const u = new URL(mirror.url); if (secs > 0) u.searchParams.set("t", `${secs}s`); return { original: { label: "YouTube", url: u.toString() }, downloads: [archive, torrent] }; } return { original: archive, downloads: [torrent] }; } // ─── Playing it ─── // Browser-playable containers, best first. archive.org transcodes most video // originals to an h.264 mp4 derivative, which every browser plays — an .avi or // .mpeg original does not. const PLAYABLE_EXTS = ["mp4", "m4v", "webm", "m4a", "mp3", "ogg", "oga", "opus"]; type FormatLike = { url?: unknown; ext?: unknown; format_note?: unknown }; // The one file of a record a