// archive.org ITEM AND FILE IDENTITY — what an archive.org URL names, and the // canonical video id (the `data//` dir name) it maps to. // // A leaf with no imports, like lib/videoId.ts (which calls into it): the roster // store, the browser and the export build all canonicalize URLs through here. // // AN ITEM is archive.org's unit of upload: `https://archive.org/details/`. // An identifier is ASCII letters, digits, `.`, `-` and `_` — already a safe // directory name, so a WHOLE ITEM (one that holds a single media file) has the // identifier itself as its video id. // // A FILE INSIDE AN ITEM (a channel-archive item can hold a hundred and sixty // videos) is `https://archive.org/details//` — the form // yt-dlp's ArchiveOrg extractor resolves to that one entry. A file path is free // text (spaces, brackets, unicode, sub-directories), so its id is // // __-<8 hex> // // where the slug keeps `[A-Za-z0-9_-]` and folds every other run into one `-` // (cut to 48 characters), and the hex is a 32-bit FNV-1a hash of the EXACT file // path. The slug keeps the id readable; the hash keeps it unique — two paths // that slug alike ("a b.mp4", "a_b.mp4", "a b.mkv") still differ. Stable: the // same path always gives the same id, so a re-import lands in the same dir. // // `/embed/` and `/download/` URLs name the same things and canonicalize to the // same ids. `web.archive.org` (the Wayback Machine) is NOT an item host and is // not handled here. export type ArchiveOrgRef = { identifier: string; // The file's path inside the item, decoded; absent for the whole item. file?: string; }; const IDENTIFIER_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/; // Hosts that serve items. Not `web.archive.org` (Wayback) and not the // `iaNNNNNN.us.archive.org` storage nodes a download redirects to. export function isArchiveOrgItemHost(host: string): boolean { const h = host.toLowerCase(); return h === "archive.org" || h === "www.archive.org"; } // yt-dlp reads the path with `unquote_plus`; so do we, so the file we name is // the file it resolves. function unquotePlus(s: string): string { try { return decodeURIComponent(s.replace(/\+/g, " ")); } catch { return s; } } // The item (and file) an archive.org URL names, or null for anything else. export function parseArchiveOrgUrl(url: string): ArchiveOrgRef | null { let u: URL; try { u = new URL(url); } catch { return null; } if (!isArchiveOrgItemHost(u.hostname)) return null; const segs = u.pathname.split("/").filter(Boolean); if (segs.length < 2) return null; const kind = segs[0]; if (kind !== "details" && kind !== "embed" && kind !== "download") return null; const identifier = unquotePlus(segs[1]); if (!IDENTIFIER_RE.test(identifier)) return null; const rest = segs.slice(2).map(unquotePlus); const file = rest.length > 0 ? rest.join("/") : undefined; // A download URL with no file is the item's file listing, i.e. the item. return file ? { identifier, file } : { identifier }; } // FNV-1a, 32 bits, over the UTF-16 code units — dependency-free and the same in // every runtime this module is imported into. function fnv1a32(s: string): string { let h = 0x811c9dc5; for (let i = 0; i < s.length; i++) { h ^= s.charCodeAt(i); h = Math.imul(h, 0x01000193) >>> 0; } return h.toString(16).padStart(8, "0"); } const SLUG_MAX = 48; function fileSlug(file: string): string { const base = file.replace(/\.[A-Za-z0-9]{1,8}$/, ""); const slug = base .replace(/[^A-Za-z0-9_-]+/g, "-") .replace(/-{2,}/g, "-") .replace(/^-+|-+$/g, "") .slice(0, SLUG_MAX) .replace(/-+$/, ""); return slug || "file"; } // The canonical video id of an item, or of one file inside it. export function archiveOrgVideoId(ref: ArchiveOrgRef): string { if (!ref.file) return ref.identifier; return `${ref.identifier}__${fileSlug(ref.file)}-${fnv1a32(ref.file)}`; } // yt-dlp's own id for an archive.org record: the identifier, or // `/` for an entry of a multi-file item. The canonical id // of that record, or null when it is not one. export function archiveOrgVideoIdFromNativeId( nativeId: string | null | undefined, ): string | null { if (!nativeId) return null; const slash = nativeId.indexOf("/"); const identifier = slash < 0 ? nativeId : nativeId.slice(0, slash); if (!IDENTIFIER_RE.test(identifier)) return null; const file = slash < 0 ? "" : nativeId.slice(slash + 1); return archiveOrgVideoId(file ? { identifier, file } : { identifier }); } function encodePath(file: string): string { return file.split("/").map(encodeURIComponent).join("/"); } // The item page, or the file's own page inside it — the URL a record keeps as // its `webpage_url`, and the one every link to it uses. export function archiveOrgDetailsUrl(ref: ArchiveOrgRef): string { const base = `https://archive.org/details/${ref.identifier}`; return ref.file ? `${base}/${encodePath(ref.file)}` : base; } // The file's bytes. export function archiveOrgDownloadUrl(identifier: string, file: string): string { return `https://archive.org/download/${identifier}/${encodePath(file)}`; } // The item's BitTorrent file. archive.org derives one for every item, named // `_archive.torrent`; it covers every file in the item. export function archiveOrgTorrentUrl(identifier: string): string { return `https://archive.org/download/${identifier}/${identifier}_archive.torrent`; } // The item's metadata API (one JSON document: the item's fields and its file // list). export function archiveOrgMetadataUrl(identifier: string): string { return `https://archive.org/metadata/${identifier}`; } // THE YOUTUBE ID A MIRROR CARRIES, when it says so in its name. // // `youtube-` is the identifier tubeup (the usual YouTube → archive.org // mirroring tool) gives an item; a file yt-dlp named carries the id as // `-<id>.<ext>` or `<title> [<id>].<ext>`. The bracketed form is // unambiguous. The dashed one is a guess at the last eleven characters before // the extension, so it is only taken when the token is not plain lowercase // letters — a real id is random base64 and almost never is, while an English // word of eleven letters ("performance") always is. const YT_ID = "[A-Za-z0-9_-]{11}"; export function youtubeIdFromIdentifier(identifier: string): string | null { const m = new RegExp(`^youtube-(${YT_ID})$`).exec(identifier); return m ? m[1] : null; } export function youtubeIdFromFileName(file: string): string | null { const name = file.split("/").pop() ?? file; const stem = name.replace(/(\.[A-Za-z0-9]{1,8})+$/, ""); const bracket = new RegExp(`\\[(${YT_ID})\\]$`).exec(stem); if (bracket) return bracket[1]; const dashed = new RegExp(`(?:^|[-_ ])(${YT_ID})$`).exec(stem); if (dashed && /[A-Z0-9_-]/.test(dashed[1])) return dashed[1]; return null; } // THE TITLE A MIRRORED FILE'S NAME CARRIES, for one file of a multi-file item // that has no title of its own in the item. Archiving tools name a file // `[<collection>_]YYYYMMDD <title> [<n> views]-<id>.<ext>`, and a file-name // sanitiser writes a title's ` | ` as ` _ `. What is taken off: // // - the extension (one), and the YouTube id youtubeIdFromFileName finds, // with its `-`/`_`/space or brackets; // - a trailing ` [<n> views]` count; // - a leading `YYYYMMDD` date, alone or after one `<word>_` prefix, when it // is a calendar date (dateFromMirrorFileName reads it); the raw name stays // in the provenance (`file`); // // and what is put back: ` _ ` → ` | `, and `word_ next` → `word: next` (an // underscore glued to a word and followed by a space is a sanitised colon; an // underscore anywhere else is left as it is). Spaces collapse. Null when no // letter or digit remains. const MIRROR_DATE_PREFIX = /^(?:[A-Za-z0-9]+_)?((?:19|20)\d{2})(\d{2})(\d{2})(?:\s+-\s+|\s+|_|$)/; // The leading `[<word>_]YYYYMMDD` of a file name's stem, when it is a real // calendar day (no 31st of April, no 29th of February outside a leap year): // the date as YYYYMMDD and the stem after it. Null otherwise. function mirrorDatePrefix(stem: string): { date: string; rest: string } | null { const m = MIRROR_DATE_PREFIX.exec(stem); if (!m) return null; const [y, mo, d] = [Number(m[1]), Number(m[2]), Number(m[3])]; const day = new Date(Date.UTC(y, mo - 1, d)); if (day.getUTCFullYear() !== y || day.getUTCMonth() !== mo - 1 || day.getUTCDate() !== d) return null; return { date: `${m[1]}${m[2]}${m[3]}`, rest: stem.slice(m[0].length) }; } function mirrorFileStem(file: string): string { return (file.split("/").pop() ?? file).replace(/\.[A-Za-z0-9]{1,8}$/, ""); } // THE DATE A MIRRORED FILE'S NAME CARRIES: the leading `[<word>_]YYYYMMDD` // archiving tools write (the original's upload day), as YYYYMMDD, only for a // real calendar day. A file whose own name has none takes its folder's (an // archive that keeps one folder per upload: `<YYYYMMDD>_<title>/<title>.mkv`). // Null when neither starts with such a date. export function dateFromMirrorFileName(file: string): string | null { const own = mirrorDatePrefix(mirrorFileStem(file))?.date; if (own) return own; const parts = file.split("/"); const folder = parts.length > 1 ? parts[parts.length - 2] : ""; return folder ? (mirrorDatePrefix(folder)?.date ?? null) : null; } export function titleFromMirrorFileName(file: string): string | null { const name = file.split("/").pop() ?? file; let t = mirrorFileStem(name); const id = youtubeIdFromFileName(name); if (id) { if (t.endsWith(`[${id}]`)) t = t.slice(0, -(id.length + 2)); else if (t.endsWith(id)) t = t.slice(0, -id.length).replace(/[-_ ]$/, ""); } t = t.replace(/\s*\[\d[\d,]*\s+views?\]\s*$/i, ""); t = mirrorDatePrefix(t)?.rest ?? t; t = t .replace(/\s+_\s+/g, " | ") .replace(/(?<=[\p{L}\p{N})\]!?'’"])_(?=\s+\S)/gu, ":") .replace(/\s+/g, " ") .trim(); return /[\p{L}\p{N}]/u.test(t) ? t : null; }