// THE WAYBACK MACHINE — a capture of some other URL, and what it was a copy of. // // A Wayback capture URL wraps the original: `https://web.archive.org/web/ // []/`, where the timestamp is 4–14 digits // (yyyy[MM[dd[hh[mm[ss]]]]]) and the modifier is the replay mode — none (the // page in the Wayback frame, which is what plays), `id_` (the raw bytes as // captured), `im_`, `if_`, `js_`, `cs_`, `oe_`, … The original keeps its own // query string, so it is the rest of the path PLUS the capture URL's search. // // yt-dlp downloads either kind through the editor's import: an archived // YouTube page through its `YoutubeWebArchive` extractor (the record's id is // the YouTube id), a raw media file through `generic` (an id that is the file // name). Neither knows it is a copy; the `wayback.json` sidecar // (lib/wayback-server.ts) records that, and lib/videoId.ts names the record by // what the capture is OF. // // A LEAF: no imports, so lib/videoId.ts (itself a leaf apart from // archiveOrgId.ts) can unwrap a capture without a cycle. export const WAYBACK_PROVENANCE_FILENAME = "wayback.json"; // Hosts that serve Wayback captures. archive.org's own hosts serve items // (lib/archiveOrgId.ts), never captures. const WAYBACK_HOSTS = new Set(["web.archive.org", "wayback.archive.org"]); export function isWaybackHost(host: string): boolean { return WAYBACK_HOSTS.has(host.toLowerCase()); } export type WaybackRef = { // The capture's timestamp as the URL gives it (4–14 digits). captureTs: string; // The replay modifier (`id_`, `im_`, …), "" for the framed page. modifier: string; // The URL the capture is of, as the capture URL names it (scheme added // when the capture URL left it off). originalUrl: string; }; // `/web//`, or the older `//`. const CAPTURE_PATH_RE = /^\/(?:web\/)?(\d{4,14})([a-z]{2}_)?\/(.+)$/s; // A Wayback capture URL's parts, or null for anything else (a calendar page // `/web/*/`, a search, another host). export function parseWaybackUrl(url: string | null | undefined): WaybackRef | null { if (!url) return null; let u: URL; try { u = new URL(url); } catch { return null; } if (!isWaybackHost(u.hostname)) return null; const m = CAPTURE_PATH_RE.exec(u.pathname); if (!m) return null; let rest = m[3]; // A percent-encoded original (`https%3A%2F%2F…`) is the same URL. if (/^https?%3a/i.test(rest)) { try { rest = decodeURIComponent(rest); } catch { return null; } } // The WHATWG parser keeps `https://` inside a path as written, but a capture // URL that went through a path normaliser arrives as `https:/host`. rest = rest.replace(/^(https?):\/(?!\/)/i, "$1://"); if (!/^https?:\/\//i.test(rest)) rest = `http://${rest}`; const original = `${rest}${u.search}${u.hash}`; try { new URL(original); } catch { return null; } return { captureTs: m[1], modifier: m[2] ?? "", originalUrl: original }; } // The capture as a page that plays (the Wayback frame, no modifier). export function waybackPageUrl(ref: Pick): string { return `https://web.archive.org/web/${ref.captureTs}/${ref.originalUrl}`; } // The capture's raw bytes (`id_`), the form yt-dlp fetches a media file by. export function waybackRawUrl(ref: Pick): string { return `https://web.archive.org/web/${ref.captureTs}id_/${ref.originalUrl}`; } // `YYYY-MM-DD` of a capture timestamp, or the year (/month) when that is all // it names. export function waybackCaptureDate(captureTs: string): string { const y = captureTs.slice(0, 4); const mo = captureTs.slice(4, 6); const d = captureTs.slice(6, 8); return [y, mo, d].filter((p) => p.length === 2 || p.length === 4).join("-"); } // ─── JW Player ─── // The hosts JW Player serves a media file from. A file there is // `…/videos/-.` (cdn.jwplayer.com/videos/…, // content.jwplatform.com/videos/…, videos-fms.jwpsrv.com/content/conversions/ // /videos/…); the media id is eight alphanumerics and names the video // across every rendition. const JW_HOSTS = ["jwplayer.com", "jwplatform.com", "jwpsrv.com"]; export function isJwPlayerHost(host: string): boolean { const h = host.toLowerCase(); return JW_HOSTS.some((d) => h === d || h.endsWith(`.${d}`)); } const JW_FILE_RE = /\/videos\/([A-Za-z0-9]{8})(?:-[A-Za-z0-9]+)?\.[A-Za-z0-9]+$/; const JW_MEDIA_RE = /\/(?:manifests|v2\/media|previews)\/([A-Za-z0-9]{8})(?:[-./]|$)/; // The JW media id a JW Player file or manifest URL names, or null. export function jwPlayerMediaId(url: string | URL): string | null { let u: URL; try { u = typeof url === "string" ? new URL(url) : url; } catch { return null; } if (!isJwPlayerHost(u.hostname)) return null; const m = JW_FILE_RE.exec(u.pathname) ?? JW_MEDIA_RE.exec(u.pathname); return m ? m[1] : null; } // ─── The sidecar's record ─── export type WaybackProvenance = { // What the capture is a copy of. For an archived YouTube page, its watch // URL (`https://www.youtube.com/watch?v=`), whatever form was captured. originalUrl: string; // The capture's timestamp (4–14 digits). captureTs: string; // The capture as a page that plays. waybackUrl: string; // The capture's raw bytes. rawUrl: string; }; // A YouTube URL as its watch page, or null for any other URL. function youtubeWatchUrl(url: string): string | null { let u: URL; try { u = new URL(url); } catch { return null; } const host = u.hostname.toLowerCase(); let id: string | null = null; if (host === "youtu.be") id = u.pathname.split("/").filter(Boolean)[0] ?? null; else if (host === "youtube.com" || host.endsWith(".youtube.com")) { id = u.searchParams.get("v"); if (!id) { const segs = u.pathname.split("/").filter(Boolean); if ((segs[0] === "embed" || segs[0] === "shorts" || segs[0] === "v" || segs[0] === "live") && segs[1]) id = segs[1]; } } else return null; return id ? `https://www.youtube.com/watch?v=${id}` : null; } // The sidecar for a capture URL, or null when the URL is not one. export function buildWaybackProvenance(url: string): WaybackProvenance | null { const ref = parseWaybackUrl(url); if (!ref) return null; return { originalUrl: youtubeWatchUrl(ref.originalUrl) ?? ref.originalUrl, captureTs: ref.captureTs, waybackUrl: waybackPageUrl(ref), rawUrl: waybackRawUrl(ref), }; } export function coerceWaybackProvenance(value: unknown): WaybackProvenance | null { if (!value || typeof value !== "object" || Array.isArray(value)) return null; const v = value as Record; const str = (k: string) => (typeof v[k] === "string" && v[k] ? (v[k] as string) : null); const originalUrl = str("originalUrl"); const captureTs = str("captureTs"); const waybackUrl = str("waybackUrl"); const rawUrl = str("rawUrl"); if (!originalUrl || !captureTs || !/^\d{4,14}$/.test(captureTs) || !waybackUrl || !rawUrl) return null; return { originalUrl, captureTs, waybackUrl, rawUrl }; } export function sameWaybackProvenance(a: WaybackProvenance, b: WaybackProvenance): boolean { return ( a.originalUrl === b.originalUrl && a.captureTs === b.captureTs && a.waybackUrl === b.waybackUrl && a.rawUrl === b.rawUrl ); } // ─── Citing it ─── export type WaybackLink = { label: string; url: string }; // What a citation of an archived copy links: the original, named as the // original and as possibly gone (a capture exists because it may be), and the // Wayback copy, which plays. `originalMomentUrl` is the original at the cited // second when its platform takes one (lib/momentUrl.ts). export function waybackCitationLinks( prov: WaybackProvenance, opts: { originalMomentUrl?: string | null } = {}, ): { original: WaybackLink; copy: WaybackLink } { return { original: { label: "Original (may be gone)", url: opts.originalMomentUrl || prov.originalUrl }, copy: { label: `Wayback Machine copy, ${waybackCaptureDate(prov.captureTs)}`, url: prov.waybackUrl }, }; }