Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 9d8e7303ace208fdf95a13d21810268235868d0e
parent 9b649912101756b6a746db374b3b40af5966564d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 08:51:35 -0400

wayback: a capture is named by what it is a copy of, and says so

extractVideoId unwraps a Wayback Machine capture (web.archive.org/web/<ts>[mod_]/<original>)
into its original: an archived YouTube page is its YouTube id, a JW Player file its media id
(cdn.jwplayer.com, content.jwplatform.com, videos-fms.jwpsrv.com) — not `watch` or
`<id>-<rendition>.mp4`. summarize names such a record by the same id. A capture URL is its
own moment link (no time param). Every managed download of a capture writes the
`wayback.json` sidecar: original URL, capture timestamp, the page that plays, the raw bytes.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/lib/momentUrl.ts | 4++++
Mcommon/lib/sidecar-server.test.ts | 4+++-
Mcommon/lib/transcripts-server.ts | 14+++++++++++---
Mcommon/lib/videoId.ts | 20++++++++++++++++++++
Acommon/lib/wayback-server.ts | 43+++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/wayback.test.ts | 162+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/wayback.ts | 211+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/downloadOneManaged.ts | 31+++++++++++++++++++++++++++++--
8 files changed, 483 insertions(+), 6 deletions(-)

diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts @@ -19,6 +19,7 @@ // report-citation UI, and build tools alike. import { detectPlatform, type Platform } from "./platform"; +import { parseWaybackUrl } from "./wayback"; export type MomentUrlInput = { // Public origin of the archilyzer viewer that owns this video (a RemoteSource @@ -55,6 +56,9 @@ export function platformMomentUrl( if (!webpageUrl) return null; const secs = Math.max(0, Math.floor(seconds || 0)); if (secs <= 0) return webpageUrl; + // A Wayback capture (lib/wayback.ts) is the page that plays, and a time + // param would name a different URL — one the Wayback Machine never captured. + if (parseWaybackUrl(webpageUrl)) return webpageUrl; const plat = platform ?? detectPlatform(webpageUrl); let u: URL; try { diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts @@ -11,6 +11,7 @@ import "./attribution-server"; import "./diarization-server"; import "./digest-server"; import "./metadataHistory-server"; +import "./wayback-server"; import { availabilitySidecar, loadAvailability, @@ -42,7 +43,7 @@ async function scratch(): Promise<string> { return mkdtemp(path.join(os.tmpdir(), "sidecar-")); } -test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => { +test("every declared sidecar filename escapes SUB_FILE_RE, and all twelve are declared", () => { assert.deepEqual([...SIDECAR_FILENAMES].sort(), [ "ai-digest.json", "ai-digest.overrides.json", @@ -55,6 +56,7 @@ test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are de "exclude-truncated-check.json", "metadata.history.json", "transcribe-outcome.json", + "wayback.json", ]); for (const name of SIDECAR_FILENAMES) { assert.ok(!SUB_FILE_RE.test(name), name); diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -5,6 +5,8 @@ import { defaultWebpageUrl, detectPlatform } from "./platform"; import { archiveOrgPlayableUrl } from "./archiveOrg"; import { bitchutePlayableUrl } from "./bitchute"; import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId"; +import { parseWaybackUrl } from "./wayback"; +import { extractVideoId } from "./videoId"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { MediaType, VideoStat, VideoStatus } from "./stats"; import type { VideoState } from "./availability"; @@ -77,7 +79,8 @@ export function loadRawMetadataFromDir( // The platform a record is from, by yt-dlp's extractor. archive.org's // extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's -// `YoutubeWebArchive`, which is a YouTube video. +// `YoutubeWebArchive`, which is a YouTube video: the original's platform (the +// record's `wayback.json` says it is a copy, lib/wayback-server.ts). // // AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the // app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has @@ -144,12 +147,17 @@ export function summarize( const platform = platformFromMetadata(meta); // archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`, // which is not a slug; the canonical id (lib/archiveOrgId.ts) is. + // A Wayback capture (lib/wayback.ts): the id its dir is named by — what + // the capture is of (lib/videoId.ts) — not yt-dlp's, which for a raw media + // file is the file's name (`<jwId>-<rendition>`). + const waybackId = parseWaybackUrl(meta.webpage_url) ? extractVideoId(meta.webpage_url!) : null; const id = - platform === "odysee" + waybackId ?? + (platform === "odysee" ? (meta.webpage_url_basename ?? meta.id ?? videoDir) : platform === "archiveorg" ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir) - : (meta.id ?? videoDir); + : (meta.id ?? videoDir)); const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1]; return { slug: `${channelSlug}/${id}`, diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts @@ -10,11 +10,31 @@ // for the native-id resolution that reads metadata.info.json. import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId"; +import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback"; export function extractVideoId(url: string): string | null { + return extractVideoIdAt(url, 0); +} + +function extractVideoIdAt(url: string, depth: number): string | null { try { const u = new URL(url); const host = u.hostname.toLowerCase(); + if (isWaybackHost(host)) { + // A Wayback capture is named by what it is a capture OF + // (lib/wayback.ts): an archived YouTube page by its YouTube id, a JW + // Player file by its media id. Its own path's last segment is the + // original's (`watch`, `<id>-<rendition>.mp4`) — a name two captures + // share. A capture of a capture is not unwrapped twice. + const ref = depth === 0 ? parseWaybackUrl(url) : null; + return ref ? extractVideoIdAt(ref.originalUrl, depth + 1) : null; + } + if (isJwPlayerHost(host)) { + // One media id across every rendition and host; a JW URL that names + // none falls through to the last segment below. + const jw = jwPlayerMediaId(u); + if (jw) return jw; + } if (isArchiveOrgItemHost(host)) { // A whole item → its identifier; one file inside an item → a stable // `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names diff --git a/common/lib/wayback-server.ts b/common/lib/wayback-server.ts @@ -0,0 +1,43 @@ +// WAYBACK PROVENANCE ON DISK — the `wayback.json` sidecar. +// +// A record downloaded from a Wayback Machine capture (lib/wayback.ts) carries +// what it is a copy of: the original URL, the capture's timestamp, the capture +// as a page that plays and as its raw bytes. Built from the capture URL alone +// — no request — so it is written on every download of one +// (ytdlp/downloadOneManaged.ts) and by `archilyzer wayback refresh` for a +// record imported before this existed (controller/waybackRefresh.ts). + +import { + WAYBACK_PROVENANCE_FILENAME, + buildWaybackProvenance, + coerceWaybackProvenance, + sameWaybackProvenance, + type WaybackProvenance, +} from "./wayback"; +import { sidecar, sidecarField } from "./sidecar-server"; + +export const waybackProvenanceSidecar = sidecar( + WAYBACK_PROVENANCE_FILENAME, + sidecarField(coerceWaybackProvenance), +); + +export const { load: loadWaybackProvenance, write: writeWaybackProvenance } = + waybackProvenanceSidecar; + +// The sidecar for a record fetched by `url`: written when the URL is a +// capture and the one on disk is absent or says otherwise. Returns what the +// record now carries (null for a URL that is not a capture) and whether it was +// (or, with `dryRun`, would be) written. +export async function ensureWaybackProvenance( + videoDir: string, + url: string, + opts: { dryRun?: boolean; onLog?: (line: string) => void } = {}, +): Promise<{ provenance: WaybackProvenance | null; written: boolean }> { + const next = buildWaybackProvenance(url); + if (!next) return { provenance: null, written: false }; + const prev = await loadWaybackProvenance(videoDir); + if (prev && sameWaybackProvenance(prev, next)) return { provenance: prev, written: false }; + if (!opts.dryRun) await writeWaybackProvenance(videoDir, next); + opts.onLog?.(`Wayback provenance: a capture of ${next.originalUrl} (${next.captureTs}).\n`); + return { provenance: next, written: true }; +} diff --git a/common/lib/wayback.test.ts b/common/lib/wayback.test.ts @@ -0,0 +1,162 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + buildWaybackProvenance, + coerceWaybackProvenance, + jwPlayerMediaId, + parseWaybackUrl, + waybackCaptureDate, + waybackCitationLinks, +} from "./wayback"; +import { ensureWaybackProvenance, loadWaybackProvenance } from "./wayback-server"; +import { extractVideoId } from "./videoId"; +import { platformMomentUrl } from "./momentUrl"; +import { platformFromMetadata, summarize } from "./transcripts-server"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/wayback.test.ts +// +// Every id, account and token here is invented. + +const YT = "Abc123def45"; +const JW = "Qw3rTy12"; +const ARCHIVED_PAGE = `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`; +const JW_FILE = `https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${JW}-12345678.mp4?token=0_abc_0xdef`; +const ARCHIVED_FILE = `https://web.archive.org/web/20200102030405id_/${JW_FILE}`; + +test("parseWaybackUrl: timestamp, modifier and the original with its own query", () => { + assert.deepEqual(parseWaybackUrl(ARCHIVED_PAGE), { + captureTs: "20210102030405", + modifier: "", + originalUrl: `https://www.youtube.com/watch?v=${YT}`, + }); + assert.deepEqual(parseWaybackUrl(ARCHIVED_FILE), { + captureTs: "20200102030405", + modifier: "id_", + originalUrl: JW_FILE, + }); + // A short timestamp, another modifier, no scheme, a collapsed scheme, an + // encoded original, the older path without /web/. + assert.equal(parseWaybackUrl("https://web.archive.org/web/2019im_/example.com/a.png")?.originalUrl, "http://example.com/a.png"); + assert.equal(parseWaybackUrl("https://web.archive.org/web/2019/https:/example.com/x")?.originalUrl, "https://example.com/x"); + assert.equal( + parseWaybackUrl("https://web.archive.org/web/2019/https%3A%2F%2Fexample.com%2Fx")?.originalUrl, + "https://example.com/x", + ); + assert.equal(parseWaybackUrl("http://wayback.archive.org/20190101000000/http://example.com/")?.captureTs, "20190101000000"); + // Not captures. + assert.equal(parseWaybackUrl("https://web.archive.org/web/*/example.com"), null); + assert.equal(parseWaybackUrl("https://archive.org/details/some-item"), null); + assert.equal(parseWaybackUrl(`https://www.youtube.com/watch?v=${YT}`), null); + assert.equal(parseWaybackUrl("not a url"), null); +}); + +test("jwPlayerMediaId: the media id across hosts and renditions", () => { + assert.equal(jwPlayerMediaId(JW_FILE), JW); + assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/videos/${JW}-AbCdEf12.mp4`), JW); + assert.equal(jwPlayerMediaId(`https://content.jwplatform.com/videos/${JW}.mp4`), JW); + assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/manifests/${JW}.m3u8`), JW); + assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/v2/media/${JW}`), JW); + assert.equal(jwPlayerMediaId("https://cdn.jwplayer.com/libraries/AbCd1234.js"), null); + assert.equal(jwPlayerMediaId(`https://example.com/videos/${JW}-1.mp4`), null); +}); + +test("extractVideoId unwraps a capture into what it is a capture of", () => { + assert.equal(extractVideoId(ARCHIVED_PAGE), YT); + assert.equal(extractVideoId(`https://web.archive.org/web/2021if_/https://youtu.be/${YT}`), YT); + assert.equal(extractVideoId(ARCHIVED_FILE), JW); + assert.equal(extractVideoId(JW_FILE), JW); + // Unwrapped once: a capture of a capture names no id. + assert.equal(extractVideoId(`https://web.archive.org/web/2021/${ARCHIVED_PAGE}`), null); + // A Wayback page that is not a capture names no record. + assert.equal(extractVideoId("https://web.archive.org/web/*/example.com"), null); + // A capture of a page the app does not know: the original's last segment. + assert.equal(extractVideoId("https://web.archive.org/web/2021/https://example.com/media/clip-7.mp4"), "clip-7.mp4"); + // Nothing else moved. + assert.equal(extractVideoId(`https://www.youtube.com/watch?v=${YT}`), YT); + assert.equal(extractVideoId("https://example.com/feed/episode-1.mp3"), "episode-1.mp3"); +}); + +test("buildWaybackProvenance: an archived YouTube page's original is its watch URL", () => { + assert.deepEqual(buildWaybackProvenance(`https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`), { + originalUrl: `https://www.youtube.com/watch?v=${YT}`, + captureTs: "20210102", + waybackUrl: `https://web.archive.org/web/20210102/https://m.youtube.com/watch?v=${YT}&feature=share`, + rawUrl: `https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`, + }); + const file = buildWaybackProvenance(ARCHIVED_FILE)!; + assert.equal(file.originalUrl, JW_FILE); + assert.equal(file.waybackUrl, `https://web.archive.org/web/20200102030405/${JW_FILE}`); + assert.equal(file.rawUrl, ARCHIVED_FILE); + assert.equal(buildWaybackProvenance(JW_FILE), null); + assert.deepEqual(coerceWaybackProvenance(JSON.parse(JSON.stringify(file))), file); + assert.equal(coerceWaybackProvenance({ ...file, captureTs: "yesterday" }), null); + assert.equal(coerceWaybackProvenance([]), null); +}); + +test("waybackCaptureDate and the citation links", () => { + assert.equal(waybackCaptureDate("20210102030405"), "2021-01-02"); + assert.equal(waybackCaptureDate("2021"), "2021"); + assert.equal(waybackCaptureDate("202101"), "2021-01"); + const prov = buildWaybackProvenance(ARCHIVED_PAGE)!; + const links = waybackCitationLinks(prov, { + originalMomentUrl: platformMomentUrl(prov.originalUrl, null, 90), + }); + assert.deepEqual(links.original, { + label: "Original (may be gone)", + url: `https://www.youtube.com/watch?v=${YT}&t=90s`, + }); + assert.deepEqual(links.copy, { + label: "Wayback Machine copy, 2021-01-02", + url: `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`, + }); +}); + +test("platformMomentUrl: a capture is the page that plays, never given a time param", () => { + assert.equal(platformMomentUrl(ARCHIVED_PAGE, "youtube", 125), ARCHIVED_PAGE); + assert.equal(platformMomentUrl(`https://www.youtube.com/watch?v=${YT}`, "youtube", 125), `https://www.youtube.com/watch?v=${YT}&t=125s`); +}); + +test("platformFromMetadata: YoutubeWebArchive is the original's platform, YouTube", () => { + assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive", extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube"); + assert.equal(platformFromMetadata({ extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube"); +}); + +test("summarize: a capture's id is its dir's name, not yt-dlp's file-name id", () => { + const s = summarize("demo", `${JW}`, { + id: `${JW}-12345678`, + title: `${JW}-12345678`, + extractor_key: "Generic", + webpage_url: ARCHIVED_FILE, + }); + assert.equal(s.id, JW); + assert.equal(s.slug, `demo/${JW}`); + const page = summarize("demo", YT, { id: YT, title: "A title", extractor_key: "YoutubeWebArchive", webpage_url: ARCHIVED_PAGE }); + assert.equal(page.id, YT); + assert.equal(page.platform, "youtube"); + assert.equal(page.webpageUrl, ARCHIVED_PAGE); +}); + +test("ensureWaybackProvenance writes the sidecar once, and nothing for a non-capture", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "wayback-sidecar-")); + assert.deepEqual(await ensureWaybackProvenance(dir, JW_FILE), { provenance: null, written: false }); + assert.equal(await loadWaybackProvenance(dir), null); + + const dry = await ensureWaybackProvenance(dir, ARCHIVED_PAGE, { dryRun: true }); + assert.equal(dry.written, true); + assert.equal(await loadWaybackProvenance(dir), null); + + const first = await ensureWaybackProvenance(dir, ARCHIVED_PAGE); + assert.equal(first.written, true); + const onDisk = JSON.parse(await readFile(path.join(dir, "wayback.json"), "utf8")); + assert.deepEqual(onDisk, buildWaybackProvenance(ARCHIVED_PAGE)); + assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, false); + + // A sidecar for another capture is rewritten. + await writeFile(path.join(dir, "wayback.json"), JSON.stringify(buildWaybackProvenance(ARCHIVED_FILE))); + assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, true); + assert.deepEqual(await loadWaybackProvenance(dir), buildWaybackProvenance(ARCHIVED_PAGE)); +}); diff --git a/common/lib/wayback.ts b/common/lib/wayback.ts @@ -0,0 +1,211 @@ +// THE WAYBACK MACHINE — a capture of some other URL, and what it was a copy of. +// +// A Wayback capture URL wraps the original: `https://web.archive.org/web/ +// <timestamp>[<modifier>]/<original>`, where the timestamp is 4–14 digits +// (yyyy[MM[dd[hh[mm[ss]]]]]) and the modifier is the replay mode — none (the +// page in the Wayback frame, which is what plays), `id_` (the raw bytes as +// captured), `im_`, `if_`, `js_`, `cs_`, `oe_`, … The original keeps its own +// query string, so it is the rest of the path PLUS the capture URL's search. +// +// yt-dlp downloads either kind through the editor's import: an archived +// YouTube page through its `YoutubeWebArchive` extractor (the record's id is +// the YouTube id), a raw media file through `generic` (an id that is the file +// name). Neither knows it is a copy; the `wayback.json` sidecar +// (lib/wayback-server.ts) records that, and lib/videoId.ts names the record by +// what the capture is OF. +// +// A LEAF: no imports, so lib/videoId.ts (itself a leaf apart from +// archiveOrgId.ts) can unwrap a capture without a cycle. + +export const WAYBACK_PROVENANCE_FILENAME = "wayback.json"; + +// Hosts that serve Wayback captures. archive.org's own hosts serve items +// (lib/archiveOrgId.ts), never captures. +const WAYBACK_HOSTS = new Set(["web.archive.org", "wayback.archive.org"]); + +export function isWaybackHost(host: string): boolean { + return WAYBACK_HOSTS.has(host.toLowerCase()); +} + +export type WaybackRef = { + // The capture's timestamp as the URL gives it (4–14 digits). + captureTs: string; + // The replay modifier (`id_`, `im_`, …), "" for the framed page. + modifier: string; + // The URL the capture is of, as the capture URL names it (scheme added + // when the capture URL left it off). + originalUrl: string; +}; + +// `/web/<ts><mod>/<original>`, or the older `/<ts><mod>/<original>`. +const CAPTURE_PATH_RE = /^\/(?:web\/)?(\d{4,14})([a-z]{2}_)?\/(.+)$/s; + +// A Wayback capture URL's parts, or null for anything else (a calendar page +// `/web/*/<url>`, a search, another host). +export function parseWaybackUrl(url: string | null | undefined): WaybackRef | null { + if (!url) return null; + let u: URL; + try { + u = new URL(url); + } catch { + return null; + } + if (!isWaybackHost(u.hostname)) return null; + const m = CAPTURE_PATH_RE.exec(u.pathname); + if (!m) return null; + let rest = m[3]; + // A percent-encoded original (`https%3A%2F%2F…`) is the same URL. + if (/^https?%3a/i.test(rest)) { + try { + rest = decodeURIComponent(rest); + } catch { + return null; + } + } + // The WHATWG parser keeps `https://` inside a path as written, but a capture + // URL that went through a path normaliser arrives as `https:/host`. + rest = rest.replace(/^(https?):\/(?!\/)/i, "$1://"); + if (!/^https?:\/\//i.test(rest)) rest = `http://${rest}`; + const original = `${rest}${u.search}${u.hash}`; + try { + new URL(original); + } catch { + return null; + } + return { captureTs: m[1], modifier: m[2] ?? "", originalUrl: original }; +} + +// The capture as a page that plays (the Wayback frame, no modifier). +export function waybackPageUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string { + return `https://web.archive.org/web/${ref.captureTs}/${ref.originalUrl}`; +} + +// The capture's raw bytes (`id_`), the form yt-dlp fetches a media file by. +export function waybackRawUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string { + return `https://web.archive.org/web/${ref.captureTs}id_/${ref.originalUrl}`; +} + +// `YYYY-MM-DD` of a capture timestamp, or the year (/month) when that is all +// it names. +export function waybackCaptureDate(captureTs: string): string { + const y = captureTs.slice(0, 4); + const mo = captureTs.slice(4, 6); + const d = captureTs.slice(6, 8); + return [y, mo, d].filter((p) => p.length === 2 || p.length === 4).join("-"); +} + +// ─── JW Player ─── + +// The hosts JW Player serves a media file from. A file there is +// `…/videos/<mediaId>-<rendition>.<ext>` (cdn.jwplayer.com/videos/…, +// content.jwplatform.com/videos/…, videos-fms.jwpsrv.com/content/conversions/ +// <account>/videos/…); the media id is eight alphanumerics and names the video +// across every rendition. +const JW_HOSTS = ["jwplayer.com", "jwplatform.com", "jwpsrv.com"]; + +export function isJwPlayerHost(host: string): boolean { + const h = host.toLowerCase(); + return JW_HOSTS.some((d) => h === d || h.endsWith(`.${d}`)); +} + +const JW_FILE_RE = /\/videos\/([A-Za-z0-9]{8})(?:-[A-Za-z0-9]+)?\.[A-Za-z0-9]+$/; +const JW_MEDIA_RE = /\/(?:manifests|v2\/media|previews)\/([A-Za-z0-9]{8})(?:[-./]|$)/; + +// The JW media id a JW Player file or manifest URL names, or null. +export function jwPlayerMediaId(url: string | URL): string | null { + let u: URL; + try { + u = typeof url === "string" ? new URL(url) : url; + } catch { + return null; + } + if (!isJwPlayerHost(u.hostname)) return null; + const m = JW_FILE_RE.exec(u.pathname) ?? JW_MEDIA_RE.exec(u.pathname); + return m ? m[1] : null; +} + +// ─── The sidecar's record ─── + +export type WaybackProvenance = { + // What the capture is a copy of. For an archived YouTube page, its watch + // URL (`https://www.youtube.com/watch?v=<id>`), whatever form was captured. + originalUrl: string; + // The capture's timestamp (4–14 digits). + captureTs: string; + // The capture as a page that plays. + waybackUrl: string; + // The capture's raw bytes. + rawUrl: string; +}; + +// A YouTube URL as its watch page, or null for any other URL. +function youtubeWatchUrl(url: string): string | null { + let u: URL; + try { + u = new URL(url); + } catch { + return null; + } + const host = u.hostname.toLowerCase(); + let id: string | null = null; + if (host === "youtu.be") id = u.pathname.split("/").filter(Boolean)[0] ?? null; + else if (host === "youtube.com" || host.endsWith(".youtube.com")) { + id = u.searchParams.get("v"); + if (!id) { + const segs = u.pathname.split("/").filter(Boolean); + if ((segs[0] === "embed" || segs[0] === "shorts" || segs[0] === "v" || segs[0] === "live") && segs[1]) id = segs[1]; + } + } else return null; + return id ? `https://www.youtube.com/watch?v=${id}` : null; +} + +// The sidecar for a capture URL, or null when the URL is not one. +export function buildWaybackProvenance(url: string): WaybackProvenance | null { + const ref = parseWaybackUrl(url); + if (!ref) return null; + return { + originalUrl: youtubeWatchUrl(ref.originalUrl) ?? ref.originalUrl, + captureTs: ref.captureTs, + waybackUrl: waybackPageUrl(ref), + rawUrl: waybackRawUrl(ref), + }; +} + +export function coerceWaybackProvenance(value: unknown): WaybackProvenance | null { + if (!value || typeof value !== "object" || Array.isArray(value)) return null; + const v = value as Record<string, unknown>; + const str = (k: string) => (typeof v[k] === "string" && v[k] ? (v[k] as string) : null); + const originalUrl = str("originalUrl"); + const captureTs = str("captureTs"); + const waybackUrl = str("waybackUrl"); + const rawUrl = str("rawUrl"); + if (!originalUrl || !captureTs || !/^\d{4,14}$/.test(captureTs) || !waybackUrl || !rawUrl) return null; + return { originalUrl, captureTs, waybackUrl, rawUrl }; +} + +export function sameWaybackProvenance(a: WaybackProvenance, b: WaybackProvenance): boolean { + return ( + a.originalUrl === b.originalUrl && + a.captureTs === b.captureTs && + a.waybackUrl === b.waybackUrl && + a.rawUrl === b.rawUrl + ); +} + +// ─── Citing it ─── + +export type WaybackLink = { label: string; url: string }; + +// What a citation of an archived copy links: the original, named as the +// original and as possibly gone (a capture exists because it may be), and the +// Wayback copy, which plays. `originalMomentUrl` is the original at the cited +// second when its platform takes one (lib/momentUrl.ts). +export function waybackCitationLinks( + prov: WaybackProvenance, + opts: { originalMomentUrl?: string | null } = {}, +): { original: WaybackLink; copy: WaybackLink } { + return { + original: { label: "Original (may be gone)", url: opts.originalMomentUrl || prov.originalUrl }, + copy: { label: `Wayback Machine copy, ${waybackCaptureDate(prov.captureTs)}`, url: prov.waybackUrl }, + }; +} diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -1,7 +1,7 @@ import { removeMediaFile } from "../lib/mediaTier-server"; import { tierVideoDir } from "../lib/mediaTier-server"; import path from "node:path"; -import { appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises"; +import { access, appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises"; import { createWriteStream, type Dirent, type WriteStream } from "node:fs"; import { execa } from "execa"; import { @@ -50,6 +50,8 @@ import { writeDownloadOutcome } from "../lib/downloadOutcome-server"; import { formatBytes } from "../lib/format"; import { recordAvailability } from "../lib/availability-server"; import { withMetadataHistory } from "../lib/metadataHistory-server"; +import { ensureWaybackProvenance } from "../lib/wayback-server"; +import { parseWaybackUrl } from "../lib/wayback"; import { loadRawMetadata, loadRawMetadataFromDir, @@ -656,12 +658,37 @@ export async function downloadOneManaged( if (detectPlatform(opts.videoUrl) === "archiveorg") { return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps); } - return await runManagedDownload(opts, channelDir, startedAt, canonicalId); + const outcome = await runManagedDownload(opts, channelDir, startedAt, canonicalId); + await recordWaybackProvenance(opts, channelDir, canonicalId); + return outcome; } finally { logStream?.end(); } } +// A WAYBACK CAPTURE IS A COPY (lib/wayback.ts): once the record exists — its +// metadata.info.json written by the prefetch or the download — the +// `wayback.json` sidecar says of what. Built from the URL alone, so it costs no +// request; never fails the download. +async function recordWaybackProvenance( + opts: ManagedDownloadOpts, + channelDir: string, + canonicalId: string | null, +): Promise<void> { + if (!canonicalId || !parseWaybackUrl(opts.videoUrl)) return; + const videoDir = path.join(channelDir, "data", canonicalId); + try { + await access(path.join(videoDir, "metadata.info.json")); + } catch { + return; + } + try { + await ensureWaybackProvenance(videoDir, opts.videoUrl, { onLog: opts.onLog }); + } catch (err) { + opts.onLog(`Wayback provenance not written: ${(err as Error).message}\n`); + } +} + async function runManagedDownload( opts: ManagedDownloadOpts, channelDir: string,