commit 9d8e7303ace208fdf95a13d21810268235868d0e
parent 9b649912101756b6a746db374b3b40af5966564d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 08:51:35 -0400
wayback: a capture is named by what it is a copy of, and says so
extractVideoId unwraps a Wayback Machine capture (web.archive.org/web/<ts>[mod_]/<original>)
into its original: an archived YouTube page is its YouTube id, a JW Player file its media id
(cdn.jwplayer.com, content.jwplatform.com, videos-fms.jwpsrv.com) — not `watch` or
`<id>-<rendition>.mp4`. summarize names such a record by the same id. A capture URL is its
own moment link (no time param). Every managed download of a capture writes the
`wayback.json` sidecar: original URL, capture timestamp, the page that plays, the raw bytes.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 483 insertions(+), 6 deletions(-)
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -19,6 +19,7 @@
// report-citation UI, and build tools alike.
import { detectPlatform, type Platform } from "./platform";
+import { parseWaybackUrl } from "./wayback";
export type MomentUrlInput = {
// Public origin of the archilyzer viewer that owns this video (a RemoteSource
@@ -55,6 +56,9 @@ export function platformMomentUrl(
if (!webpageUrl) return null;
const secs = Math.max(0, Math.floor(seconds || 0));
if (secs <= 0) return webpageUrl;
+ // A Wayback capture (lib/wayback.ts) is the page that plays, and a time
+ // param would name a different URL — one the Wayback Machine never captured.
+ if (parseWaybackUrl(webpageUrl)) return webpageUrl;
const plat = platform ?? detectPlatform(webpageUrl);
let u: URL;
try {
diff --git a/common/lib/sidecar-server.test.ts b/common/lib/sidecar-server.test.ts
@@ -11,6 +11,7 @@ import "./attribution-server";
import "./diarization-server";
import "./digest-server";
import "./metadataHistory-server";
+import "./wayback-server";
import {
availabilitySidecar,
loadAvailability,
@@ -42,7 +43,7 @@ async function scratch(): Promise<string> {
return mkdtemp(path.join(os.tmpdir(), "sidecar-"));
}
-test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are declared", () => {
+test("every declared sidecar filename escapes SUB_FILE_RE, and all twelve are declared", () => {
assert.deepEqual([...SIDECAR_FILENAMES].sort(), [
"ai-digest.json",
"ai-digest.overrides.json",
@@ -55,6 +56,7 @@ test("every declared sidecar filename escapes SUB_FILE_RE, and all eleven are de
"exclude-truncated-check.json",
"metadata.history.json",
"transcribe-outcome.json",
+ "wayback.json",
]);
for (const name of SIDECAR_FILENAMES) {
assert.ok(!SUB_FILE_RE.test(name), name);
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -5,6 +5,8 @@ import { defaultWebpageUrl, detectPlatform } from "./platform";
import { archiveOrgPlayableUrl } from "./archiveOrg";
import { bitchutePlayableUrl } from "./bitchute";
import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId";
+import { parseWaybackUrl } from "./wayback";
+import { extractVideoId } from "./videoId";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { MediaType, VideoStat, VideoStatus } from "./stats";
import type { VideoState } from "./availability";
@@ -77,7 +79,8 @@ export function loadRawMetadataFromDir(
// The platform a record is from, by yt-dlp's extractor. archive.org's
// extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's
-// `YoutubeWebArchive`, which is a YouTube video.
+// `YoutubeWebArchive`, which is a YouTube video: the original's platform (the
+// record's `wayback.json` says it is a copy, lib/wayback-server.ts).
//
// AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the
// app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has
@@ -144,12 +147,17 @@ export function summarize(
const platform = platformFromMetadata(meta);
// archive.org: yt-dlp's id for one file of an item is `<identifier>/<path>`,
// which is not a slug; the canonical id (lib/archiveOrgId.ts) is.
+ // A Wayback capture (lib/wayback.ts): the id its dir is named by — what
+ // the capture is of (lib/videoId.ts) — not yt-dlp's, which for a raw media
+ // file is the file's name (`<jwId>-<rendition>`).
+ const waybackId = parseWaybackUrl(meta.webpage_url) ? extractVideoId(meta.webpage_url!) : null;
const id =
- platform === "odysee"
+ waybackId ??
+ (platform === "odysee"
? (meta.webpage_url_basename ?? meta.id ?? videoDir)
: platform === "archiveorg"
? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir)
- : (meta.id ?? videoDir);
+ : (meta.id ?? videoDir));
const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1];
return {
slug: `${channelSlug}/${id}`,
diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts
@@ -10,11 +10,31 @@
// for the native-id resolution that reads metadata.info.json.
import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId";
+import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback";
export function extractVideoId(url: string): string | null {
+ return extractVideoIdAt(url, 0);
+}
+
+function extractVideoIdAt(url: string, depth: number): string | null {
try {
const u = new URL(url);
const host = u.hostname.toLowerCase();
+ if (isWaybackHost(host)) {
+ // A Wayback capture is named by what it is a capture OF
+ // (lib/wayback.ts): an archived YouTube page by its YouTube id, a JW
+ // Player file by its media id. Its own path's last segment is the
+ // original's (`watch`, `<id>-<rendition>.mp4`) — a name two captures
+ // share. A capture of a capture is not unwrapped twice.
+ const ref = depth === 0 ? parseWaybackUrl(url) : null;
+ return ref ? extractVideoIdAt(ref.originalUrl, depth + 1) : null;
+ }
+ if (isJwPlayerHost(host)) {
+ // One media id across every rendition and host; a JW URL that names
+ // none falls through to the last segment below.
+ const jw = jwPlayerMediaId(u);
+ if (jw) return jw;
+ }
if (isArchiveOrgItemHost(host)) {
// A whole item → its identifier; one file inside an item → a stable
// `<identifier>__<slug>-<hash>` (lib/archiveOrgId.ts). A URL that names
diff --git a/common/lib/wayback-server.ts b/common/lib/wayback-server.ts
@@ -0,0 +1,43 @@
+// WAYBACK PROVENANCE ON DISK — the `wayback.json` sidecar.
+//
+// A record downloaded from a Wayback Machine capture (lib/wayback.ts) carries
+// what it is a copy of: the original URL, the capture's timestamp, the capture
+// as a page that plays and as its raw bytes. Built from the capture URL alone
+// — no request — so it is written on every download of one
+// (ytdlp/downloadOneManaged.ts) and by `archilyzer wayback refresh` for a
+// record imported before this existed (controller/waybackRefresh.ts).
+
+import {
+ WAYBACK_PROVENANCE_FILENAME,
+ buildWaybackProvenance,
+ coerceWaybackProvenance,
+ sameWaybackProvenance,
+ type WaybackProvenance,
+} from "./wayback";
+import { sidecar, sidecarField } from "./sidecar-server";
+
+export const waybackProvenanceSidecar = sidecar(
+ WAYBACK_PROVENANCE_FILENAME,
+ sidecarField(coerceWaybackProvenance),
+);
+
+export const { load: loadWaybackProvenance, write: writeWaybackProvenance } =
+ waybackProvenanceSidecar;
+
+// The sidecar for a record fetched by `url`: written when the URL is a
+// capture and the one on disk is absent or says otherwise. Returns what the
+// record now carries (null for a URL that is not a capture) and whether it was
+// (or, with `dryRun`, would be) written.
+export async function ensureWaybackProvenance(
+ videoDir: string,
+ url: string,
+ opts: { dryRun?: boolean; onLog?: (line: string) => void } = {},
+): Promise<{ provenance: WaybackProvenance | null; written: boolean }> {
+ const next = buildWaybackProvenance(url);
+ if (!next) return { provenance: null, written: false };
+ const prev = await loadWaybackProvenance(videoDir);
+ if (prev && sameWaybackProvenance(prev, next)) return { provenance: prev, written: false };
+ if (!opts.dryRun) await writeWaybackProvenance(videoDir, next);
+ opts.onLog?.(`Wayback provenance: a capture of ${next.originalUrl} (${next.captureTs}).\n`);
+ return { provenance: next, written: true };
+}
diff --git a/common/lib/wayback.test.ts b/common/lib/wayback.test.ts
@@ -0,0 +1,162 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ buildWaybackProvenance,
+ coerceWaybackProvenance,
+ jwPlayerMediaId,
+ parseWaybackUrl,
+ waybackCaptureDate,
+ waybackCitationLinks,
+} from "./wayback";
+import { ensureWaybackProvenance, loadWaybackProvenance } from "./wayback-server";
+import { extractVideoId } from "./videoId";
+import { platformMomentUrl } from "./momentUrl";
+import { platformFromMetadata, summarize } from "./transcripts-server";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/wayback.test.ts
+//
+// Every id, account and token here is invented.
+
+const YT = "Abc123def45";
+const JW = "Qw3rTy12";
+const ARCHIVED_PAGE = `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`;
+const JW_FILE = `https://videos-fms.jwpsrv.com/content/conversions/AcCt1234/videos/${JW}-12345678.mp4?token=0_abc_0xdef`;
+const ARCHIVED_FILE = `https://web.archive.org/web/20200102030405id_/${JW_FILE}`;
+
+test("parseWaybackUrl: timestamp, modifier and the original with its own query", () => {
+ assert.deepEqual(parseWaybackUrl(ARCHIVED_PAGE), {
+ captureTs: "20210102030405",
+ modifier: "",
+ originalUrl: `https://www.youtube.com/watch?v=${YT}`,
+ });
+ assert.deepEqual(parseWaybackUrl(ARCHIVED_FILE), {
+ captureTs: "20200102030405",
+ modifier: "id_",
+ originalUrl: JW_FILE,
+ });
+ // A short timestamp, another modifier, no scheme, a collapsed scheme, an
+ // encoded original, the older path without /web/.
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/2019im_/example.com/a.png")?.originalUrl, "http://example.com/a.png");
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/2019/https:/example.com/x")?.originalUrl, "https://example.com/x");
+ assert.equal(
+ parseWaybackUrl("https://web.archive.org/web/2019/https%3A%2F%2Fexample.com%2Fx")?.originalUrl,
+ "https://example.com/x",
+ );
+ assert.equal(parseWaybackUrl("http://wayback.archive.org/20190101000000/http://example.com/")?.captureTs, "20190101000000");
+ // Not captures.
+ assert.equal(parseWaybackUrl("https://web.archive.org/web/*/example.com"), null);
+ assert.equal(parseWaybackUrl("https://archive.org/details/some-item"), null);
+ assert.equal(parseWaybackUrl(`https://www.youtube.com/watch?v=${YT}`), null);
+ assert.equal(parseWaybackUrl("not a url"), null);
+});
+
+test("jwPlayerMediaId: the media id across hosts and renditions", () => {
+ assert.equal(jwPlayerMediaId(JW_FILE), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/videos/${JW}-AbCdEf12.mp4`), JW);
+ assert.equal(jwPlayerMediaId(`https://content.jwplatform.com/videos/${JW}.mp4`), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/manifests/${JW}.m3u8`), JW);
+ assert.equal(jwPlayerMediaId(`https://cdn.jwplayer.com/v2/media/${JW}`), JW);
+ assert.equal(jwPlayerMediaId("https://cdn.jwplayer.com/libraries/AbCd1234.js"), null);
+ assert.equal(jwPlayerMediaId(`https://example.com/videos/${JW}-1.mp4`), null);
+});
+
+test("extractVideoId unwraps a capture into what it is a capture of", () => {
+ assert.equal(extractVideoId(ARCHIVED_PAGE), YT);
+ assert.equal(extractVideoId(`https://web.archive.org/web/2021if_/https://youtu.be/${YT}`), YT);
+ assert.equal(extractVideoId(ARCHIVED_FILE), JW);
+ assert.equal(extractVideoId(JW_FILE), JW);
+ // Unwrapped once: a capture of a capture names no id.
+ assert.equal(extractVideoId(`https://web.archive.org/web/2021/${ARCHIVED_PAGE}`), null);
+ // A Wayback page that is not a capture names no record.
+ assert.equal(extractVideoId("https://web.archive.org/web/*/example.com"), null);
+ // A capture of a page the app does not know: the original's last segment.
+ assert.equal(extractVideoId("https://web.archive.org/web/2021/https://example.com/media/clip-7.mp4"), "clip-7.mp4");
+ // Nothing else moved.
+ assert.equal(extractVideoId(`https://www.youtube.com/watch?v=${YT}`), YT);
+ assert.equal(extractVideoId("https://example.com/feed/episode-1.mp3"), "episode-1.mp3");
+});
+
+test("buildWaybackProvenance: an archived YouTube page's original is its watch URL", () => {
+ assert.deepEqual(buildWaybackProvenance(`https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`), {
+ originalUrl: `https://www.youtube.com/watch?v=${YT}`,
+ captureTs: "20210102",
+ waybackUrl: `https://web.archive.org/web/20210102/https://m.youtube.com/watch?v=${YT}&feature=share`,
+ rawUrl: `https://web.archive.org/web/20210102id_/https://m.youtube.com/watch?v=${YT}&feature=share`,
+ });
+ const file = buildWaybackProvenance(ARCHIVED_FILE)!;
+ assert.equal(file.originalUrl, JW_FILE);
+ assert.equal(file.waybackUrl, `https://web.archive.org/web/20200102030405/${JW_FILE}`);
+ assert.equal(file.rawUrl, ARCHIVED_FILE);
+ assert.equal(buildWaybackProvenance(JW_FILE), null);
+ assert.deepEqual(coerceWaybackProvenance(JSON.parse(JSON.stringify(file))), file);
+ assert.equal(coerceWaybackProvenance({ ...file, captureTs: "yesterday" }), null);
+ assert.equal(coerceWaybackProvenance([]), null);
+});
+
+test("waybackCaptureDate and the citation links", () => {
+ assert.equal(waybackCaptureDate("20210102030405"), "2021-01-02");
+ assert.equal(waybackCaptureDate("2021"), "2021");
+ assert.equal(waybackCaptureDate("202101"), "2021-01");
+ const prov = buildWaybackProvenance(ARCHIVED_PAGE)!;
+ const links = waybackCitationLinks(prov, {
+ originalMomentUrl: platformMomentUrl(prov.originalUrl, null, 90),
+ });
+ assert.deepEqual(links.original, {
+ label: "Original (may be gone)",
+ url: `https://www.youtube.com/watch?v=${YT}&t=90s`,
+ });
+ assert.deepEqual(links.copy, {
+ label: "Wayback Machine copy, 2021-01-02",
+ url: `https://web.archive.org/web/20210102030405/https://www.youtube.com/watch?v=${YT}`,
+ });
+});
+
+test("platformMomentUrl: a capture is the page that plays, never given a time param", () => {
+ assert.equal(platformMomentUrl(ARCHIVED_PAGE, "youtube", 125), ARCHIVED_PAGE);
+ assert.equal(platformMomentUrl(`https://www.youtube.com/watch?v=${YT}`, "youtube", 125), `https://www.youtube.com/watch?v=${YT}&t=125s`);
+});
+
+test("platformFromMetadata: YoutubeWebArchive is the original's platform, YouTube", () => {
+ assert.equal(platformFromMetadata({ extractor_key: "YoutubeWebArchive", extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube");
+ assert.equal(platformFromMetadata({ extractor: "web.archive:youtube", webpage_url: ARCHIVED_PAGE }), "youtube");
+});
+
+test("summarize: a capture's id is its dir's name, not yt-dlp's file-name id", () => {
+ const s = summarize("demo", `${JW}`, {
+ id: `${JW}-12345678`,
+ title: `${JW}-12345678`,
+ extractor_key: "Generic",
+ webpage_url: ARCHIVED_FILE,
+ });
+ assert.equal(s.id, JW);
+ assert.equal(s.slug, `demo/${JW}`);
+ const page = summarize("demo", YT, { id: YT, title: "A title", extractor_key: "YoutubeWebArchive", webpage_url: ARCHIVED_PAGE });
+ assert.equal(page.id, YT);
+ assert.equal(page.platform, "youtube");
+ assert.equal(page.webpageUrl, ARCHIVED_PAGE);
+});
+
+test("ensureWaybackProvenance writes the sidecar once, and nothing for a non-capture", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "wayback-sidecar-"));
+ assert.deepEqual(await ensureWaybackProvenance(dir, JW_FILE), { provenance: null, written: false });
+ assert.equal(await loadWaybackProvenance(dir), null);
+
+ const dry = await ensureWaybackProvenance(dir, ARCHIVED_PAGE, { dryRun: true });
+ assert.equal(dry.written, true);
+ assert.equal(await loadWaybackProvenance(dir), null);
+
+ const first = await ensureWaybackProvenance(dir, ARCHIVED_PAGE);
+ assert.equal(first.written, true);
+ const onDisk = JSON.parse(await readFile(path.join(dir, "wayback.json"), "utf8"));
+ assert.deepEqual(onDisk, buildWaybackProvenance(ARCHIVED_PAGE));
+ assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, false);
+
+ // A sidecar for another capture is rewritten.
+ await writeFile(path.join(dir, "wayback.json"), JSON.stringify(buildWaybackProvenance(ARCHIVED_FILE)));
+ assert.equal((await ensureWaybackProvenance(dir, ARCHIVED_PAGE)).written, true);
+ assert.deepEqual(await loadWaybackProvenance(dir), buildWaybackProvenance(ARCHIVED_PAGE));
+});
diff --git a/common/lib/wayback.ts b/common/lib/wayback.ts
@@ -0,0 +1,211 @@
+// THE WAYBACK MACHINE — a capture of some other URL, and what it was a copy of.
+//
+// A Wayback capture URL wraps the original: `https://web.archive.org/web/
+// <timestamp>[<modifier>]/<original>`, where the timestamp is 4–14 digits
+// (yyyy[MM[dd[hh[mm[ss]]]]]) and the modifier is the replay mode — none (the
+// page in the Wayback frame, which is what plays), `id_` (the raw bytes as
+// captured), `im_`, `if_`, `js_`, `cs_`, `oe_`, … The original keeps its own
+// query string, so it is the rest of the path PLUS the capture URL's search.
+//
+// yt-dlp downloads either kind through the editor's import: an archived
+// YouTube page through its `YoutubeWebArchive` extractor (the record's id is
+// the YouTube id), a raw media file through `generic` (an id that is the file
+// name). Neither knows it is a copy; the `wayback.json` sidecar
+// (lib/wayback-server.ts) records that, and lib/videoId.ts names the record by
+// what the capture is OF.
+//
+// A LEAF: no imports, so lib/videoId.ts (itself a leaf apart from
+// archiveOrgId.ts) can unwrap a capture without a cycle.
+
+export const WAYBACK_PROVENANCE_FILENAME = "wayback.json";
+
+// Hosts that serve Wayback captures. archive.org's own hosts serve items
+// (lib/archiveOrgId.ts), never captures.
+const WAYBACK_HOSTS = new Set(["web.archive.org", "wayback.archive.org"]);
+
+export function isWaybackHost(host: string): boolean {
+ return WAYBACK_HOSTS.has(host.toLowerCase());
+}
+
+export type WaybackRef = {
+ // The capture's timestamp as the URL gives it (4–14 digits).
+ captureTs: string;
+ // The replay modifier (`id_`, `im_`, …), "" for the framed page.
+ modifier: string;
+ // The URL the capture is of, as the capture URL names it (scheme added
+ // when the capture URL left it off).
+ originalUrl: string;
+};
+
+// `/web/<ts><mod>/<original>`, or the older `/<ts><mod>/<original>`.
+const CAPTURE_PATH_RE = /^\/(?:web\/)?(\d{4,14})([a-z]{2}_)?\/(.+)$/s;
+
+// A Wayback capture URL's parts, or null for anything else (a calendar page
+// `/web/*/<url>`, a search, another host).
+export function parseWaybackUrl(url: string | null | undefined): WaybackRef | null {
+ if (!url) return null;
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ if (!isWaybackHost(u.hostname)) return null;
+ const m = CAPTURE_PATH_RE.exec(u.pathname);
+ if (!m) return null;
+ let rest = m[3];
+ // A percent-encoded original (`https%3A%2F%2F…`) is the same URL.
+ if (/^https?%3a/i.test(rest)) {
+ try {
+ rest = decodeURIComponent(rest);
+ } catch {
+ return null;
+ }
+ }
+ // The WHATWG parser keeps `https://` inside a path as written, but a capture
+ // URL that went through a path normaliser arrives as `https:/host`.
+ rest = rest.replace(/^(https?):\/(?!\/)/i, "$1://");
+ if (!/^https?:\/\//i.test(rest)) rest = `http://${rest}`;
+ const original = `${rest}${u.search}${u.hash}`;
+ try {
+ new URL(original);
+ } catch {
+ return null;
+ }
+ return { captureTs: m[1], modifier: m[2] ?? "", originalUrl: original };
+}
+
+// The capture as a page that plays (the Wayback frame, no modifier).
+export function waybackPageUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string {
+ return `https://web.archive.org/web/${ref.captureTs}/${ref.originalUrl}`;
+}
+
+// The capture's raw bytes (`id_`), the form yt-dlp fetches a media file by.
+export function waybackRawUrl(ref: Pick<WaybackRef, "captureTs" | "originalUrl">): string {
+ return `https://web.archive.org/web/${ref.captureTs}id_/${ref.originalUrl}`;
+}
+
+// `YYYY-MM-DD` of a capture timestamp, or the year (/month) when that is all
+// it names.
+export function waybackCaptureDate(captureTs: string): string {
+ const y = captureTs.slice(0, 4);
+ const mo = captureTs.slice(4, 6);
+ const d = captureTs.slice(6, 8);
+ return [y, mo, d].filter((p) => p.length === 2 || p.length === 4).join("-");
+}
+
+// ─── JW Player ───
+
+// The hosts JW Player serves a media file from. A file there is
+// `…/videos/<mediaId>-<rendition>.<ext>` (cdn.jwplayer.com/videos/…,
+// content.jwplatform.com/videos/…, videos-fms.jwpsrv.com/content/conversions/
+// <account>/videos/…); the media id is eight alphanumerics and names the video
+// across every rendition.
+const JW_HOSTS = ["jwplayer.com", "jwplatform.com", "jwpsrv.com"];
+
+export function isJwPlayerHost(host: string): boolean {
+ const h = host.toLowerCase();
+ return JW_HOSTS.some((d) => h === d || h.endsWith(`.${d}`));
+}
+
+const JW_FILE_RE = /\/videos\/([A-Za-z0-9]{8})(?:-[A-Za-z0-9]+)?\.[A-Za-z0-9]+$/;
+const JW_MEDIA_RE = /\/(?:manifests|v2\/media|previews)\/([A-Za-z0-9]{8})(?:[-./]|$)/;
+
+// The JW media id a JW Player file or manifest URL names, or null.
+export function jwPlayerMediaId(url: string | URL): string | null {
+ let u: URL;
+ try {
+ u = typeof url === "string" ? new URL(url) : url;
+ } catch {
+ return null;
+ }
+ if (!isJwPlayerHost(u.hostname)) return null;
+ const m = JW_FILE_RE.exec(u.pathname) ?? JW_MEDIA_RE.exec(u.pathname);
+ return m ? m[1] : null;
+}
+
+// ─── The sidecar's record ───
+
+export type WaybackProvenance = {
+ // What the capture is a copy of. For an archived YouTube page, its watch
+ // URL (`https://www.youtube.com/watch?v=<id>`), whatever form was captured.
+ originalUrl: string;
+ // The capture's timestamp (4–14 digits).
+ captureTs: string;
+ // The capture as a page that plays.
+ waybackUrl: string;
+ // The capture's raw bytes.
+ rawUrl: string;
+};
+
+// A YouTube URL as its watch page, or null for any other URL.
+function youtubeWatchUrl(url: string): string | null {
+ let u: URL;
+ try {
+ u = new URL(url);
+ } catch {
+ return null;
+ }
+ const host = u.hostname.toLowerCase();
+ let id: string | null = null;
+ if (host === "youtu.be") id = u.pathname.split("/").filter(Boolean)[0] ?? null;
+ else if (host === "youtube.com" || host.endsWith(".youtube.com")) {
+ id = u.searchParams.get("v");
+ if (!id) {
+ const segs = u.pathname.split("/").filter(Boolean);
+ if ((segs[0] === "embed" || segs[0] === "shorts" || segs[0] === "v" || segs[0] === "live") && segs[1]) id = segs[1];
+ }
+ } else return null;
+ return id ? `https://www.youtube.com/watch?v=${id}` : null;
+}
+
+// The sidecar for a capture URL, or null when the URL is not one.
+export function buildWaybackProvenance(url: string): WaybackProvenance | null {
+ const ref = parseWaybackUrl(url);
+ if (!ref) return null;
+ return {
+ originalUrl: youtubeWatchUrl(ref.originalUrl) ?? ref.originalUrl,
+ captureTs: ref.captureTs,
+ waybackUrl: waybackPageUrl(ref),
+ rawUrl: waybackRawUrl(ref),
+ };
+}
+
+export function coerceWaybackProvenance(value: unknown): WaybackProvenance | null {
+ if (!value || typeof value !== "object" || Array.isArray(value)) return null;
+ const v = value as Record<string, unknown>;
+ const str = (k: string) => (typeof v[k] === "string" && v[k] ? (v[k] as string) : null);
+ const originalUrl = str("originalUrl");
+ const captureTs = str("captureTs");
+ const waybackUrl = str("waybackUrl");
+ const rawUrl = str("rawUrl");
+ if (!originalUrl || !captureTs || !/^\d{4,14}$/.test(captureTs) || !waybackUrl || !rawUrl) return null;
+ return { originalUrl, captureTs, waybackUrl, rawUrl };
+}
+
+export function sameWaybackProvenance(a: WaybackProvenance, b: WaybackProvenance): boolean {
+ return (
+ a.originalUrl === b.originalUrl &&
+ a.captureTs === b.captureTs &&
+ a.waybackUrl === b.waybackUrl &&
+ a.rawUrl === b.rawUrl
+ );
+}
+
+// ─── Citing it ───
+
+export type WaybackLink = { label: string; url: string };
+
+// What a citation of an archived copy links: the original, named as the
+// original and as possibly gone (a capture exists because it may be), and the
+// Wayback copy, which plays. `originalMomentUrl` is the original at the cited
+// second when its platform takes one (lib/momentUrl.ts).
+export function waybackCitationLinks(
+ prov: WaybackProvenance,
+ opts: { originalMomentUrl?: string | null } = {},
+): { original: WaybackLink; copy: WaybackLink } {
+ return {
+ original: { label: "Original (may be gone)", url: opts.originalMomentUrl || prov.originalUrl },
+ copy: { label: `Wayback Machine copy, ${waybackCaptureDate(prov.captureTs)}`, url: prov.waybackUrl },
+ };
+}
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -1,7 +1,7 @@
import { removeMediaFile } from "../lib/mediaTier-server";
import { tierVideoDir } from "../lib/mediaTier-server";
import path from "node:path";
-import { appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises";
+import { access, appendFile, mkdir, readdir, readFile, rm, stat } from "node:fs/promises";
import { createWriteStream, type Dirent, type WriteStream } from "node:fs";
import { execa } from "execa";
import {
@@ -50,6 +50,8 @@ import { writeDownloadOutcome } from "../lib/downloadOutcome-server";
import { formatBytes } from "../lib/format";
import { recordAvailability } from "../lib/availability-server";
import { withMetadataHistory } from "../lib/metadataHistory-server";
+import { ensureWaybackProvenance } from "../lib/wayback-server";
+import { parseWaybackUrl } from "../lib/wayback";
import {
loadRawMetadata,
loadRawMetadataFromDir,
@@ -656,12 +658,37 @@ export async function downloadOneManaged(
if (detectPlatform(opts.videoUrl) === "archiveorg") {
return await downloadArchiveOrgManaged(opts, opts.archiveOrgDeps);
}
- return await runManagedDownload(opts, channelDir, startedAt, canonicalId);
+ const outcome = await runManagedDownload(opts, channelDir, startedAt, canonicalId);
+ await recordWaybackProvenance(opts, channelDir, canonicalId);
+ return outcome;
} finally {
logStream?.end();
}
}
+// A WAYBACK CAPTURE IS A COPY (lib/wayback.ts): once the record exists — its
+// metadata.info.json written by the prefetch or the download — the
+// `wayback.json` sidecar says of what. Built from the URL alone, so it costs no
+// request; never fails the download.
+async function recordWaybackProvenance(
+ opts: ManagedDownloadOpts,
+ channelDir: string,
+ canonicalId: string | null,
+): Promise<void> {
+ if (!canonicalId || !parseWaybackUrl(opts.videoUrl)) return;
+ const videoDir = path.join(channelDir, "data", canonicalId);
+ try {
+ await access(path.join(videoDir, "metadata.info.json"));
+ } catch {
+ return;
+ }
+ try {
+ await ensureWaybackProvenance(videoDir, opts.videoUrl, { onLog: opts.onLog });
+ } catch (err) {
+ opts.onLog(`Wayback provenance not written: ${(err as Error).message}\n`);
+ }
+}
+
async function runManagedDownload(
opts: ManagedDownloadOpts,
channelDir: string,