// The canonical video id derived from a video URL — the name every video's // data// dir carries, on every platform. Deliberately a leaf module whose // one import (archiveOrgId.ts) is itself a leaf with no imports: it lives here // rather than in ytdlp/runYtdlp.ts (its original home, which still re-exports it) so that low-level stores like // controller/rosterStore.ts can canonicalize a URL without pulling in execa, // the settings loader and the whole download pipeline. // // Canonical is NOT the same as yt-dlp's native extractor id: the two coincide // on YouTube and diverge everywhere else. See archiveIdForUrl in runYtdlp.ts // for the native-id resolution that reads metadata.info.json. import { isArchiveOrgItemHost, parseArchiveOrgUrl, archiveOrgVideoId } from "./archiveOrgId"; import { isJwPlayerHost, isWaybackHost, jwPlayerMediaId, parseWaybackUrl } from "./wayback"; export function extractVideoId(url: string): string | null { return extractVideoIdAt(url, 0); } function extractVideoIdAt(url: string, depth: number): string | null { try { const u = new URL(url); const host = u.hostname.toLowerCase(); if (isWaybackHost(host)) { // A Wayback capture is named by what it is a capture OF // (lib/wayback.ts): an archived YouTube page by its YouTube id, a JW // Player file by its media id. Its own path's last segment is the // original's (`watch`, `-.mp4`) — a name two captures // share. A capture of a capture is not unwrapped twice. const ref = depth === 0 ? parseWaybackUrl(url) : null; return ref ? extractVideoIdAt(ref.originalUrl, depth + 1) : null; } if (isJwPlayerHost(host)) { // One media id across every rendition and host; a JW URL that names // none falls through to the last segment below. const jw = jwPlayerMediaId(u); if (jw) return jw; } if (isArchiveOrgItemHost(host)) { // A whole item → its identifier; one file inside an item → a stable // `__-` (lib/archiveOrgId.ts). A URL that names // no item (a search page, a collection listing) has no id. const ref = parseArchiveOrgUrl(url); return ref ? archiveOrgVideoId(ref) : null; } if (host.endsWith("youtube.com") || host === "youtu.be") { const v = u.searchParams.get("v"); if (v) return v; const seg = u.pathname.split("/").filter(Boolean).pop(); return seg ?? null; } if (host.endsWith("rumble.com")) { const seg = u.pathname.split("/").filter(Boolean).pop(); // Rumble paths often look like /v123abc-some-title.html if (seg) { const trimmed = seg.replace(/\.html?$/i, ""); const dashIdx = trimmed.indexOf("-"); return dashIdx > 0 ? trimmed.slice(0, dashIdx) : trimmed; } return null; } if (host.endsWith("odysee.com")) { const decoded = decodeURIComponent(u.pathname); const lastColon = decoded.lastIndexOf(":"); if (lastColon > 0) return decoded.slice(lastColon + 1); return null; } if (host === "bitchute.com" || host.endsWith(".bitchute.com")) { // A video page or its embed: /video//, /embed// (www. or old.). // A channel, playlist or profile page names no video. const segs = u.pathname.split("/").filter(Boolean); return (segs[0] === "video" || segs[0] === "embed") && segs[1] ? segs[1] : null; } if (host.endsWith("twitch.tv")) { // VODs: /videos/ or legacy //v/; clips: // clips.twitch.tv/ or //clip/. Grab the segment // after the marker; otherwise fall back to the last path segment. const segs = u.pathname.split("/").filter(Boolean); for (const marker of ["videos", "v", "clip", "clips"]) { const idx = segs.indexOf(marker); if (idx >= 0 && segs[idx + 1]) return segs[idx + 1]; } return segs.pop() ?? null; } if (host.endsWith("kick.com")) { // VODs: //videos/ or /video/; clips: // //clips/ or /clip/. The UUID after the marker is // yt-dlp's native Kick id; fall back to the last segment otherwise. const segs = u.pathname.split("/").filter(Boolean); for (const marker of ["videos", "video", "clips", "clip"]) { const idx = segs.indexOf(marker); if (idx >= 0 && segs[idx + 1]) return segs[idx + 1]; } return segs.pop() ?? null; } const seg = u.pathname.split("/").filter(Boolean).pop(); return seg ?? null; } catch { return null; } }