// Video platforms plus the social-post platforms (see common/lib/posts.ts). // Widening this allowlist is purely additive: it cannot invalidate existing // cached entries, so no transcriptStore.ts DB_VERSION bump is needed. Posts get // their own IDB store rather than sharing the transcript cache. export type Platform = | "youtube" | "rumble" | "odysee" | "twitch" | "kick" // archive.org items (lib/archiveOrgId.ts): a whole item, or one file inside // a multi-file item. | "archiveorg" // BitChute videos (bitchute.com/video//) and channels. | "bitchute" | "twitter" | "bluesky" | "xenforo"; export const PLATFORM_VALUES: ReadonlyArray = [ "youtube", "rumble", "odysee", "twitch", "kick", "archiveorg", "bitchute", "twitter", "bluesky", "xenforo", ]; // The subset that carries social posts rather than videos. A channel on one of // these is a `sourceKind: "social"` channel — the video scan skips it and the // posts pipeline picks it up. export const SOCIAL_PLATFORM_VALUES: ReadonlyArray = [ "twitter", "bluesky", "xenforo", ]; export function isSocialPlatform(platform: Platform | null | undefined): boolean { return platform === "twitter" || platform === "bluesky" || platform === "xenforo"; } // Host → platform. Lives in `detectPlatform.mjs` (plain JS so umtool's `.mjs` // scripts can import the same copy); every TS caller imports it from here. import { detectPlatform, sameXenforoForum } from "./detectPlatform.mjs"; export { detectPlatform, sameXenforoForum }; // Best-effort canonical webpage URL for a video given its platform and // canonical id. Used both as a metadata fallback in summarize() and to // re-create a download URL when a video has no metadata.info.json. Note: for // Odysee the canonical id is only the claim hash (the channel-name prefix is // dropped by extractVideoId), so a reconstructed Odysee URL may not resolve. export function defaultWebpageUrl(platform: Platform, id: string): string { if (platform === "rumble") return `https://rumble.com/${id}`; if (platform === "odysee") return `https://odysee.com/${id}`; if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`; // Kick's canonical id is the VOD UUID; /video/ resolves to the VOD. if (platform === "kick") return `https://kick.com/video/${id}`; // A whole item's id IS its identifier. A file's id is a slug + hash the file // path cannot be recovered from, so this is the right page only for a whole // item; every archived record carries its own `webpage_url`. if (platform === "archiveorg") { const file = /^(.+?)__.*-[0-9a-f]{8}$/.exec(id); return `https://archive.org/details/${file ? file[1] : id}`; } // BitChute's canonical id is its video id, which is also yt-dlp's. if (platform === "bitchute") return `https://www.bitchute.com/video/${id}/`; // Social posts: /i/status/ resolves without knowing the handle. Bluesky // has no handle-free permalink, so this is only a last-resort fallback — // every archived post carries its own canonical `url` (see postPermalink). if (platform === "twitter") return `https://x.com/i/status/${id}`; if (platform === "bluesky") return `https://bsky.app/profile/${id}`; // A forum post's URL needs its forum's host, which an id alone does not // carry; every archived forum post has its own `url`. A thread URL passed as // the id is returned as it is. if (platform === "xenforo") return /^https?:\/\//.test(id) ? id : ""; return `https://www.youtube.com/watch?v=${id}`; } export function platformQueueKey( platform: Platform | null | undefined, ): string { return `platform:${platform ?? "unknown"}`; } // Best-effort registrable domain for an arbitrary URL, used to route // unrecognized sources to their own job queue. Uses a dependency-free // "last two labels" heuristic (clips.twitch.tv -> twitch.tv, // player.vimeo.com -> vimeo.com). Multi-part TLDs coarsen (example.co.uk // -> co.uk), which is harmless for a queue key — at worst two unrelated // sources share a serial queue. Returns null when the URL can't be parsed. export function registrableDomain(url: string | undefined | null): string | null { if (!url) return null; try { const host = new URL(url).hostname.toLowerCase(); if (!host) return null; const labels = host.split(".").filter(Boolean); if (labels.length <= 2) return labels.join(".") || null; return labels.slice(-2).join("."); } catch { return null; } } // Queue key for a channel/video URL. Known platforms get their canonical // `platform:` queue; unrecognized hosts fall back to a per-domain queue // (e.g. `platform:vimeo.com`) instead of all colliding in `platform:unknown`. export function queueKeyForUrl(url: string | undefined | null): string { const detected = detectPlatform(url); if (detected) return platformQueueKey(detected); const host = registrableDomain(url); return host ? `platform:${host}` : platformQueueKey(null); } // System-wide queue for resource-bound local jobs (whisper, ffmpeg). // Channels share this queue so two heavy local jobs never run in parallel. export const TRANSCRIPTION_QUEUE = "transcription";