import { readFile } from "node:fs/promises"; import path from "node:path"; import { formatDate, formatDuration } from "./format"; import { defaultWebpageUrl, detectPlatform } from "./platform"; import { archiveOrgPlayableUrl } from "./archiveOrg"; import { bitchutePlayableUrl } from "./bitchute"; import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId"; import { parseWaybackUrl } from "./wayback"; import { extractVideoId } from "./videoId"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { MediaType, VideoStat, VideoStatus } from "./stats"; import type { VideoState } from "./availability"; export type RawMetadata = { id?: string; title?: string; upload_date?: string; duration?: number; channel?: string; uploader?: string; description?: string; is_live?: boolean; was_live?: boolean; live_status?: string; age_limit?: number; extractor?: string; extractor_key?: string; webpage_url?: string; webpage_url_basename?: string; // HLS master playlist URL yt-dlp resolves for stream/VOD extractors. Kick // VODs have no iframe embed, so we persist this to play them via hls.js in // the export. Unsigned and CORS-open on Kick (stream.kick.com/…/master.m3u8). manifest_url?: string; // Engagement / categorical fields used by the charts feature. Present in // yt-dlp's metadata.info.json but not all platforms/older archives carry them. timestamp?: number; view_count?: number; like_count?: number; comment_count?: number; channel_follower_count?: number; categories?: string[]; tags?: string[]; language?: string; // yt-dlp's coarse kind: "video" | "livestream" | "short" (YouTube). Absent on // platforms that don't distinguish, where we fall back to is/was_live. media_type?: string; // Every format the extractor offered. Read only for archive.org and // BitChute, where each is a plain download URL and one of them is what the // player plays (lib/archiveOrg.ts archiveOrgPlayableUrl, lib/bitchute.ts). formats?: unknown; // The chosen format's URL, which yt-dlp writes at the top level. Read only // for BitChute, as the playable file's fallback. url?: unknown; ext?: unknown; }; // Read and parse a video's metadata.info.json into the typed RawMetadata // shape. Returns null on a missing/unreadable/invalid file. The canonical // loader for the four spots that previously parsed it inline. export async function loadRawMetadata( infoJsonPath: string, ): Promise { try { const raw = await readFile(infoJsonPath, "utf8"); const parsed = JSON.parse(raw); if (!parsed || typeof parsed !== "object") return null; return parsed as RawMetadata; } catch { return null; } } // Convenience: load metadata.info.json from a video's data dir. export function loadRawMetadataFromDir( videoDir: string, ): Promise { return loadRawMetadata(path.join(videoDir, "metadata.info.json")); } // The platform a record is from, by yt-dlp's extractor. archive.org's // extractor is `ArchiveOrg` (key) / `archive.org` (name) — NOT web.archive.org's // `YoutubeWebArchive`, which is a YouTube video: the original's platform (the // record's `wayback.json` says it is a copy, lib/wayback-server.ts). // // AN UNKNOWN EXTRACTOR: the record's own page decides when its host is one the // app knows (lib/detectPlatform.mjs); otherwise "youtube", as it always has // been — the generic extractor (a podcast episode imported by its enclosure // URL) still lands there, and changing that would relabel records already // published. export function platformFromMetadata(meta: RawMetadata): Platform { const key = meta.extractor_key ?? meta.extractor ?? ""; if (/^rumble/i.test(key)) return "rumble"; if (/^lbry/i.test(key)) return "odysee"; if (/^twitch/i.test(key)) return "twitch"; if (/^kick/i.test(key)) return "kick"; if (/^archive\.?org$/i.test(key)) return "archiveorg"; // `BitChute` (a video) and `BitChuteChannel` (a listing). if (/^bitchute/i.test(key)) return "bitchute"; if (/^youtube/i.test(key)) return "youtube"; const fromPage = detectPlatform(meta.webpage_url); if (fromPage && PAGE_DECIDES.includes(fromPage)) return fromPage; return "youtube"; } // The platforms whose page decides a record's label even under an extractor // the app does not know (above). const PAGE_DECIDES: ReadonlyArray = ["archiveorg", "bitchute"]; // A SUMMARY LABELLED BEFORE ITS PLATFORM WAS KNOWN: its own page is on a // platform whose page decides the label, and the label says otherwise — what a // BitChute record summarized before the bitchute platform existed carries // ("youtube", and no file to play). The index re-derives such a summary from // the record's metadata (controller/buildIndex.ts), and so does every reader // of a normalized transcript.cues.json, whose summary was frozen when it was // written. Pure: the summary alone decides. export function platformLabelStale(summary: { platform?: Platform; webpageUrl?: string; }): boolean { const fromPage = detectPlatform(summary.webpageUrl); return fromPage !== null && PAGE_DECIDES.includes(fromPage) && fromPage !== summary.platform; } // The broad "is this a livestream (or stream VOD/upcoming)" notion used by the // coverage detector and the download-time duration guard, so both skip the same // content (stream captures have unreliable metadata durations). Distinct from // the narrower download-filter `skipLive`, which intentionally lets finished // stream VODs through. export function isLivestreamMetadata(meta: RawMetadata): boolean { const liveStatus = meta.live_status ?? ""; return ( meta.was_live === true || meta.is_live === true || liveStatus === "was_live" || liveStatus === "is_live" || liveStatus === "is_upcoming" ); } export function summarize( channelSlug: string, videoDir: string, meta: RawMetadata, configName?: string, ): TranscriptSummary { const isLivestream = isLivestreamMetadata(meta); const platform = platformFromMetadata(meta); // archive.org: yt-dlp's id for one file of an item is `/`, // which is not a slug; the canonical id (lib/archiveOrgId.ts) is. // A Wayback capture (lib/wayback.ts): the id its dir is named by — what // the capture is of (lib/videoId.ts) — not yt-dlp's, which for a raw media // file is the file's name (`-`). const waybackId = parseWaybackUrl(meta.webpage_url) ? extractVideoId(meta.webpage_url!) : null; const id = waybackId ?? (platform === "odysee" ? (meta.webpage_url_basename ?? meta.id ?? videoDir) : platform === "archiveorg" ? (archiveOrgVideoIdFromNativeId(meta.id) ?? videoDir) : (meta.id ?? videoDir)); const dateFromDir = videoDir.match(/^(\d{8})(?:_|$)/)?.[1]; return { slug: `${channelSlug}/${id}`, id, channelSlug, title: meta.title ?? id, uploadDate: meta.upload_date ?? dateFromDir ?? "", duration: meta.duration ?? 0, channel: configName ?? meta.channel ?? meta.uploader ?? "", description: meta.description ?? "", tags: tagsOf(meta), isLivestream, ageRestricted: (meta.age_limit ?? 0) > 0, platform, webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id), // Kick VODs play from a persisted HLS manifest (no iframe embed exists). hlsUrl: platform === "kick" ? meta.manifest_url : undefined, // archive.org and BitChute play the file itself in a native