// A PODCAST CHANNEL'S RECORDS, COMPLETED FROM ITS RSS FEED. // // An episode imported one by one by its enclosure URL (`import-video` with a // direct .mp3 link) goes through yt-dlp's generic extractor, which knows // nothing but the file: the record's title is the file name, and it has no // upload_date, no description and no duration. The index skips a record with // no upload_date, and every surface shows the file id as its title. The feed // the episode came from has all four; this reads them back. // // ONE FETCH, OF ONE URL — the channel's configured url, or the one the caller // names — and nothing else on the network: no enclosure is requested, no media // is downloaded. The editor's job takes its turn on the channel's download // queue (`downloadQueueKey`), behind whatever else is fetching from that host. // // WHICH RECORDS. A record is COMPLETE — and left alone — when it has an // 8-digit upload_date and a title that is not a placeholder (the dir name, its // stem, or yt-dlp's id). Every other record is matched to a feed item: // // guid the record's yt-dlp id (or display_id) is the item's guid — // what yt-dlp itself records when it reads an episode out of the // feed (it forces the guid as the id); // enclosure one of the record's URLs (webpage_url, original_url, url) is the // item's enclosure URL, compared without its fragment, then // without its query; // file name the record's dir name is the enclosure's file name, as the // import named it (`extractVideoId` of the URL). // // The first rule that finds exactly ONE item wins. A rule that finds several // (a feed listing the same file twice) decides nothing; a record no rule // settles is reported as ambiguous or unmatched, never guessed. // // WHAT IS WRITTEN, and only where the record lacks it: the title, the // description, the duration (a value measured from the media beats the // feed's declared one, so an existing duration stays), the upload_date and // timestamp (UTC, from pubDate) — and webpage_url, under one rule below. // Through `patchMetadataInfo`, the one non-yt-dlp writer of // metadata.info.json, so every change is in `metadata.history.json` as // `feed-backfill`. // // WEBPAGE_URL IS THE VIDEO'S ID, NOT A LINK. The snapshot reconciles every // video dir to `extractVideoId(webpage_url)` (reconcileVideoDirs.ts) and a // re-download fetches it (undownloadedVideos.ts). So the item's — an // episode page, sometimes the show's page shared by every item — is written // only when its id IS the record's dir name, else the enclosure URL under the // same test, else nothing: a page link that renamed or merged the dir would // cost the record, not decorate it. import path from "node:path"; import { getPaths, type Paths } from "../lib/paths"; import { readJsonFile } from "../lib/jsonFile-server"; import { mapConcurrent } from "../lib/concurrency"; import { assertChannelTextReadable } from "../lib/channelMedia"; import { extractVideoId } from "../lib/videoId"; import { PROJECT_NAME, PROJECT_URL } from "../lib/project"; import { patchMetadataInfo } from "../lib/metadataHistory-server"; import { feedDateToTimestamp, feedDateToUploadDate, parseRssFeed, type FeedItem, } from "../lib/rssFeed"; import { readChannelConfig } from "./channels"; import { listChannelVideoIds } from "./keptVideos"; const INFO_JSON = "metadata.info.json"; const READ_CONCURRENCY = 16; // One request, but not one that may hang a queue: a feed that has not // answered in this long is not going to. const FEED_FETCH_TIMEOUT_MS = 60_000; type Info = Record; export type FeedMatchRule = "guid" | "enclosure" | "file-name"; export type FeedMetadataPatch = { title?: string; description?: string; duration?: number; webpage_url?: string; timestamp?: number; upload_date?: string; }; export type FeedRecord = { id: string; info: Info | null }; export type FeedBackfillMatch = { id: string; by: FeedMatchRule; item: FeedItem; patch: FeedMetadataPatch; }; export type FeedBackfillPlan = { complete: string[]; matched: FeedBackfillMatch[]; unmatched: string[]; ambiguous: { id: string; rule: FeedMatchRule; candidates: number }[]; // A video dir with no readable metadata.info.json: nothing to complete. noMetadata: string[]; }; // ── pure: what a record lacks ───────────────────────────────────────────── function str(v: unknown): string | null { return typeof v === "string" && v.trim() ? v.trim() : null; } function stem(name: string): string { const dot = name.lastIndexOf("."); return dot > 0 ? name.slice(0, dot) : name; } // A title that only restates the file: what the generic extractor writes for a // direct media URL. export function isPlaceholderTitle(id: string, info: Info): boolean { const title = str(info.title); if (!title) return true; const fileNames = [id, str(info.webpage_url_basename)].filter( (s): s is string => s !== null, ); const placeholders = new Set([ ...fileNames, ...fileNames.map(stem), ...[str(info.id), str(info.display_id)].filter((s): s is string => s !== null), ]); return placeholders.has(title); } function hasUploadDate(info: Info): boolean { return typeof info.upload_date === "string" && /^\d{8}$/.test(info.upload_date); } export function isRecordComplete(id: string, info: Info): boolean { return hasUploadDate(info) && !isPlaceholderTitle(id, info); } // ── pure: matching ──────────────────────────────────────────────────────── function urlKeys(raw: string | null): string[] { if (!raw) return []; try { const u = new URL(raw); u.hash = ""; const full = u.href; u.search = ""; return full === u.href ? [full] : [full, u.href]; } catch { return []; } } function safeDecode(s: string): string { try { return decodeURIComponent(s); } catch { return s; } } function fileNameKeys(raw: string | null): string[] { if (!raw) return []; const keys = new Set([raw, safeDecode(raw)]); return [...keys]; } type FeedIndex = Record>; function add(map: Map, key: string, item: FeedItem): void { const list = map.get(key); if (!list) map.set(key, [item]); else if (!list.includes(item)) list.push(item); } export function indexFeedItems(items: readonly FeedItem[]): FeedIndex { const index: FeedIndex = { guid: new Map(), enclosure: new Map(), "file-name": new Map(), }; for (const item of items) { if (item.guid) add(index.guid, item.guid, item); for (const k of urlKeys(item.enclosureUrl)) add(index.enclosure, k, item); const fileName = item.enclosureUrl ? extractVideoId(item.enclosureUrl) : null; for (const k of fileNameKeys(fileName)) add(index["file-name"], k, item); } return index; } const RULES: readonly FeedMatchRule[] = ["guid", "enclosure", "file-name"]; function recordKeys(id: string, info: Info, rule: FeedMatchRule): string[] { switch (rule) { case "guid": return [str(info.id), str(info.display_id)].filter( (s): s is string => s !== null, ); case "enclosure": return [info.webpage_url, info.original_url, info.url].flatMap((u) => urlKeys(str(u)), ); case "file-name": return [id, str(info.webpage_url_basename)].flatMap((n) => fileNameKeys(n)); } } export type RecordMatch = | { kind: "matched"; item: FeedItem; by: FeedMatchRule } | { kind: "ambiguous"; rule: FeedMatchRule; candidates: number } | { kind: "unmatched" }; export function matchRecord(id: string, info: Info, index: FeedIndex): RecordMatch { let ambiguous: RecordMatch | null = null; for (const rule of RULES) { const found = new Set(); for (const key of recordKeys(id, info, rule)) { for (const item of index[rule].get(key) ?? []) found.add(item); } if (found.size === 1) return { kind: "matched", item: [...found][0], by: rule }; if (found.size > 1 && !ambiguous) { ambiguous = { kind: "ambiguous", rule, candidates: found.size }; } } return ambiguous ?? { kind: "unmatched" }; } // ── pure: what to write ─────────────────────────────────────────────────── // The fields a matched record lacks, from its item. Key order is the order // they are appended to a file that lacks them: the long description before the // date, so the date stays in the tail recencyIndex.ts reads. export function feedPatchFor( id: string, info: Info, item: FeedItem, ): FeedMetadataPatch { const patch: FeedMetadataPatch = {}; if (item.title && isPlaceholderTitle(id, info)) patch.title = item.title; if (item.description && !str(info.description)) { patch.description = item.description; } const duration = info.duration; if ( item.durationSeconds !== null && item.durationSeconds > 0 && !(typeof duration === "number" && duration > 0) ) { patch.duration = item.durationSeconds; } // See the header: a URL here must keep the record's id. const webpage = [item.link, item.enclosureUrl].find( (u): u is string => !!u && extractVideoId(u) === id, ); if (webpage && webpage !== info.webpage_url) patch.webpage_url = webpage; if (!hasUploadDate(info)) { const uploadDate = feedDateToUploadDate(item.pubDate); const timestamp = feedDateToTimestamp(item.pubDate); if (uploadDate) { if (timestamp !== null && typeof info.timestamp !== "number") { patch.timestamp = timestamp; } patch.upload_date = uploadDate; } } return patch; } export function planFeedBackfill( records: readonly FeedRecord[], items: readonly FeedItem[], ): FeedBackfillPlan { const index = indexFeedItems(items); const plan: FeedBackfillPlan = { complete: [], matched: [], unmatched: [], ambiguous: [], noMetadata: [], }; for (const { id, info } of records) { if (!info) { plan.noMetadata.push(id); continue; } if (isRecordComplete(id, info)) { plan.complete.push(id); continue; } const m = matchRecord(id, info, index); if (m.kind === "matched") { plan.matched.push({ id, by: m.by, item: m.item, patch: feedPatchFor(id, info, m.item) }); } else if (m.kind === "ambiguous") { plan.ambiguous.push({ id, rule: m.rule, candidates: m.candidates }); } else { plan.unmatched.push(id); } } return plan; } // ── the fetch ───────────────────────────────────────────────────────────── export type FetchFeed = (url: string, signal?: AbortSignal) => Promise; // The one request. Identified, bounded in time, and never retried here: a // failure is the job's failure, and the operator re-runs it. export const fetchFeedText: FetchFeed = async (url, signal) => { const timeout = AbortSignal.timeout(FEED_FETCH_TIMEOUT_MS); const res = await fetch(url, { signal: signal ? AbortSignal.any([signal, timeout]) : timeout, redirect: "follow", headers: { accept: "application/rss+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.1", "user-agent": `${PROJECT_NAME} feed backfill (+${PROJECT_URL})`, }, }); if (!res.ok) { throw new Error(`the feed answered HTTP ${res.status} ${res.statusText}`.trim()); } return res.text(); }; // ── the run ─────────────────────────────────────────────────────────────── export type FeedBackfillResult = { slug: string; feedUrl: string; dryRun: boolean; feedItems: number; records: number; complete: number; matched: number; written: number; unchanged: number; unmatched: string[]; ambiguous: string[]; noMetadata: number; failed: { id: string; error: string }[]; }; export type FeedBackfillOpts = { paths?: Paths; slug: string; // Default: the channel's configured url. feedUrl?: string; dryRun?: boolean; // Who asked, for the history entry ("cli", "ops", …). requestedBy?: string; onLog?: (line: string) => void; signal?: AbortSignal; // The fetch, injectable so a test never touches the network. fetchFeed?: FetchFeed; }; function describePatch(p: FeedMetadataPatch): string { const parts: string[] = []; if (p.title !== undefined) parts.push(`title ${JSON.stringify(p.title)}`); if (p.upload_date !== undefined) parts.push(`date ${p.upload_date}`); if (p.duration !== undefined) parts.push(`duration ${p.duration}s`); if (p.description !== undefined) parts.push(`description (${p.description.length} chars)`); if (p.webpage_url !== undefined) parts.push("webpage_url"); return parts.join(", ") || "nothing it lacks"; } export async function backfillFeedMetadata( opts: FeedBackfillOpts, ): Promise { const paths = opts.paths ?? getPaths(); const { slug } = opts; const log = opts.onLog ?? (() => {}); const dryRun = opts.dryRun === true; const config = await readChannelConfig(paths, slug); if (!config) throw new Error(`Channel "${slug}" not found`); const feedUrl = (opts.feedUrl ?? config.url ?? "").trim(); if (!/^https?:\/\//i.test(feedUrl)) { throw new Error( feedUrl ? `"${feedUrl}" is not an http(s) feed URL` : `Channel "${slug}" has no url; pass the feed URL`, ); } // AN UNREADABLE TEXT TIER IS NOT AN EMPTY CHANNEL (AGENTS.md): the walk below // would find no records and report a clean run. await assertChannelTextReadable(paths, slug, config); const dataDir = path.join(paths.channelsDir, slug, "data"); const ids = await listChannelVideoIds(paths, slug); const records = await mapConcurrent(ids, READ_CONCURRENCY, async (id) => { const read = await readJsonFile(path.join(dataDir, id, INFO_JSON)); const value = read.ok ? read.value : null; const info = value && typeof value === "object" && !Array.isArray(value) ? (value as Info) : null; return { id, info }; }); const incomplete = records.filter( (r) => r.info && !isRecordComplete(r.id, r.info), ).length; log( `${slug}: ${records.length} record(s), ${incomplete} lacking a title or a date.`, ); const result: FeedBackfillResult = { slug, feedUrl, dryRun, feedItems: 0, records: records.length, complete: 0, matched: 0, written: 0, unchanged: 0, unmatched: [], ambiguous: [], noMetadata: 0, failed: [], }; if (incomplete === 0) { // Nothing to complete is nothing to fetch: the feed is not asked. result.complete = records.filter((r) => r.info).length; result.noMetadata = records.length - result.complete; log("Every record has a title and a date; the feed was not fetched."); return result; } log(`Fetching the feed (one request): ${feedUrl}`); const xml = await (opts.fetchFeed ?? fetchFeedText)(feedUrl, opts.signal); const feed = parseRssFeed(xml); if (feed.items.length === 0) { throw new Error("the feed has no — is this URL an RSS feed?"); } result.feedItems = feed.items.length; log( `Feed${feed.title ? ` "${feed.title}"` : ""}: ${feed.items.length} item(s).`, ); const plan = planFeedBackfill(records, feed.items); result.complete = plan.complete.length; result.matched = plan.matched.length; result.unmatched = plan.unmatched; result.ambiguous = plan.ambiguous.map((a) => a.id); result.noMetadata = plan.noMetadata.length; for (const m of plan.matched) { if (opts.signal?.aborted) break; const what = describePatch(m.patch); if (dryRun) { log(` would write ${m.id} (by ${m.by}): ${what}`); continue; } if (Object.keys(m.patch).length === 0) { result.unchanged += 1; log(` ${m.id} (by ${m.by}): the item has nothing it lacks`); continue; } try { const { written } = await patchMetadataInfo( path.join(dataDir, m.id), { ...m.patch }, { by: "feed-backfill", ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), onLog: log, }, ); if (written) { result.written += 1; log(` wrote ${m.id} (by ${m.by}): ${what}`); } else { result.unchanged += 1; } } catch (err) { result.failed.push({ id: m.id, error: (err as Error).message }); log(` FAILED ${m.id}: ${(err as Error).message}`); } } for (const a of plan.ambiguous) { log(` ambiguous ${a.id}: ${a.candidates} items share its ${a.rule}`); } for (const id of plan.unmatched) log(` unmatched ${id}`); return result; } export function feedBackfillSummary(r: FeedBackfillResult): string { const head = r.dryRun ? "Dry run" : "Done"; const parts = [ `${r.matched} matched`, `${r.unmatched.length} unmatched`, `${r.complete} already complete`, ]; if (r.ambiguous.length) parts.push(`${r.ambiguous.length} ambiguous`); if (r.noMetadata) parts.push(`${r.noMetadata} without metadata`); if (!r.dryRun) { parts.push(`${r.written} written`); if (r.unchanged) parts.push(`${r.unchanged} unchanged`); if (r.failed.length) parts.push(`${r.failed.length} failed`); } let line = `${head}: ${r.slug} — ${parts.join(", ")} (of ${r.records} record(s)`; line += r.feedItems ? `, ${r.feedItems} feed item(s)).` : ")."; if (!r.dryRun && r.written > 0) { line += " Rebuild the index for the new titles and dates to reach search and the sites."; } return line; }