// A PODCAST CHANNEL'S RECORDS, COMPLETED FROM ITS RSS FEED.
//
// An episode imported one by one by its enclosure URL (`import-video` with a
// direct .mp3 link) goes through yt-dlp's generic extractor, which knows
// nothing but the file: the record's title is the file name, and it has no
// upload_date, no description and no duration. The index skips a record with
// no upload_date, and every surface shows the file id as its title. The feed
// the episode came from has all four; this reads them back.
//
// ONE FETCH, OF ONE URL — the channel's configured url, or the one the caller
// names — and nothing else on the network: no enclosure is requested, no media
// is downloaded. The editor's job takes its turn on the channel's download
// queue (`downloadQueueKey`), behind whatever else is fetching from that host.
//
// WHICH RECORDS. A record is COMPLETE — and left alone — when it has an
// 8-digit upload_date and a title that is not a placeholder (the dir name, its
// stem, or yt-dlp's id). Every other record is matched to a feed item:
//
// guid the record's yt-dlp id (or display_id) is the item's guid —
// what yt-dlp itself records when it reads an episode out of the
// feed (it forces the guid as the id);
// enclosure one of the record's URLs (webpage_url, original_url, url) is the
// item's enclosure URL, compared without its fragment, then
// without its query;
// file name the record's dir name is the enclosure's file name, as the
// import named it (`extractVideoId` of the URL).
//
// The first rule that finds exactly ONE item wins. A rule that finds several
// (a feed listing the same file twice) decides nothing; a record no rule
// settles is reported as ambiguous or unmatched, never guessed.
//
// WHAT IS WRITTEN, and only where the record lacks it: the title, the
// description, the duration (a value measured from the media beats the
// feed's declared one, so an existing duration stays), the upload_date and
// timestamp (UTC, from pubDate) — and webpage_url, under one rule below.
// Through `patchMetadataInfo`, the one non-yt-dlp writer of
// metadata.info.json, so every change is in `metadata.history.json` as
// `feed-backfill`.
//
// WEBPAGE_URL IS THE VIDEO'S ID, NOT A LINK. The snapshot reconciles every
// video dir to `extractVideoId(webpage_url)` (reconcileVideoDirs.ts) and a
// re-download fetches it (undownloadedVideos.ts). So the item's — an
// episode page, sometimes the show's page shared by every item — is written
// only when its id IS the record's dir name, else the enclosure URL under the
// same test, else nothing: a page link that renamed or merged the dir would
// cost the record, not decorate it.
import path from "node:path";
import { getPaths, type Paths } from "../lib/paths";
import { readJsonFile } from "../lib/jsonFile-server";
import { mapConcurrent } from "../lib/concurrency";
import { assertChannelTextReadable } from "../lib/channelMedia";
import { extractVideoId } from "../lib/videoId";
import { PROJECT_NAME, PROJECT_URL } from "../lib/project";
import { patchMetadataInfo } from "../lib/metadataHistory-server";
import {
feedDateToTimestamp,
feedDateToUploadDate,
parseRssFeed,
type FeedItem,
} from "../lib/rssFeed";
import { readChannelConfig } from "./channels";
import { listChannelVideoIds } from "./keptVideos";
const INFO_JSON = "metadata.info.json";
const READ_CONCURRENCY = 16;
// One request, but not one that may hang a queue: a feed that has not
// answered in this long is not going to.
const FEED_FETCH_TIMEOUT_MS = 60_000;
type Info = Record;
export type FeedMatchRule = "guid" | "enclosure" | "file-name";
export type FeedMetadataPatch = {
title?: string;
description?: string;
duration?: number;
webpage_url?: string;
timestamp?: number;
upload_date?: string;
};
export type FeedRecord = { id: string; info: Info | null };
export type FeedBackfillMatch = {
id: string;
by: FeedMatchRule;
item: FeedItem;
patch: FeedMetadataPatch;
};
export type FeedBackfillPlan = {
complete: string[];
matched: FeedBackfillMatch[];
unmatched: string[];
ambiguous: { id: string; rule: FeedMatchRule; candidates: number }[];
// A video dir with no readable metadata.info.json: nothing to complete.
noMetadata: string[];
};
// ── pure: what a record lacks ─────────────────────────────────────────────
function str(v: unknown): string | null {
return typeof v === "string" && v.trim() ? v.trim() : null;
}
function stem(name: string): string {
const dot = name.lastIndexOf(".");
return dot > 0 ? name.slice(0, dot) : name;
}
// A title that only restates the file: what the generic extractor writes for a
// direct media URL.
export function isPlaceholderTitle(id: string, info: Info): boolean {
const title = str(info.title);
if (!title) return true;
const fileNames = [id, str(info.webpage_url_basename)].filter(
(s): s is string => s !== null,
);
const placeholders = new Set([
...fileNames,
...fileNames.map(stem),
...[str(info.id), str(info.display_id)].filter((s): s is string => s !== null),
]);
return placeholders.has(title);
}
function hasUploadDate(info: Info): boolean {
return typeof info.upload_date === "string" && /^\d{8}$/.test(info.upload_date);
}
export function isRecordComplete(id: string, info: Info): boolean {
return hasUploadDate(info) && !isPlaceholderTitle(id, info);
}
// ── pure: matching ────────────────────────────────────────────────────────
function urlKeys(raw: string | null): string[] {
if (!raw) return [];
try {
const u = new URL(raw);
u.hash = "";
const full = u.href;
u.search = "";
return full === u.href ? [full] : [full, u.href];
} catch {
return [];
}
}
function safeDecode(s: string): string {
try {
return decodeURIComponent(s);
} catch {
return s;
}
}
function fileNameKeys(raw: string | null): string[] {
if (!raw) return [];
const keys = new Set([raw, safeDecode(raw)]);
return [...keys];
}
type FeedIndex = Record>;
function add(map: Map, key: string, item: FeedItem): void {
const list = map.get(key);
if (!list) map.set(key, [item]);
else if (!list.includes(item)) list.push(item);
}
export function indexFeedItems(items: readonly FeedItem[]): FeedIndex {
const index: FeedIndex = {
guid: new Map(),
enclosure: new Map(),
"file-name": new Map(),
};
for (const item of items) {
if (item.guid) add(index.guid, item.guid, item);
for (const k of urlKeys(item.enclosureUrl)) add(index.enclosure, k, item);
const fileName = item.enclosureUrl ? extractVideoId(item.enclosureUrl) : null;
for (const k of fileNameKeys(fileName)) add(index["file-name"], k, item);
}
return index;
}
const RULES: readonly FeedMatchRule[] = ["guid", "enclosure", "file-name"];
function recordKeys(id: string, info: Info, rule: FeedMatchRule): string[] {
switch (rule) {
case "guid":
return [str(info.id), str(info.display_id)].filter(
(s): s is string => s !== null,
);
case "enclosure":
return [info.webpage_url, info.original_url, info.url].flatMap((u) =>
urlKeys(str(u)),
);
case "file-name":
return [id, str(info.webpage_url_basename)].flatMap((n) => fileNameKeys(n));
}
}
export type RecordMatch =
| { kind: "matched"; item: FeedItem; by: FeedMatchRule }
| { kind: "ambiguous"; rule: FeedMatchRule; candidates: number }
| { kind: "unmatched" };
export function matchRecord(id: string, info: Info, index: FeedIndex): RecordMatch {
let ambiguous: RecordMatch | null = null;
for (const rule of RULES) {
const found = new Set();
for (const key of recordKeys(id, info, rule)) {
for (const item of index[rule].get(key) ?? []) found.add(item);
}
if (found.size === 1) return { kind: "matched", item: [...found][0], by: rule };
if (found.size > 1 && !ambiguous) {
ambiguous = { kind: "ambiguous", rule, candidates: found.size };
}
}
return ambiguous ?? { kind: "unmatched" };
}
// ── pure: what to write ───────────────────────────────────────────────────
// The fields a matched record lacks, from its item. Key order is the order
// they are appended to a file that lacks them: the long description before the
// date, so the date stays in the tail recencyIndex.ts reads.
export function feedPatchFor(
id: string,
info: Info,
item: FeedItem,
): FeedMetadataPatch {
const patch: FeedMetadataPatch = {};
if (item.title && isPlaceholderTitle(id, info)) patch.title = item.title;
if (item.description && !str(info.description)) {
patch.description = item.description;
}
const duration = info.duration;
if (
item.durationSeconds !== null &&
item.durationSeconds > 0 &&
!(typeof duration === "number" && duration > 0)
) {
patch.duration = item.durationSeconds;
}
// See the header: a URL here must keep the record's id.
const webpage = [item.link, item.enclosureUrl].find(
(u): u is string => !!u && extractVideoId(u) === id,
);
if (webpage && webpage !== info.webpage_url) patch.webpage_url = webpage;
if (!hasUploadDate(info)) {
const uploadDate = feedDateToUploadDate(item.pubDate);
const timestamp = feedDateToTimestamp(item.pubDate);
if (uploadDate) {
if (timestamp !== null && typeof info.timestamp !== "number") {
patch.timestamp = timestamp;
}
patch.upload_date = uploadDate;
}
}
return patch;
}
export function planFeedBackfill(
records: readonly FeedRecord[],
items: readonly FeedItem[],
): FeedBackfillPlan {
const index = indexFeedItems(items);
const plan: FeedBackfillPlan = {
complete: [],
matched: [],
unmatched: [],
ambiguous: [],
noMetadata: [],
};
for (const { id, info } of records) {
if (!info) {
plan.noMetadata.push(id);
continue;
}
if (isRecordComplete(id, info)) {
plan.complete.push(id);
continue;
}
const m = matchRecord(id, info, index);
if (m.kind === "matched") {
plan.matched.push({ id, by: m.by, item: m.item, patch: feedPatchFor(id, info, m.item) });
} else if (m.kind === "ambiguous") {
plan.ambiguous.push({ id, rule: m.rule, candidates: m.candidates });
} else {
plan.unmatched.push(id);
}
}
return plan;
}
// ── the fetch ─────────────────────────────────────────────────────────────
export type FetchFeed = (url: string, signal?: AbortSignal) => Promise;
// The one request. Identified, bounded in time, and never retried here: a
// failure is the job's failure, and the operator re-runs it.
export const fetchFeedText: FetchFeed = async (url, signal) => {
const timeout = AbortSignal.timeout(FEED_FETCH_TIMEOUT_MS);
const res = await fetch(url, {
signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
redirect: "follow",
headers: {
accept:
"application/rss+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.1",
"user-agent": `${PROJECT_NAME} feed backfill (+${PROJECT_URL})`,
},
});
if (!res.ok) {
throw new Error(`the feed answered HTTP ${res.status} ${res.statusText}`.trim());
}
return res.text();
};
// ── the run ───────────────────────────────────────────────────────────────
export type FeedBackfillResult = {
slug: string;
feedUrl: string;
dryRun: boolean;
feedItems: number;
records: number;
complete: number;
matched: number;
written: number;
unchanged: number;
unmatched: string[];
ambiguous: string[];
noMetadata: number;
failed: { id: string; error: string }[];
};
export type FeedBackfillOpts = {
paths?: Paths;
slug: string;
// Default: the channel's configured url.
feedUrl?: string;
dryRun?: boolean;
// Who asked, for the history entry ("cli", "ops", …).
requestedBy?: string;
onLog?: (line: string) => void;
signal?: AbortSignal;
// The fetch, injectable so a test never touches the network.
fetchFeed?: FetchFeed;
};
function describePatch(p: FeedMetadataPatch): string {
const parts: string[] = [];
if (p.title !== undefined) parts.push(`title ${JSON.stringify(p.title)}`);
if (p.upload_date !== undefined) parts.push(`date ${p.upload_date}`);
if (p.duration !== undefined) parts.push(`duration ${p.duration}s`);
if (p.description !== undefined) parts.push(`description (${p.description.length} chars)`);
if (p.webpage_url !== undefined) parts.push("webpage_url");
return parts.join(", ") || "nothing it lacks";
}
export async function backfillFeedMetadata(
opts: FeedBackfillOpts,
): Promise {
const paths = opts.paths ?? getPaths();
const { slug } = opts;
const log = opts.onLog ?? (() => {});
const dryRun = opts.dryRun === true;
const config = await readChannelConfig(paths, slug);
if (!config) throw new Error(`Channel "${slug}" not found`);
const feedUrl = (opts.feedUrl ?? config.url ?? "").trim();
if (!/^https?:\/\//i.test(feedUrl)) {
throw new Error(
feedUrl
? `"${feedUrl}" is not an http(s) feed URL`
: `Channel "${slug}" has no url; pass the feed URL`,
);
}
// AN UNREADABLE TEXT TIER IS NOT AN EMPTY CHANNEL (AGENTS.md): the walk below
// would find no records and report a clean run.
await assertChannelTextReadable(paths, slug, config);
const dataDir = path.join(paths.channelsDir, slug, "data");
const ids = await listChannelVideoIds(paths, slug);
const records = await mapConcurrent(ids, READ_CONCURRENCY, async (id) => {
const read = await readJsonFile(path.join(dataDir, id, INFO_JSON));
const value = read.ok ? read.value : null;
const info =
value && typeof value === "object" && !Array.isArray(value)
? (value as Info)
: null;
return { id, info };
});
const incomplete = records.filter(
(r) => r.info && !isRecordComplete(r.id, r.info),
).length;
log(
`${slug}: ${records.length} record(s), ${incomplete} lacking a title or a date.`,
);
const result: FeedBackfillResult = {
slug,
feedUrl,
dryRun,
feedItems: 0,
records: records.length,
complete: 0,
matched: 0,
written: 0,
unchanged: 0,
unmatched: [],
ambiguous: [],
noMetadata: 0,
failed: [],
};
if (incomplete === 0) {
// Nothing to complete is nothing to fetch: the feed is not asked.
result.complete = records.filter((r) => r.info).length;
result.noMetadata = records.length - result.complete;
log("Every record has a title and a date; the feed was not fetched.");
return result;
}
log(`Fetching the feed (one request): ${feedUrl}`);
const xml = await (opts.fetchFeed ?? fetchFeedText)(feedUrl, opts.signal);
const feed = parseRssFeed(xml);
if (feed.items.length === 0) {
throw new Error("the feed has no — is this URL an RSS feed?");
}
result.feedItems = feed.items.length;
log(
`Feed${feed.title ? ` "${feed.title}"` : ""}: ${feed.items.length} item(s).`,
);
const plan = planFeedBackfill(records, feed.items);
result.complete = plan.complete.length;
result.matched = plan.matched.length;
result.unmatched = plan.unmatched;
result.ambiguous = plan.ambiguous.map((a) => a.id);
result.noMetadata = plan.noMetadata.length;
for (const m of plan.matched) {
if (opts.signal?.aborted) break;
const what = describePatch(m.patch);
if (dryRun) {
log(` would write ${m.id} (by ${m.by}): ${what}`);
continue;
}
if (Object.keys(m.patch).length === 0) {
result.unchanged += 1;
log(` ${m.id} (by ${m.by}): the item has nothing it lacks`);
continue;
}
try {
const { written } = await patchMetadataInfo(
path.join(dataDir, m.id),
{ ...m.patch },
{
by: "feed-backfill",
...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}),
onLog: log,
},
);
if (written) {
result.written += 1;
log(` wrote ${m.id} (by ${m.by}): ${what}`);
} else {
result.unchanged += 1;
}
} catch (err) {
result.failed.push({ id: m.id, error: (err as Error).message });
log(` FAILED ${m.id}: ${(err as Error).message}`);
}
}
for (const a of plan.ambiguous) {
log(` ambiguous ${a.id}: ${a.candidates} items share its ${a.rule}`);
}
for (const id of plan.unmatched) log(` unmatched ${id}`);
return result;
}
export function feedBackfillSummary(r: FeedBackfillResult): string {
const head = r.dryRun ? "Dry run" : "Done";
const parts = [
`${r.matched} matched`,
`${r.unmatched.length} unmatched`,
`${r.complete} already complete`,
];
if (r.ambiguous.length) parts.push(`${r.ambiguous.length} ambiguous`);
if (r.noMetadata) parts.push(`${r.noMetadata} without metadata`);
if (!r.dryRun) {
parts.push(`${r.written} written`);
if (r.unchanged) parts.push(`${r.unchanged} unchanged`);
if (r.failed.length) parts.push(`${r.failed.length} failed`);
}
let line = `${head}: ${r.slug} — ${parts.join(", ")} (of ${r.records} record(s)`;
line += r.feedItems ? `, ${r.feedItems} feed item(s)).` : ").";
if (!r.dryRun && r.written > 0) {
line += " Rebuild the index for the new titles and dates to reach search and the sites.";
}
return line;
}