commit 3751f19e76a1a186483fb18482cc9dc819c59b6b
parent c8668fe58c84cd97129c41d13f02918e7b444e6f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:22:44 -0400
common: backfill a podcast channel's record metadata from its RSS feed
One fetch of the feed, then each record lacking a date or a real title is
matched to an item by guid, enclosure URL or enclosure file name (the first
rule that finds exactly one item wins) and completed where it lacks a title,
description, duration, upload_date and timestamp. Written through
patchMetadataInfo, the one writer of metadata.info.json that is not yt-dlp,
which merges in place and records the rewrite in metadata.history.json as
feed-backfill. webpage_url is only set to a URL whose id is the record's dir
name, since the snapshot reconciles every dir to that id. A new job kind,
feed-metadata, on the channel's download queue.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 1065 insertions(+), 2 deletions(-)
diff --git a/common/controller/feedMetadataBackfill.test.ts b/common/controller/feedMetadataBackfill.test.ts
@@ -0,0 +1,363 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { readFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import type { Paths } from "../lib/paths";
+import type { FeedItem } from "../lib/rssFeed";
+import { loadMetadataHistory } from "../lib/metadataHistory-server";
+import {
+ backfillFeedMetadata,
+ feedBackfillSummary,
+ feedPatchFor,
+ isPlaceholderTitle,
+ isRecordComplete,
+ planFeedBackfill,
+ type FetchFeed,
+} from "./feedMetadataBackfill";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/feedMetadataBackfill.test.ts
+//
+// No network: the fetch is a fake that serves the fixture feed and counts.
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FEED_XML = readFileSync(
+ path.join(HERE, "..", "lib", "__fixtures__", "demo-podcast.rss"),
+ "utf8",
+);
+const FEED_URL = "https://feeds.example.com/demo.rss";
+const SLUG = "demo-channel";
+
+// What the generic extractor writes for a direct .mp3 URL: the file name as
+// the title, no date. Key order as yt-dlp writes it.
+function genericInfo(fileName: string, extra: Record<string, unknown> = {}) {
+ return {
+ id: `uuid-${fileName}`,
+ title: fileName.replace(/\.mp3$/, ""),
+ direct: true,
+ formats: [{ format_id: "mpeg", url: `https://cdn.example.com/audio/${fileName}?key=k` }],
+ webpage_url: `https://cdn.example.com/audio/${fileName}?key=k`,
+ webpage_url_basename: fileName,
+ extractor: "generic",
+ _version: { version: "2026.01.01" },
+ ...extra,
+ };
+}
+
+type Seed = Record<string, Record<string, unknown> | null>;
+
+async function withCorpus(
+ seed: Seed,
+ fn: (paths: Paths, dataDir: string) => Promise<void>,
+): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-feed-"));
+ const transcriptsDir = path.join(dir, "corpus");
+ const paths = {
+ transcriptsDir,
+ channelsDir: path.join(transcriptsDir, "channels"),
+ } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ const dataDir = path.join(channelDir, "data");
+ await mkdir(dataDir, { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Demo", url: FEED_URL, handling: "transcribe" }),
+ );
+ for (const [id, info] of Object.entries(seed)) {
+ await mkdir(path.join(dataDir, id), { recursive: true });
+ if (info) {
+ // yt-dlp's own separators, so a byte comparison means something.
+ await writeFile(
+ path.join(dataDir, id, "metadata.info.json"),
+ JSON.stringify(info),
+ );
+ }
+ }
+ try {
+ await fn(paths, dataDir);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function fakeFetch(): FetchFeed & { calls: string[] } {
+ const calls: string[] = [];
+ const f = (async (url: string) => {
+ calls.push(url);
+ return FEED_XML;
+ }) as FetchFeed & { calls: string[] };
+ f.calls = calls;
+ return f;
+}
+
+const readInfo = async (dataDir: string, id: string) =>
+ JSON.parse(await readFile(path.join(dataDir, id, "metadata.info.json"), "utf8")) as Record<
+ string,
+ unknown
+ >;
+
+const SEED: Seed = {
+ // Matched by its enclosure URL: same path, a different signed query.
+ "abc123.mp3": genericInfo("abc123.mp3"),
+ // Matched by guid: yt-dlp forced the feed's guid as the id.
+ "def456.mp3": genericInfo("def456.mp3", {
+ id: "guid-0002",
+ webpage_url: "https://cdn2.example.org/def456.mp3?key=q",
+ }),
+ // Matched by file name only, and already carries a measured duration.
+ "ghi789.mp3": genericInfo("ghi789.mp3", {
+ webpage_url: "https://mirror.example.org/x/ghi789.mp3",
+ duration: 3700,
+ }),
+ "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" },
+ "zzz999.mp3": genericInfo("zzz999.mp3"),
+ "nometa.mp3": null,
+};
+
+test("a run completes every matched record, through the history, with one fetch", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const fetchFeed = fakeFetch();
+ const lines: string[] = [];
+ const r = await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed,
+ requestedBy: "test",
+ onLog: (l) => lines.push(l),
+ });
+ assert.deepEqual(fetchFeed.calls, [FEED_URL], "the channel's url, once");
+ assert.equal(r.feedItems, 5);
+ assert.equal(r.matched, 3);
+ assert.equal(r.written, 3);
+ assert.equal(r.complete, 1);
+ assert.deepEqual(r.unmatched, ["zzz999.mp3"]);
+ assert.equal(r.noMetadata, 1);
+ assert.deepEqual(r.failed, []);
+
+ const one = await readInfo(dataDir, "abc123.mp3");
+ assert.equal(one.title, "Episode 1: Cats & <Dogs>");
+ assert.equal(one.upload_date, "20240305");
+ assert.equal(one.timestamp, 1709632800);
+ assert.equal(one.duration, 3723);
+ assert.equal(one.description, "First line & more.\nSecond line\nthird");
+ // The item's <link> would rename the dir (its id is "one"), so the
+ // enclosure — whose file name IS the dir name — is the webpage_url.
+ assert.equal(
+ one.webpage_url,
+ "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1",
+ );
+ // Everything else is untouched, in place; new keys are appended with the
+ // date LAST (the tail reader's 8 KB).
+ const keys = Object.keys(one);
+ assert.deepEqual(keys.slice(0, 8), Object.keys(genericInfo("abc123.mp3")));
+ assert.equal(keys.at(-1), "upload_date");
+ assert.deepEqual(one.formats, genericInfo("abc123.mp3").formats);
+
+ const two = await readInfo(dataDir, "def456.mp3");
+ assert.equal(two.title, "Two & a half ’quotes’ \"here\"");
+ assert.equal(two.upload_date, "20240305", "22:30 at -0500 is the next UTC day");
+ assert.equal(two.description, "Hello & goodbye");
+
+ const three = await readInfo(dataDir, "ghi789.mp3");
+ assert.equal(three.title, "Episode three");
+ assert.equal(three.duration, 3700, "a measured duration is kept");
+ assert.equal(three.upload_date, "20240306");
+
+ // Through the history: one entry per written record, by feed-backfill.
+ const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3"));
+ assert.equal(h?.entries.length, 1);
+ const e = h!.entries[0];
+ assert.equal(e.by, "feed-backfill");
+ assert.equal(e.requestedBy, "test");
+ assert.deepEqual(e.changed.title, { from: "abc123", to: "Episode 1: Cats & <Dogs>" });
+ assert.equal(e.added.upload_date, "20240305");
+
+ // The complete and the unmatched records are not touched.
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "done.mp3")), null);
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "zzz999.mp3")), null);
+ assert.match(feedBackfillSummary(r), /3 matched, 1 unmatched, 1 already complete/);
+ assert.ok(lines.some((l) => l.includes("unmatched zzz999.mp3")));
+ });
+});
+
+test("a second run finds the records complete and still asks the feed only for the rest", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: fakeFetch() });
+ const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ const fetchFeed = fakeFetch();
+ const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed });
+ assert.equal(r.complete, 4);
+ assert.equal(r.written, 0);
+ assert.deepEqual(r.unmatched, ["zzz999.mp3"]);
+ const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ assert.deepEqual(after, before, "a complete record is not rewritten");
+ const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3"));
+ assert.equal(h?.entries.length, 1, "and no second history entry");
+ });
+});
+
+test("nothing incomplete: the feed is not fetched at all", async () => {
+ await withCorpus(
+ { "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" } },
+ async (paths) => {
+ const fetchFeed = fakeFetch();
+ const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed });
+ assert.deepEqual(fetchFeed.calls, []);
+ assert.equal(r.complete, 1);
+ assert.equal(r.matched, 0);
+ },
+ );
+});
+
+test("a dry run reports matched / unmatched / complete and writes nothing", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ const lines: string[] = [];
+ const r = await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ dryRun: true,
+ fetchFeed: fakeFetch(),
+ onLog: (l) => lines.push(l),
+ });
+ assert.equal(r.matched, 3);
+ assert.equal(r.unmatched.length, 1);
+ assert.equal(r.complete, 1);
+ assert.equal(r.written, 0);
+ const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ assert.deepEqual(after, before);
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null);
+ assert.ok(lines.some((l) => l.startsWith(" would write abc123.mp3 (by enclosure)")));
+ assert.match(feedBackfillSummary(r), /^Dry run: demo-channel — 3 matched, 1 unmatched, 1 already complete/);
+ });
+});
+
+test("--feed overrides the channel's url; a non-http url and a feed with no items refuse", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const fetchFeed = fakeFetch();
+ await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ feedUrl: "https://mirror.example.net/feed.xml",
+ dryRun: true,
+ fetchFeed,
+ });
+ assert.deepEqual(fetchFeed.calls, ["https://mirror.example.net/feed.xml"]);
+
+ await assert.rejects(
+ backfillFeedMetadata({ paths, slug: SLUG, feedUrl: "file:///etc/passwd", fetchFeed }),
+ /not an http\(s\) feed URL/,
+ );
+ await assert.rejects(
+ backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed: async () => "<html><body>not a feed</body></html>",
+ }),
+ /no <item>/,
+ );
+ await assert.rejects(
+ backfillFeedMetadata({ paths, slug: "no-such-channel", fetchFeed }),
+ /not found/,
+ );
+ // A failed fetch writes nothing.
+ await assert.rejects(
+ backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed: async () => {
+ throw new Error("the feed answered HTTP 503");
+ },
+ }),
+ /503/,
+ );
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null);
+ });
+});
+
+// ── the pure rules ──────────────────────────────────────────────────────────
+
+const item = (over: Partial<FeedItem>): FeedItem => ({
+ title: "T",
+ link: null,
+ guid: null,
+ pubDate: "Tue, 05 Mar 2024 10:00:00 GMT",
+ enclosureUrl: null,
+ durationSeconds: null,
+ description: null,
+ ...over,
+});
+
+test("placeholder titles: the dir name, its stem, the extractor's ids, or none", () => {
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc.mp3" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "uuid-1", id: "uuid-1" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: " " }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", {}), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "A real title" }), false);
+ assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "20240101" }), true);
+ assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "2024" }), false);
+ assert.equal(isRecordComplete("abc.mp3", { title: "abc", upload_date: "20240101" }), false);
+});
+
+test("a rule that finds several items decides nothing; the next unique rule still can", () => {
+ const dupA = item({ guid: "g-a", enclosureUrl: "https://a.example.com/x/same.mp3" });
+ const dupB = item({ guid: "g-b", enclosureUrl: "https://b.example.com/y/same.mp3" });
+ const plan = planFeedBackfill(
+ [
+ { id: "same.mp3", info: { id: "nope", title: "same" } },
+ { id: "same.mp3", info: { id: "g-b", title: "same" } },
+ ],
+ [dupA, dupB],
+ );
+ assert.deepEqual(plan.ambiguous, [{ id: "same.mp3", rule: "file-name", candidates: 2 }]);
+ assert.equal(plan.matched.length, 1);
+ assert.equal(plan.matched[0].by, "guid");
+ assert.equal(plan.matched[0].item, dupB);
+});
+
+test("a URL is compared without its fragment, then without its query", () => {
+ const it = item({ enclosureUrl: "https://cdn.example.com/a/ep.mp3?updated=1" });
+ const plan = planFeedBackfill(
+ [
+ { id: "x1", info: { webpage_url: "https://cdn.example.com/a/ep.mp3?updated=1#__youtubedl_smuggle=1" } },
+ { id: "x2", info: { url: "https://cdn.example.com/a/ep.mp3?key=signed" } },
+ { id: "x3", info: { webpage_url: "https://cdn.example.com/b/ep.mp3" } },
+ ],
+ [it],
+ );
+ assert.deepEqual(plan.matched.map((m) => [m.id, m.by]), [
+ ["x1", "enclosure"],
+ ["x2", "enclosure"],
+ ]);
+ assert.deepEqual(plan.unmatched, ["x3"]);
+});
+
+test("the patch fills only what is missing, and a URL only when it keeps the record's id", () => {
+ const it = item({
+ title: "Real",
+ description: "About it",
+ durationSeconds: 60,
+ link: "https://example.com/episodes/ep.mp3",
+ enclosureUrl: "https://cdn.example.com/ep.mp3",
+ });
+ // The link's id is the dir name: the link wins.
+ assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, it).webpage_url, "https://example.com/episodes/ep.mp3");
+ // A show-page link would rename the dir: the enclosure is used instead.
+ const showLink = { ...it, link: "https://example.com/show" };
+ assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, showLink).webpage_url, "https://cdn.example.com/ep.mp3");
+ // Neither keeps the id: webpage_url is left alone.
+ assert.equal("webpage_url" in feedPatchFor("other.mp3", { title: "other" }, showLink), false);
+ // A record with a real title and description keeps them; only the date goes in.
+ assert.deepEqual(
+ Object.keys(
+ feedPatchFor("ep.mp3", { title: "Mine", description: "Mine too", duration: 59, webpage_url: "https://example.com/episodes/ep.mp3" }, it),
+ ),
+ ["timestamp", "upload_date"],
+ );
+ // An item with an unreadable date adds no date.
+ assert.equal("upload_date" in feedPatchFor("ep.mp3", { title: "ep" }, { ...it, pubDate: "soon" }), false);
+});
diff --git a/common/controller/feedMetadataBackfill.ts b/common/controller/feedMetadataBackfill.ts
@@ -0,0 +1,505 @@
+// A PODCAST CHANNEL'S RECORDS, COMPLETED FROM ITS RSS FEED.
+//
+// An episode imported one by one by its enclosure URL (`import-video` with a
+// direct .mp3 link) goes through yt-dlp's generic extractor, which knows
+// nothing but the file: the record's title is the file name, and it has no
+// upload_date, no description and no duration. The index skips a record with
+// no upload_date, and every surface shows the file id as its title. The feed
+// the episode came from has all four; this reads them back.
+//
+// ONE FETCH, OF ONE URL — the channel's configured url, or the one the caller
+// names — and nothing else on the network: no enclosure is requested, no media
+// is downloaded. The editor's job takes its turn on the channel's download
+// queue (`downloadQueueKey`), behind whatever else is fetching from that host.
+//
+// WHICH RECORDS. A record is COMPLETE — and left alone — when it has an
+// 8-digit upload_date and a title that is not a placeholder (the dir name, its
+// stem, or yt-dlp's id). Every other record is matched to a feed item:
+//
+// guid the record's yt-dlp id (or display_id) is the item's guid —
+// what yt-dlp itself records when it reads an episode out of the
+// feed (it forces the guid as the id);
+// enclosure one of the record's URLs (webpage_url, original_url, url) is the
+// item's enclosure URL, compared without its fragment, then
+// without its query;
+// file name the record's dir name is the enclosure's file name, as the
+// import named it (`extractVideoId` of the URL).
+//
+// The first rule that finds exactly ONE item wins. A rule that finds several
+// (a feed listing the same file twice) decides nothing; a record no rule
+// settles is reported as ambiguous or unmatched, never guessed.
+//
+// WHAT IS WRITTEN, and only where the record lacks it: the title, the
+// description, the duration (a value measured from the media beats the
+// feed's declared one, so an existing duration stays), the upload_date and
+// timestamp (UTC, from pubDate) — and webpage_url, under one rule below.
+// Through `patchMetadataInfo`, the one non-yt-dlp writer of
+// metadata.info.json, so every change is in `metadata.history.json` as
+// `feed-backfill`.
+//
+// WEBPAGE_URL IS THE VIDEO'S ID, NOT A LINK. The snapshot reconciles every
+// video dir to `extractVideoId(webpage_url)` (reconcileVideoDirs.ts) and a
+// re-download fetches it (undownloadedVideos.ts). So the item's <link> — an
+// episode page, sometimes the show's page shared by every item — is written
+// only when its id IS the record's dir name, else the enclosure URL under the
+// same test, else nothing: a page link that renamed or merged the dir would
+// cost the record, not decorate it.
+
+import path from "node:path";
+import { getPaths, type Paths } from "../lib/paths";
+import { readJsonFile } from "../lib/jsonFile-server";
+import { mapConcurrent } from "../lib/concurrency";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { extractVideoId } from "../lib/videoId";
+import { PROJECT_NAME, PROJECT_URL } from "../lib/project";
+import { patchMetadataInfo } from "../lib/metadataHistory-server";
+import {
+ feedDateToTimestamp,
+ feedDateToUploadDate,
+ parseRssFeed,
+ type FeedItem,
+} from "../lib/rssFeed";
+import { readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+
+const INFO_JSON = "metadata.info.json";
+const READ_CONCURRENCY = 16;
+// One request, but not one that may hang a queue: a feed that has not
+// answered in this long is not going to.
+const FEED_FETCH_TIMEOUT_MS = 60_000;
+
+type Info = Record<string, unknown>;
+
+export type FeedMatchRule = "guid" | "enclosure" | "file-name";
+
+export type FeedMetadataPatch = {
+ title?: string;
+ description?: string;
+ duration?: number;
+ webpage_url?: string;
+ timestamp?: number;
+ upload_date?: string;
+};
+
+export type FeedRecord = { id: string; info: Info | null };
+
+export type FeedBackfillMatch = {
+ id: string;
+ by: FeedMatchRule;
+ item: FeedItem;
+ patch: FeedMetadataPatch;
+};
+
+export type FeedBackfillPlan = {
+ complete: string[];
+ matched: FeedBackfillMatch[];
+ unmatched: string[];
+ ambiguous: { id: string; rule: FeedMatchRule; candidates: number }[];
+ // A video dir with no readable metadata.info.json: nothing to complete.
+ noMetadata: string[];
+};
+
+// ── pure: what a record lacks ─────────────────────────────────────────────
+
+function str(v: unknown): string | null {
+ return typeof v === "string" && v.trim() ? v.trim() : null;
+}
+
+function stem(name: string): string {
+ const dot = name.lastIndexOf(".");
+ return dot > 0 ? name.slice(0, dot) : name;
+}
+
+// A title that only restates the file: what the generic extractor writes for a
+// direct media URL.
+export function isPlaceholderTitle(id: string, info: Info): boolean {
+ const title = str(info.title);
+ if (!title) return true;
+ const fileNames = [id, str(info.webpage_url_basename)].filter(
+ (s): s is string => s !== null,
+ );
+ const placeholders = new Set<string>([
+ ...fileNames,
+ ...fileNames.map(stem),
+ ...[str(info.id), str(info.display_id)].filter((s): s is string => s !== null),
+ ]);
+ return placeholders.has(title);
+}
+
+function hasUploadDate(info: Info): boolean {
+ return typeof info.upload_date === "string" && /^\d{8}$/.test(info.upload_date);
+}
+
+export function isRecordComplete(id: string, info: Info): boolean {
+ return hasUploadDate(info) && !isPlaceholderTitle(id, info);
+}
+
+// ── pure: matching ────────────────────────────────────────────────────────
+
+function urlKeys(raw: string | null): string[] {
+ if (!raw) return [];
+ try {
+ const u = new URL(raw);
+ u.hash = "";
+ const full = u.href;
+ u.search = "";
+ return full === u.href ? [full] : [full, u.href];
+ } catch {
+ return [];
+ }
+}
+
+function safeDecode(s: string): string {
+ try {
+ return decodeURIComponent(s);
+ } catch {
+ return s;
+ }
+}
+
+function fileNameKeys(raw: string | null): string[] {
+ if (!raw) return [];
+ const keys = new Set([raw, safeDecode(raw)]);
+ return [...keys];
+}
+
+type FeedIndex = Record<FeedMatchRule, Map<string, FeedItem[]>>;
+
+function add(map: Map<string, FeedItem[]>, key: string, item: FeedItem): void {
+ const list = map.get(key);
+ if (!list) map.set(key, [item]);
+ else if (!list.includes(item)) list.push(item);
+}
+
+export function indexFeedItems(items: readonly FeedItem[]): FeedIndex {
+ const index: FeedIndex = {
+ guid: new Map(),
+ enclosure: new Map(),
+ "file-name": new Map(),
+ };
+ for (const item of items) {
+ if (item.guid) add(index.guid, item.guid, item);
+ for (const k of urlKeys(item.enclosureUrl)) add(index.enclosure, k, item);
+ const fileName = item.enclosureUrl ? extractVideoId(item.enclosureUrl) : null;
+ for (const k of fileNameKeys(fileName)) add(index["file-name"], k, item);
+ }
+ return index;
+}
+
+const RULES: readonly FeedMatchRule[] = ["guid", "enclosure", "file-name"];
+
+function recordKeys(id: string, info: Info, rule: FeedMatchRule): string[] {
+ switch (rule) {
+ case "guid":
+ return [str(info.id), str(info.display_id)].filter(
+ (s): s is string => s !== null,
+ );
+ case "enclosure":
+ return [info.webpage_url, info.original_url, info.url].flatMap((u) =>
+ urlKeys(str(u)),
+ );
+ case "file-name":
+ return [id, str(info.webpage_url_basename)].flatMap((n) => fileNameKeys(n));
+ }
+}
+
+export type RecordMatch =
+ | { kind: "matched"; item: FeedItem; by: FeedMatchRule }
+ | { kind: "ambiguous"; rule: FeedMatchRule; candidates: number }
+ | { kind: "unmatched" };
+
+export function matchRecord(id: string, info: Info, index: FeedIndex): RecordMatch {
+ let ambiguous: RecordMatch | null = null;
+ for (const rule of RULES) {
+ const found = new Set<FeedItem>();
+ for (const key of recordKeys(id, info, rule)) {
+ for (const item of index[rule].get(key) ?? []) found.add(item);
+ }
+ if (found.size === 1) return { kind: "matched", item: [...found][0], by: rule };
+ if (found.size > 1 && !ambiguous) {
+ ambiguous = { kind: "ambiguous", rule, candidates: found.size };
+ }
+ }
+ return ambiguous ?? { kind: "unmatched" };
+}
+
+// ── pure: what to write ───────────────────────────────────────────────────
+
+// The fields a matched record lacks, from its item. Key order is the order
+// they are appended to a file that lacks them: the long description before the
+// date, so the date stays in the tail recencyIndex.ts reads.
+export function feedPatchFor(
+ id: string,
+ info: Info,
+ item: FeedItem,
+): FeedMetadataPatch {
+ const patch: FeedMetadataPatch = {};
+ if (item.title && isPlaceholderTitle(id, info)) patch.title = item.title;
+ if (item.description && !str(info.description)) {
+ patch.description = item.description;
+ }
+ const duration = info.duration;
+ if (
+ item.durationSeconds !== null &&
+ item.durationSeconds > 0 &&
+ !(typeof duration === "number" && duration > 0)
+ ) {
+ patch.duration = item.durationSeconds;
+ }
+ // See the header: a URL here must keep the record's id.
+ const webpage = [item.link, item.enclosureUrl].find(
+ (u): u is string => !!u && extractVideoId(u) === id,
+ );
+ if (webpage && webpage !== info.webpage_url) patch.webpage_url = webpage;
+ if (!hasUploadDate(info)) {
+ const uploadDate = feedDateToUploadDate(item.pubDate);
+ const timestamp = feedDateToTimestamp(item.pubDate);
+ if (uploadDate) {
+ if (timestamp !== null && typeof info.timestamp !== "number") {
+ patch.timestamp = timestamp;
+ }
+ patch.upload_date = uploadDate;
+ }
+ }
+ return patch;
+}
+
+export function planFeedBackfill(
+ records: readonly FeedRecord[],
+ items: readonly FeedItem[],
+): FeedBackfillPlan {
+ const index = indexFeedItems(items);
+ const plan: FeedBackfillPlan = {
+ complete: [],
+ matched: [],
+ unmatched: [],
+ ambiguous: [],
+ noMetadata: [],
+ };
+ for (const { id, info } of records) {
+ if (!info) {
+ plan.noMetadata.push(id);
+ continue;
+ }
+ if (isRecordComplete(id, info)) {
+ plan.complete.push(id);
+ continue;
+ }
+ const m = matchRecord(id, info, index);
+ if (m.kind === "matched") {
+ plan.matched.push({ id, by: m.by, item: m.item, patch: feedPatchFor(id, info, m.item) });
+ } else if (m.kind === "ambiguous") {
+ plan.ambiguous.push({ id, rule: m.rule, candidates: m.candidates });
+ } else {
+ plan.unmatched.push(id);
+ }
+ }
+ return plan;
+}
+
+// ── the fetch ─────────────────────────────────────────────────────────────
+
+export type FetchFeed = (url: string, signal?: AbortSignal) => Promise<string>;
+
+// The one request. Identified, bounded in time, and never retried here: a
+// failure is the job's failure, and the operator re-runs it.
+export const fetchFeedText: FetchFeed = async (url, signal) => {
+ const timeout = AbortSignal.timeout(FEED_FETCH_TIMEOUT_MS);
+ const res = await fetch(url, {
+ signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
+ redirect: "follow",
+ headers: {
+ accept:
+ "application/rss+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.1",
+ "user-agent": `${PROJECT_NAME} feed backfill (+${PROJECT_URL})`,
+ },
+ });
+ if (!res.ok) {
+ throw new Error(`the feed answered HTTP ${res.status} ${res.statusText}`.trim());
+ }
+ return res.text();
+};
+
+// ── the run ───────────────────────────────────────────────────────────────
+
+export type FeedBackfillResult = {
+ slug: string;
+ feedUrl: string;
+ dryRun: boolean;
+ feedItems: number;
+ records: number;
+ complete: number;
+ matched: number;
+ written: number;
+ unchanged: number;
+ unmatched: string[];
+ ambiguous: string[];
+ noMetadata: number;
+ failed: { id: string; error: string }[];
+};
+
+export type FeedBackfillOpts = {
+ paths?: Paths;
+ slug: string;
+ // Default: the channel's configured url.
+ feedUrl?: string;
+ dryRun?: boolean;
+ // Who asked, for the history entry ("cli", "ops", …).
+ requestedBy?: string;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // The fetch, injectable so a test never touches the network.
+ fetchFeed?: FetchFeed;
+};
+
+function describePatch(p: FeedMetadataPatch): string {
+ const parts: string[] = [];
+ if (p.title !== undefined) parts.push(`title ${JSON.stringify(p.title)}`);
+ if (p.upload_date !== undefined) parts.push(`date ${p.upload_date}`);
+ if (p.duration !== undefined) parts.push(`duration ${p.duration}s`);
+ if (p.description !== undefined) parts.push(`description (${p.description.length} chars)`);
+ if (p.webpage_url !== undefined) parts.push("webpage_url");
+ return parts.join(", ") || "nothing it lacks";
+}
+
+export async function backfillFeedMetadata(
+ opts: FeedBackfillOpts,
+): Promise<FeedBackfillResult> {
+ const paths = opts.paths ?? getPaths();
+ const { slug } = opts;
+ const log = opts.onLog ?? (() => {});
+ const dryRun = opts.dryRun === true;
+ const config = await readChannelConfig(paths, slug);
+ if (!config) throw new Error(`Channel "${slug}" not found`);
+ const feedUrl = (opts.feedUrl ?? config.url ?? "").trim();
+ if (!/^https?:\/\//i.test(feedUrl)) {
+ throw new Error(
+ feedUrl
+ ? `"${feedUrl}" is not an http(s) feed URL`
+ : `Channel "${slug}" has no url; pass the feed URL`,
+ );
+ }
+ // AN UNREADABLE TEXT TIER IS NOT AN EMPTY CHANNEL (AGENTS.md): the walk below
+ // would find no records and report a clean run.
+ await assertChannelTextReadable(paths, slug, config);
+
+ const dataDir = path.join(paths.channelsDir, slug, "data");
+ const ids = await listChannelVideoIds(paths, slug);
+ const records = await mapConcurrent(ids, READ_CONCURRENCY, async (id) => {
+ const read = await readJsonFile(path.join(dataDir, id, INFO_JSON));
+ const value = read.ok ? read.value : null;
+ const info =
+ value && typeof value === "object" && !Array.isArray(value)
+ ? (value as Info)
+ : null;
+ return { id, info };
+ });
+ const incomplete = records.filter(
+ (r) => r.info && !isRecordComplete(r.id, r.info),
+ ).length;
+ log(
+ `${slug}: ${records.length} record(s), ${incomplete} lacking a title or a date.`,
+ );
+
+ const result: FeedBackfillResult = {
+ slug,
+ feedUrl,
+ dryRun,
+ feedItems: 0,
+ records: records.length,
+ complete: 0,
+ matched: 0,
+ written: 0,
+ unchanged: 0,
+ unmatched: [],
+ ambiguous: [],
+ noMetadata: 0,
+ failed: [],
+ };
+ if (incomplete === 0) {
+ // Nothing to complete is nothing to fetch: the feed is not asked.
+ result.complete = records.filter((r) => r.info).length;
+ result.noMetadata = records.length - result.complete;
+ log("Every record has a title and a date; the feed was not fetched.");
+ return result;
+ }
+
+ log(`Fetching the feed (one request): ${feedUrl}`);
+ const xml = await (opts.fetchFeed ?? fetchFeedText)(feedUrl, opts.signal);
+ const feed = parseRssFeed(xml);
+ if (feed.items.length === 0) {
+ throw new Error("the feed has no <item> — is this URL an RSS feed?");
+ }
+ result.feedItems = feed.items.length;
+ log(
+ `Feed${feed.title ? ` "${feed.title}"` : ""}: ${feed.items.length} item(s).`,
+ );
+
+ const plan = planFeedBackfill(records, feed.items);
+ result.complete = plan.complete.length;
+ result.matched = plan.matched.length;
+ result.unmatched = plan.unmatched;
+ result.ambiguous = plan.ambiguous.map((a) => a.id);
+ result.noMetadata = plan.noMetadata.length;
+
+ for (const m of plan.matched) {
+ if (opts.signal?.aborted) break;
+ const what = describePatch(m.patch);
+ if (dryRun) {
+ log(` would write ${m.id} (by ${m.by}): ${what}`);
+ continue;
+ }
+ if (Object.keys(m.patch).length === 0) {
+ result.unchanged += 1;
+ log(` ${m.id} (by ${m.by}): the item has nothing it lacks`);
+ continue;
+ }
+ try {
+ const { written } = await patchMetadataInfo(
+ path.join(dataDir, m.id),
+ { ...m.patch },
+ {
+ by: "feed-backfill",
+ ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}),
+ onLog: log,
+ },
+ );
+ if (written) {
+ result.written += 1;
+ log(` wrote ${m.id} (by ${m.by}): ${what}`);
+ } else {
+ result.unchanged += 1;
+ }
+ } catch (err) {
+ result.failed.push({ id: m.id, error: (err as Error).message });
+ log(` FAILED ${m.id}: ${(err as Error).message}`);
+ }
+ }
+ for (const a of plan.ambiguous) {
+ log(` ambiguous ${a.id}: ${a.candidates} items share its ${a.rule}`);
+ }
+ for (const id of plan.unmatched) log(` unmatched ${id}`);
+ return result;
+}
+
+export function feedBackfillSummary(r: FeedBackfillResult): string {
+ const head = r.dryRun ? "Dry run" : "Done";
+ const parts = [
+ `${r.matched} matched`,
+ `${r.unmatched.length} unmatched`,
+ `${r.complete} already complete`,
+ ];
+ if (r.ambiguous.length) parts.push(`${r.ambiguous.length} ambiguous`);
+ if (r.noMetadata) parts.push(`${r.noMetadata} without metadata`);
+ if (!r.dryRun) {
+ parts.push(`${r.written} written`);
+ if (r.unchanged) parts.push(`${r.unchanged} unchanged`);
+ if (r.failed.length) parts.push(`${r.failed.length} failed`);
+ }
+ let line = `${head}: ${r.slug} — ${parts.join(", ")} (of ${r.records} record(s)`;
+ line += r.feedItems ? `, ${r.feedItems} feed item(s)).` : ").";
+ if (!r.dryRun && r.written > 0) {
+ line += " Rebuild the index for the new titles and dates to reach search and the sites.";
+ }
+ return line;
+}
diff --git a/common/controller/feedMetadataJob.ts b/common/controller/feedMetadataJob.ts
@@ -0,0 +1,61 @@
+// THE FEED METADATA BACKFILL AS A JOB. The work is
+// controller/feedMetadataBackfill.ts, which the CLI runs offline; this is what
+// the editor's action and `/api/ops/feed-metadata` enqueue. Its own module so
+// the CLI never loads the job registry.
+
+import { getPaths, type Paths } from "../lib/paths";
+import { downloadQueueKey } from "../lib/queueKeys";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "../jobs/streamCommand";
+import { readChannelConfig } from "./channels";
+import {
+ backfillFeedMetadata,
+ feedBackfillSummary,
+} from "./feedMetadataBackfill";
+
+export const FEED_METADATA_JOB_KIND = "feed-metadata";
+
+// One job per run, on the channel's DOWNLOAD queue: the fetch contends with
+// that queue's own requests to the same host, and the writes must not land
+// beside a download rewriting the same metadata.info.json. `needsText` (the
+// kind's entry in jobs/jobKinds.ts) refuses it for a channel whose text cannot
+// be read. Not replayable: a re-run is the same click.
+export async function runFeedMetadataJob(opts: {
+ paths?: Paths;
+ slug: string;
+ feedUrl?: string;
+ dryRun?: boolean;
+ requestedBy?: string;
+ afterRun?: () => void | Promise<void>;
+}): Promise<StreamActionResult> {
+ const paths = opts.paths ?? getPaths();
+ const config = await readChannelConfig(paths, opts.slug);
+ if (!config) return { ok: false, error: `Channel "${opts.slug}" not found` };
+ return runManagedFunction({
+ kind: FEED_METADATA_JOB_KIND,
+ queueKey: downloadQueueKey(config),
+ paths,
+ channelSlug: opts.slug,
+ fn: async (onLog, signal) => {
+ onLog(
+ `${opts.dryRun ? "Previewing" : "Backfilling"} feed metadata for ${opts.slug}.`,
+ );
+ const result = await backfillFeedMetadata({
+ paths,
+ slug: opts.slug,
+ ...(opts.feedUrl ? { feedUrl: opts.feedUrl } : {}),
+ ...(opts.dryRun ? { dryRun: true } : {}),
+ ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}),
+ onLog,
+ signal,
+ });
+ onLog(feedBackfillSummary(result));
+ if (result.failed.length > 0) {
+ throw new Error(`${result.failed.length} record(s) could not be written`);
+ }
+ await opts.afterRun?.();
+ },
+ });
+}
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -201,3 +201,11 @@ test("release 17: every media kind keeps needsMedia, and none is also a text kin
assert.equal(kindNeedsText(k), false, k);
}
});
+
+test("the feed metadata backfill reads and writes the text tier only", () => {
+ assert.ok(getJobKind("feed-metadata"), "registered");
+ assert.equal(kindNeedsMedia("feed-metadata"), false);
+ assert.equal(kindNeedsText("feed-metadata"), true);
+ assert.equal(jobKindLabel("feed-metadata"), "Backfill feed metadata");
+ assert.equal(isDrainableKind("feed-metadata"), false);
+});
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -577,6 +577,25 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
needsMedia: false,
needsText: true,
},
+ // A podcast channel's records completed from its RSS feed
+ // (controller/feedMetadataBackfill.ts): one fetch of the feed, then a write
+ // of title, date, description and duration into each metadata.info.json that
+ // lacks them. On the platform download queue, like the metadata scan, so the
+ // fetch takes its turn with the host and the writes never land beside a
+ // download rewriting the same file. `needsText`: it reads and writes the
+ // text tier only, and against an unreadable one would report a clean run
+ // over zero records. Not drainable (one request and a few small writes); not
+ // replayable (a re-run is the same click, and finds the completed records
+ // complete).
+ "feed-metadata": {
+ kind: "feed-metadata",
+ label: "Backfill feed metadata",
+ drainable: false,
+ replayable: false,
+ queueKeyStrategy: "platform",
+ needsMedia: false,
+ needsText: true,
+ },
// THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They
// ran from /sites — the hub since release 7, the homepage since release 11 —
// with no entry here, so /jobs showed their raw machine kinds. The labels
diff --git a/common/lib/metadataHistory-server.test.ts b/common/lib/metadataHistory-server.test.ts
@@ -6,6 +6,7 @@ import path from "node:path";
import {
loadMetadataHistory,
metadataHistoryPath,
+ patchMetadataInfo,
withMetadataHistory,
} from "./metadataHistory-server";
@@ -126,3 +127,51 @@ test("a failure to record is logged, never thrown", async () => {
assert.match(log, /Could not record the metadata history/);
});
});
+
+test("patchMetadataInfo merges in place, appends new keys in order, and records the rewrite", async () => {
+ await withDir(async (dir) => {
+ const original = { id: "v1", title: "v1", formats: [{ url: "u" }], _version: { v: 1 } };
+ await writeFile(path.join(dir, INFO), JSON.stringify(original));
+ const out = await patchMetadataInfo(
+ dir,
+ { title: "Real", description: "About", upload_date: "20240305" },
+ { by: "feed-backfill", requestedBy: "test" },
+ );
+ assert.deepEqual(out, { written: true });
+ const text = await readFile(path.join(dir, INFO), "utf8");
+ const after = JSON.parse(text);
+ assert.deepEqual(Object.keys(after), ["id", "title", "formats", "_version", "description", "upload_date"]);
+ assert.equal(after.title, "Real");
+ assert.deepEqual(after.formats, original.formats);
+ assert.equal(text.endsWith("\n"), false, "compact, no trailing newline");
+ const h = await loadMetadataHistory(dir);
+ assert.equal(h?.entries.length, 1);
+ assert.equal(h!.entries[0].by, "feed-backfill");
+ assert.equal(h!.entries[0].requestedBy, "test");
+ assert.deepEqual(h!.entries[0].changed.title, { from: "v1", to: "Real" });
+ assert.deepEqual(h!.entries[0].added, { description: "About", upload_date: "20240305" });
+ // No temp file is left behind.
+ assert.deepEqual((await readdir(dir)).sort(), [INFO, "metadata.history.json"].sort());
+ });
+});
+
+test("patchMetadataInfo: a patch that changes nothing writes nothing", async () => {
+ await withDir(async (dir) => {
+ const text = JSON.stringify({ id: "v1", title: "Same" });
+ await writeFile(path.join(dir, INFO), text);
+ const out = await patchMetadataInfo(dir, { title: "Same", description: undefined }, { by: "feed-backfill" });
+ assert.deepEqual(out, { written: false });
+ assert.equal(await readFile(path.join(dir, INFO), "utf8"), text);
+ assert.equal(await loadMetadataHistory(dir), null);
+ });
+});
+
+test("patchMetadataInfo refuses a missing file or one that is not a JSON object", async () => {
+ await withDir(async (dir) => {
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /ENOENT/);
+ await writeFile(path.join(dir, INFO), "[1,2]");
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not a JSON object/);
+ await writeFile(path.join(dir, INFO), "{torn");
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not JSON/);
+ });
+});
diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts
@@ -26,7 +26,7 @@
import path from "node:path";
import { readFile, stat } from "node:fs/promises";
-import { withJsonFileLock } from "./jsonFile-server";
+import { withJsonFileLock, writeJsonAtomic } from "./jsonFile-server";
import { sidecar, sidecarField } from "./sidecar-server";
import {
METADATA_HISTORY_FILENAME,
@@ -129,3 +129,56 @@ export async function withMetadataHistory<T>(
}
}
}
+
+// THE ONE WRITER OF metadata.info.json THAT IS NOT yt-dlp: merge `patch` into
+// the file on disk, inside `withMetadataHistory`, so the rewrite is recorded
+// exactly as a yt-dlp one is (by `ctx.by`, with what moved).
+//
+// A MERGE, NEVER A REPLACEMENT. Every key the file has stays where it is and
+// keeps its value unless `patch` names it; a key the file lacks is appended in
+// `patch`'s order. Order is not cosmetic: two readers scan the file's BYTES
+// rather than parse it — the title from a 16 KB head (videoTitles.ts) and the
+// upload_date from an 8 KB tail (recencyIndex.ts) — so a caller adding a long
+// description and a date names the description first.
+//
+// Compact JSON, no trailing newline — yt-dlp's own shape, near enough for
+// every reader (they all allow whitespace after a colon). A patch that changes
+// nothing writes nothing. A missing file, or one that is not a JSON object,
+// throws: there is no record to complete, and a write would invent one.
+//
+// Under the file's lock, so two patches of one record never drop each other.
+// It does NOT serialise against a yt-dlp spawn rewriting the same file; the
+// caller's job runs on the channel's download queue for that reason.
+export async function patchMetadataInfo(
+ videoDir: string,
+ patch: Record<string, unknown>,
+ ctx: MetadataRewriteContext & { onLog?: (line: string) => void },
+): Promise<{ written: boolean }> {
+ const file = path.join(videoDir, INFO_JSON);
+ return withJsonFileLock(file, async () => {
+ const text = await readFile(file, "utf8");
+ let current: unknown;
+ try {
+ current = JSON.parse(text);
+ } catch {
+ throw new Error(`${file} is not JSON`);
+ }
+ if (typeof current !== "object" || current === null || Array.isArray(current)) {
+ throw new Error(`${file} is not a JSON object`);
+ }
+ const before = current as Record<string, unknown>;
+ const next: Record<string, unknown> = { ...before };
+ let changed = false;
+ for (const [key, value] of Object.entries(patch)) {
+ if (value === undefined) continue;
+ if (JSON.stringify(before[key]) === JSON.stringify(value)) continue;
+ next[key] = value;
+ changed = true;
+ }
+ if (!changed) return { written: false };
+ await withMetadataHistory(videoDir, ctx, () =>
+ writeJsonAtomic(file, next, { indent: 0, newline: false }),
+ );
+ return { written: true };
+ });
+}
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -46,7 +46,8 @@ export const METADATA_HISTORY_CAP = 200;
// A stored value whose JSON is larger than this is replaced by its digest.
export const METADATA_HISTORY_MAX_VALUE_BYTES = 16 * 1024;
-// Who rewrote the file: the three yt-dlp spawns that write it.
+// Who rewrote the file: the three yt-dlp spawns that write it, and the one
+// writer that is not yt-dlp (`patchMetadataInfo`, metadataHistory-server.ts).
export const METADATA_HISTORY_WRITERS = [
// downloadOneManaged's attempt 0, on every managed download.
"prefetch",
@@ -55,6 +56,10 @@ export const METADATA_HISTORY_WRITERS = [
"audio-check",
// runYtdlp's legacy single-video `download-one-audio` job.
"download-one",
+ // The podcast feed backfill (controller/feedMetadataBackfill.ts): title,
+ // date, description and duration from the channel's RSS feed, for a record
+ // imported by its enclosure URL, which carries none of them.
+ "feed-backfill",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];