commit 84fea46778622d71c1d13fecffcbc9bfd0f4bf40
parent b3532d20c2705f3520a6bdad141b775f616bc3cc
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:26:24 -0400
Merge feed-metadata-backfill (archilyzer feeds backfill-metadata: a podcast feed's titles, dates and descriptions onto feed-imported records, through a history-recorded metadata writer; job kind + ops route)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
18 files changed, 1618 insertions(+), 2 deletions(-)
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -308,6 +308,26 @@ export const COMMANDS: Command[] = [
);
},
},
+ {
+ path: ["feeds", "backfill-metadata"],
+ usage:
+ "<slug> [--feed <url>] [--dry-run] complete a podcast channel's records (title, date, description, duration) from its RSS feed: one fetch of the feed (default: the channel's url), no media; --dry-run counts matched / unmatched / already complete and writes nothing",
+ flags: { feed: "string", "dry-run": "boolean" },
+ maxPositionals: 1,
+ run: async ({ positionals, flags }) => {
+ const [slug] = positionals;
+ if (!slug) {
+ console.error("feeds backfill-metadata: which channel? Pass its slug.");
+ return 2;
+ }
+ return (await import("./feeds-backfill-metadata")).main({
+ slug,
+ ...(typeof flags.feed === "string" ? { feedUrl: flags.feed } : {}),
+ dryRun: flags["dry-run"] === true,
+ signal: interrupted(),
+ });
+ },
+ },
// The bins that parse their own flags, run as children with their argv
// verbatim (_spawnBin.ts says why). Each row is the whole integration.
script(["duplicates"], "duplicate-shorts.ts",
diff --git a/common/bin/feeds-backfill-metadata.ts b/common/bin/feeds-backfill-metadata.ts
@@ -0,0 +1,41 @@
+// `archilyzer feeds backfill-metadata <slug> [--feed <url>] [--dry-run]` —
+// a podcast channel's records completed from its RSS feed, offline. The work
+// and its rules are controller/feedMetadataBackfill.ts; the editor runs the
+// same function as the `feed-metadata` job (`POST /api/ops/feed-metadata`).
+//
+// One fetch of the feed and nothing else on the network. It writes
+// metadata.info.json files through the history, atomically; it does not see
+// the editor's queues, so not beside a download on the same channel.
+
+import { isValidChannelSlug } from "../controller/channels";
+import {
+ backfillFeedMetadata,
+ feedBackfillSummary,
+} from "../controller/feedMetadataBackfill";
+
+export async function main(opts: {
+ slug: string;
+ feedUrl?: string;
+ dryRun: boolean;
+ signal?: AbortSignal;
+}): Promise<number> {
+ if (!isValidChannelSlug(opts.slug)) {
+ console.error(`feeds backfill-metadata: "${opts.slug}" is not a channel slug`);
+ return 2;
+ }
+ try {
+ const result = await backfillFeedMetadata({
+ slug: opts.slug,
+ ...(opts.feedUrl ? { feedUrl: opts.feedUrl } : {}),
+ dryRun: opts.dryRun,
+ requestedBy: "cli",
+ onLog: (line) => console.log(line.replace(/\n$/, "")),
+ ...(opts.signal ? { signal: opts.signal } : {}),
+ });
+ console.log(feedBackfillSummary(result));
+ return result.failed.length > 0 || opts.signal?.aborted ? 1 : 0;
+ } catch (err) {
+ console.error(`feeds backfill-metadata: ${(err as Error).message}`);
+ return 1;
+ }
+}
diff --git a/common/controller/feedMetadataBackfill.test.ts b/common/controller/feedMetadataBackfill.test.ts
@@ -0,0 +1,363 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises";
+import { readFileSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import type { Paths } from "../lib/paths";
+import type { FeedItem } from "../lib/rssFeed";
+import { loadMetadataHistory } from "../lib/metadataHistory-server";
+import {
+ backfillFeedMetadata,
+ feedBackfillSummary,
+ feedPatchFor,
+ isPlaceholderTitle,
+ isRecordComplete,
+ planFeedBackfill,
+ type FetchFeed,
+} from "./feedMetadataBackfill";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/feedMetadataBackfill.test.ts
+//
+// No network: the fetch is a fake that serves the fixture feed and counts.
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FEED_XML = readFileSync(
+ path.join(HERE, "..", "lib", "__fixtures__", "demo-podcast.rss"),
+ "utf8",
+);
+const FEED_URL = "https://feeds.example.com/demo.rss";
+const SLUG = "demo-channel";
+
+// What the generic extractor writes for a direct .mp3 URL: the file name as
+// the title, no date. Key order as yt-dlp writes it.
+function genericInfo(fileName: string, extra: Record<string, unknown> = {}) {
+ return {
+ id: `uuid-${fileName}`,
+ title: fileName.replace(/\.mp3$/, ""),
+ direct: true,
+ formats: [{ format_id: "mpeg", url: `https://cdn.example.com/audio/${fileName}?key=k` }],
+ webpage_url: `https://cdn.example.com/audio/${fileName}?key=k`,
+ webpage_url_basename: fileName,
+ extractor: "generic",
+ _version: { version: "2026.01.01" },
+ ...extra,
+ };
+}
+
+type Seed = Record<string, Record<string, unknown> | null>;
+
+async function withCorpus(
+ seed: Seed,
+ fn: (paths: Paths, dataDir: string) => Promise<void>,
+): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-feed-"));
+ const transcriptsDir = path.join(dir, "corpus");
+ const paths = {
+ transcriptsDir,
+ channelsDir: path.join(transcriptsDir, "channels"),
+ } as Paths;
+ const channelDir = path.join(paths.channelsDir, SLUG);
+ const dataDir = path.join(channelDir, "data");
+ await mkdir(dataDir, { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({ name: "Demo", url: FEED_URL, handling: "transcribe" }),
+ );
+ for (const [id, info] of Object.entries(seed)) {
+ await mkdir(path.join(dataDir, id), { recursive: true });
+ if (info) {
+ // yt-dlp's own separators, so a byte comparison means something.
+ await writeFile(
+ path.join(dataDir, id, "metadata.info.json"),
+ JSON.stringify(info),
+ );
+ }
+ }
+ try {
+ await fn(paths, dataDir);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function fakeFetch(): FetchFeed & { calls: string[] } {
+ const calls: string[] = [];
+ const f = (async (url: string) => {
+ calls.push(url);
+ return FEED_XML;
+ }) as FetchFeed & { calls: string[] };
+ f.calls = calls;
+ return f;
+}
+
+const readInfo = async (dataDir: string, id: string) =>
+ JSON.parse(await readFile(path.join(dataDir, id, "metadata.info.json"), "utf8")) as Record<
+ string,
+ unknown
+ >;
+
+const SEED: Seed = {
+ // Matched by its enclosure URL: same path, a different signed query.
+ "abc123.mp3": genericInfo("abc123.mp3"),
+ // Matched by guid: yt-dlp forced the feed's guid as the id.
+ "def456.mp3": genericInfo("def456.mp3", {
+ id: "guid-0002",
+ webpage_url: "https://cdn2.example.org/def456.mp3?key=q",
+ }),
+ // Matched by file name only, and already carries a measured duration.
+ "ghi789.mp3": genericInfo("ghi789.mp3", {
+ webpage_url: "https://mirror.example.org/x/ghi789.mp3",
+ duration: 3700,
+ }),
+ "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" },
+ "zzz999.mp3": genericInfo("zzz999.mp3"),
+ "nometa.mp3": null,
+};
+
+test("a run completes every matched record, through the history, with one fetch", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const fetchFeed = fakeFetch();
+ const lines: string[] = [];
+ const r = await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed,
+ requestedBy: "test",
+ onLog: (l) => lines.push(l),
+ });
+ assert.deepEqual(fetchFeed.calls, [FEED_URL], "the channel's url, once");
+ assert.equal(r.feedItems, 5);
+ assert.equal(r.matched, 3);
+ assert.equal(r.written, 3);
+ assert.equal(r.complete, 1);
+ assert.deepEqual(r.unmatched, ["zzz999.mp3"]);
+ assert.equal(r.noMetadata, 1);
+ assert.deepEqual(r.failed, []);
+
+ const one = await readInfo(dataDir, "abc123.mp3");
+ assert.equal(one.title, "Episode 1: Cats & <Dogs>");
+ assert.equal(one.upload_date, "20240305");
+ assert.equal(one.timestamp, 1709632800);
+ assert.equal(one.duration, 3723);
+ assert.equal(one.description, "First line & more.\nSecond line\nthird");
+ // The item's <link> would rename the dir (its id is "one"), so the
+ // enclosure — whose file name IS the dir name — is the webpage_url.
+ assert.equal(
+ one.webpage_url,
+ "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1",
+ );
+ // Everything else is untouched, in place; new keys are appended with the
+ // date LAST (the tail reader's 8 KB).
+ const keys = Object.keys(one);
+ assert.deepEqual(keys.slice(0, 8), Object.keys(genericInfo("abc123.mp3")));
+ assert.equal(keys.at(-1), "upload_date");
+ assert.deepEqual(one.formats, genericInfo("abc123.mp3").formats);
+
+ const two = await readInfo(dataDir, "def456.mp3");
+ assert.equal(two.title, "Two & a half ’quotes’ \"here\"");
+ assert.equal(two.upload_date, "20240305", "22:30 at -0500 is the next UTC day");
+ assert.equal(two.description, "Hello & goodbye");
+
+ const three = await readInfo(dataDir, "ghi789.mp3");
+ assert.equal(three.title, "Episode three");
+ assert.equal(three.duration, 3700, "a measured duration is kept");
+ assert.equal(three.upload_date, "20240306");
+
+ // Through the history: one entry per written record, by feed-backfill.
+ const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3"));
+ assert.equal(h?.entries.length, 1);
+ const e = h!.entries[0];
+ assert.equal(e.by, "feed-backfill");
+ assert.equal(e.requestedBy, "test");
+ assert.deepEqual(e.changed.title, { from: "abc123", to: "Episode 1: Cats & <Dogs>" });
+ assert.equal(e.added.upload_date, "20240305");
+
+ // The complete and the unmatched records are not touched.
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "done.mp3")), null);
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "zzz999.mp3")), null);
+ assert.match(feedBackfillSummary(r), /3 matched, 1 unmatched, 1 already complete/);
+ assert.ok(lines.some((l) => l.includes("unmatched zzz999.mp3")));
+ });
+});
+
+test("a second run finds the records complete and still asks the feed only for the rest", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: fakeFetch() });
+ const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ const fetchFeed = fakeFetch();
+ const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed });
+ assert.equal(r.complete, 4);
+ assert.equal(r.written, 0);
+ assert.deepEqual(r.unmatched, ["zzz999.mp3"]);
+ const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ assert.deepEqual(after, before, "a complete record is not rewritten");
+ const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3"));
+ assert.equal(h?.entries.length, 1, "and no second history entry");
+ });
+});
+
+test("nothing incomplete: the feed is not fetched at all", async () => {
+ await withCorpus(
+ { "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" } },
+ async (paths) => {
+ const fetchFeed = fakeFetch();
+ const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed });
+ assert.deepEqual(fetchFeed.calls, []);
+ assert.equal(r.complete, 1);
+ assert.equal(r.matched, 0);
+ },
+ );
+});
+
+test("a dry run reports matched / unmatched / complete and writes nothing", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ const lines: string[] = [];
+ const r = await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ dryRun: true,
+ fetchFeed: fakeFetch(),
+ onLog: (l) => lines.push(l),
+ });
+ assert.equal(r.matched, 3);
+ assert.equal(r.unmatched.length, 1);
+ assert.equal(r.complete, 1);
+ assert.equal(r.written, 0);
+ const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json"));
+ assert.deepEqual(after, before);
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null);
+ assert.ok(lines.some((l) => l.startsWith(" would write abc123.mp3 (by enclosure)")));
+ assert.match(feedBackfillSummary(r), /^Dry run: demo-channel — 3 matched, 1 unmatched, 1 already complete/);
+ });
+});
+
+test("--feed overrides the channel's url; a non-http url and a feed with no items refuse", async () => {
+ await withCorpus(SEED, async (paths, dataDir) => {
+ const fetchFeed = fakeFetch();
+ await backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ feedUrl: "https://mirror.example.net/feed.xml",
+ dryRun: true,
+ fetchFeed,
+ });
+ assert.deepEqual(fetchFeed.calls, ["https://mirror.example.net/feed.xml"]);
+
+ await assert.rejects(
+ backfillFeedMetadata({ paths, slug: SLUG, feedUrl: "file:///etc/passwd", fetchFeed }),
+ /not an http\(s\) feed URL/,
+ );
+ await assert.rejects(
+ backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed: async () => "<html><body>not a feed</body></html>",
+ }),
+ /no <item>/,
+ );
+ await assert.rejects(
+ backfillFeedMetadata({ paths, slug: "no-such-channel", fetchFeed }),
+ /not found/,
+ );
+ // A failed fetch writes nothing.
+ await assert.rejects(
+ backfillFeedMetadata({
+ paths,
+ slug: SLUG,
+ fetchFeed: async () => {
+ throw new Error("the feed answered HTTP 503");
+ },
+ }),
+ /503/,
+ );
+ assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null);
+ });
+});
+
+// ── the pure rules ──────────────────────────────────────────────────────────
+
+const item = (over: Partial<FeedItem>): FeedItem => ({
+ title: "T",
+ link: null,
+ guid: null,
+ pubDate: "Tue, 05 Mar 2024 10:00:00 GMT",
+ enclosureUrl: null,
+ durationSeconds: null,
+ description: null,
+ ...over,
+});
+
+test("placeholder titles: the dir name, its stem, the extractor's ids, or none", () => {
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc.mp3" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "uuid-1", id: "uuid-1" }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: " " }), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", {}), true);
+ assert.equal(isPlaceholderTitle("abc.mp3", { title: "A real title" }), false);
+ assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "20240101" }), true);
+ assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "2024" }), false);
+ assert.equal(isRecordComplete("abc.mp3", { title: "abc", upload_date: "20240101" }), false);
+});
+
+test("a rule that finds several items decides nothing; the next unique rule still can", () => {
+ const dupA = item({ guid: "g-a", enclosureUrl: "https://a.example.com/x/same.mp3" });
+ const dupB = item({ guid: "g-b", enclosureUrl: "https://b.example.com/y/same.mp3" });
+ const plan = planFeedBackfill(
+ [
+ { id: "same.mp3", info: { id: "nope", title: "same" } },
+ { id: "same.mp3", info: { id: "g-b", title: "same" } },
+ ],
+ [dupA, dupB],
+ );
+ assert.deepEqual(plan.ambiguous, [{ id: "same.mp3", rule: "file-name", candidates: 2 }]);
+ assert.equal(plan.matched.length, 1);
+ assert.equal(plan.matched[0].by, "guid");
+ assert.equal(plan.matched[0].item, dupB);
+});
+
+test("a URL is compared without its fragment, then without its query", () => {
+ const it = item({ enclosureUrl: "https://cdn.example.com/a/ep.mp3?updated=1" });
+ const plan = planFeedBackfill(
+ [
+ { id: "x1", info: { webpage_url: "https://cdn.example.com/a/ep.mp3?updated=1#__youtubedl_smuggle=1" } },
+ { id: "x2", info: { url: "https://cdn.example.com/a/ep.mp3?key=signed" } },
+ { id: "x3", info: { webpage_url: "https://cdn.example.com/b/ep.mp3" } },
+ ],
+ [it],
+ );
+ assert.deepEqual(plan.matched.map((m) => [m.id, m.by]), [
+ ["x1", "enclosure"],
+ ["x2", "enclosure"],
+ ]);
+ assert.deepEqual(plan.unmatched, ["x3"]);
+});
+
+test("the patch fills only what is missing, and a URL only when it keeps the record's id", () => {
+ const it = item({
+ title: "Real",
+ description: "About it",
+ durationSeconds: 60,
+ link: "https://example.com/episodes/ep.mp3",
+ enclosureUrl: "https://cdn.example.com/ep.mp3",
+ });
+ // The link's id is the dir name: the link wins.
+ assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, it).webpage_url, "https://example.com/episodes/ep.mp3");
+ // A show-page link would rename the dir: the enclosure is used instead.
+ const showLink = { ...it, link: "https://example.com/show" };
+ assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, showLink).webpage_url, "https://cdn.example.com/ep.mp3");
+ // Neither keeps the id: webpage_url is left alone.
+ assert.equal("webpage_url" in feedPatchFor("other.mp3", { title: "other" }, showLink), false);
+ // A record with a real title and description keeps them; only the date goes in.
+ assert.deepEqual(
+ Object.keys(
+ feedPatchFor("ep.mp3", { title: "Mine", description: "Mine too", duration: 59, webpage_url: "https://example.com/episodes/ep.mp3" }, it),
+ ),
+ ["timestamp", "upload_date"],
+ );
+ // An item with an unreadable date adds no date.
+ assert.equal("upload_date" in feedPatchFor("ep.mp3", { title: "ep" }, { ...it, pubDate: "soon" }), false);
+});
diff --git a/common/controller/feedMetadataBackfill.ts b/common/controller/feedMetadataBackfill.ts
@@ -0,0 +1,505 @@
+// A PODCAST CHANNEL'S RECORDS, COMPLETED FROM ITS RSS FEED.
+//
+// An episode imported one by one by its enclosure URL (`import-video` with a
+// direct .mp3 link) goes through yt-dlp's generic extractor, which knows
+// nothing but the file: the record's title is the file name, and it has no
+// upload_date, no description and no duration. The index skips a record with
+// no upload_date, and every surface shows the file id as its title. The feed
+// the episode came from has all four; this reads them back.
+//
+// ONE FETCH, OF ONE URL — the channel's configured url, or the one the caller
+// names — and nothing else on the network: no enclosure is requested, no media
+// is downloaded. The editor's job takes its turn on the channel's download
+// queue (`downloadQueueKey`), behind whatever else is fetching from that host.
+//
+// WHICH RECORDS. A record is COMPLETE — and left alone — when it has an
+// 8-digit upload_date and a title that is not a placeholder (the dir name, its
+// stem, or yt-dlp's id). Every other record is matched to a feed item:
+//
+// guid the record's yt-dlp id (or display_id) is the item's guid —
+// what yt-dlp itself records when it reads an episode out of the
+// feed (it forces the guid as the id);
+// enclosure one of the record's URLs (webpage_url, original_url, url) is the
+// item's enclosure URL, compared without its fragment, then
+// without its query;
+// file name the record's dir name is the enclosure's file name, as the
+// import named it (`extractVideoId` of the URL).
+//
+// The first rule that finds exactly ONE item wins. A rule that finds several
+// (a feed listing the same file twice) decides nothing; a record no rule
+// settles is reported as ambiguous or unmatched, never guessed.
+//
+// WHAT IS WRITTEN, and only where the record lacks it: the title, the
+// description, the duration (a value measured from the media beats the
+// feed's declared one, so an existing duration stays), the upload_date and
+// timestamp (UTC, from pubDate) — and webpage_url, under one rule below.
+// Through `patchMetadataInfo`, the one non-yt-dlp writer of
+// metadata.info.json, so every change is in `metadata.history.json` as
+// `feed-backfill`.
+//
+// WEBPAGE_URL IS THE VIDEO'S ID, NOT A LINK. The snapshot reconciles every
+// video dir to `extractVideoId(webpage_url)` (reconcileVideoDirs.ts) and a
+// re-download fetches it (undownloadedVideos.ts). So the item's <link> — an
+// episode page, sometimes the show's page shared by every item — is written
+// only when its id IS the record's dir name, else the enclosure URL under the
+// same test, else nothing: a page link that renamed or merged the dir would
+// cost the record, not decorate it.
+
+import path from "node:path";
+import { getPaths, type Paths } from "../lib/paths";
+import { readJsonFile } from "../lib/jsonFile-server";
+import { mapConcurrent } from "../lib/concurrency";
+import { assertChannelTextReadable } from "../lib/channelMedia";
+import { extractVideoId } from "../lib/videoId";
+import { PROJECT_NAME, PROJECT_URL } from "../lib/project";
+import { patchMetadataInfo } from "../lib/metadataHistory-server";
+import {
+ feedDateToTimestamp,
+ feedDateToUploadDate,
+ parseRssFeed,
+ type FeedItem,
+} from "../lib/rssFeed";
+import { readChannelConfig } from "./channels";
+import { listChannelVideoIds } from "./keptVideos";
+
+const INFO_JSON = "metadata.info.json";
+const READ_CONCURRENCY = 16;
+// One request, but not one that may hang a queue: a feed that has not
+// answered in this long is not going to.
+const FEED_FETCH_TIMEOUT_MS = 60_000;
+
+type Info = Record<string, unknown>;
+
+export type FeedMatchRule = "guid" | "enclosure" | "file-name";
+
+export type FeedMetadataPatch = {
+ title?: string;
+ description?: string;
+ duration?: number;
+ webpage_url?: string;
+ timestamp?: number;
+ upload_date?: string;
+};
+
+export type FeedRecord = { id: string; info: Info | null };
+
+export type FeedBackfillMatch = {
+ id: string;
+ by: FeedMatchRule;
+ item: FeedItem;
+ patch: FeedMetadataPatch;
+};
+
+export type FeedBackfillPlan = {
+ complete: string[];
+ matched: FeedBackfillMatch[];
+ unmatched: string[];
+ ambiguous: { id: string; rule: FeedMatchRule; candidates: number }[];
+ // A video dir with no readable metadata.info.json: nothing to complete.
+ noMetadata: string[];
+};
+
+// ── pure: what a record lacks ─────────────────────────────────────────────
+
+function str(v: unknown): string | null {
+ return typeof v === "string" && v.trim() ? v.trim() : null;
+}
+
+function stem(name: string): string {
+ const dot = name.lastIndexOf(".");
+ return dot > 0 ? name.slice(0, dot) : name;
+}
+
+// A title that only restates the file: what the generic extractor writes for a
+// direct media URL.
+export function isPlaceholderTitle(id: string, info: Info): boolean {
+ const title = str(info.title);
+ if (!title) return true;
+ const fileNames = [id, str(info.webpage_url_basename)].filter(
+ (s): s is string => s !== null,
+ );
+ const placeholders = new Set<string>([
+ ...fileNames,
+ ...fileNames.map(stem),
+ ...[str(info.id), str(info.display_id)].filter((s): s is string => s !== null),
+ ]);
+ return placeholders.has(title);
+}
+
+function hasUploadDate(info: Info): boolean {
+ return typeof info.upload_date === "string" && /^\d{8}$/.test(info.upload_date);
+}
+
+export function isRecordComplete(id: string, info: Info): boolean {
+ return hasUploadDate(info) && !isPlaceholderTitle(id, info);
+}
+
+// ── pure: matching ────────────────────────────────────────────────────────
+
+function urlKeys(raw: string | null): string[] {
+ if (!raw) return [];
+ try {
+ const u = new URL(raw);
+ u.hash = "";
+ const full = u.href;
+ u.search = "";
+ return full === u.href ? [full] : [full, u.href];
+ } catch {
+ return [];
+ }
+}
+
+function safeDecode(s: string): string {
+ try {
+ return decodeURIComponent(s);
+ } catch {
+ return s;
+ }
+}
+
+function fileNameKeys(raw: string | null): string[] {
+ if (!raw) return [];
+ const keys = new Set([raw, safeDecode(raw)]);
+ return [...keys];
+}
+
+type FeedIndex = Record<FeedMatchRule, Map<string, FeedItem[]>>;
+
+function add(map: Map<string, FeedItem[]>, key: string, item: FeedItem): void {
+ const list = map.get(key);
+ if (!list) map.set(key, [item]);
+ else if (!list.includes(item)) list.push(item);
+}
+
+export function indexFeedItems(items: readonly FeedItem[]): FeedIndex {
+ const index: FeedIndex = {
+ guid: new Map(),
+ enclosure: new Map(),
+ "file-name": new Map(),
+ };
+ for (const item of items) {
+ if (item.guid) add(index.guid, item.guid, item);
+ for (const k of urlKeys(item.enclosureUrl)) add(index.enclosure, k, item);
+ const fileName = item.enclosureUrl ? extractVideoId(item.enclosureUrl) : null;
+ for (const k of fileNameKeys(fileName)) add(index["file-name"], k, item);
+ }
+ return index;
+}
+
+const RULES: readonly FeedMatchRule[] = ["guid", "enclosure", "file-name"];
+
+function recordKeys(id: string, info: Info, rule: FeedMatchRule): string[] {
+ switch (rule) {
+ case "guid":
+ return [str(info.id), str(info.display_id)].filter(
+ (s): s is string => s !== null,
+ );
+ case "enclosure":
+ return [info.webpage_url, info.original_url, info.url].flatMap((u) =>
+ urlKeys(str(u)),
+ );
+ case "file-name":
+ return [id, str(info.webpage_url_basename)].flatMap((n) => fileNameKeys(n));
+ }
+}
+
+export type RecordMatch =
+ | { kind: "matched"; item: FeedItem; by: FeedMatchRule }
+ | { kind: "ambiguous"; rule: FeedMatchRule; candidates: number }
+ | { kind: "unmatched" };
+
+export function matchRecord(id: string, info: Info, index: FeedIndex): RecordMatch {
+ let ambiguous: RecordMatch | null = null;
+ for (const rule of RULES) {
+ const found = new Set<FeedItem>();
+ for (const key of recordKeys(id, info, rule)) {
+ for (const item of index[rule].get(key) ?? []) found.add(item);
+ }
+ if (found.size === 1) return { kind: "matched", item: [...found][0], by: rule };
+ if (found.size > 1 && !ambiguous) {
+ ambiguous = { kind: "ambiguous", rule, candidates: found.size };
+ }
+ }
+ return ambiguous ?? { kind: "unmatched" };
+}
+
+// ── pure: what to write ───────────────────────────────────────────────────
+
+// The fields a matched record lacks, from its item. Key order is the order
+// they are appended to a file that lacks them: the long description before the
+// date, so the date stays in the tail recencyIndex.ts reads.
+export function feedPatchFor(
+ id: string,
+ info: Info,
+ item: FeedItem,
+): FeedMetadataPatch {
+ const patch: FeedMetadataPatch = {};
+ if (item.title && isPlaceholderTitle(id, info)) patch.title = item.title;
+ if (item.description && !str(info.description)) {
+ patch.description = item.description;
+ }
+ const duration = info.duration;
+ if (
+ item.durationSeconds !== null &&
+ item.durationSeconds > 0 &&
+ !(typeof duration === "number" && duration > 0)
+ ) {
+ patch.duration = item.durationSeconds;
+ }
+ // See the header: a URL here must keep the record's id.
+ const webpage = [item.link, item.enclosureUrl].find(
+ (u): u is string => !!u && extractVideoId(u) === id,
+ );
+ if (webpage && webpage !== info.webpage_url) patch.webpage_url = webpage;
+ if (!hasUploadDate(info)) {
+ const uploadDate = feedDateToUploadDate(item.pubDate);
+ const timestamp = feedDateToTimestamp(item.pubDate);
+ if (uploadDate) {
+ if (timestamp !== null && typeof info.timestamp !== "number") {
+ patch.timestamp = timestamp;
+ }
+ patch.upload_date = uploadDate;
+ }
+ }
+ return patch;
+}
+
+export function planFeedBackfill(
+ records: readonly FeedRecord[],
+ items: readonly FeedItem[],
+): FeedBackfillPlan {
+ const index = indexFeedItems(items);
+ const plan: FeedBackfillPlan = {
+ complete: [],
+ matched: [],
+ unmatched: [],
+ ambiguous: [],
+ noMetadata: [],
+ };
+ for (const { id, info } of records) {
+ if (!info) {
+ plan.noMetadata.push(id);
+ continue;
+ }
+ if (isRecordComplete(id, info)) {
+ plan.complete.push(id);
+ continue;
+ }
+ const m = matchRecord(id, info, index);
+ if (m.kind === "matched") {
+ plan.matched.push({ id, by: m.by, item: m.item, patch: feedPatchFor(id, info, m.item) });
+ } else if (m.kind === "ambiguous") {
+ plan.ambiguous.push({ id, rule: m.rule, candidates: m.candidates });
+ } else {
+ plan.unmatched.push(id);
+ }
+ }
+ return plan;
+}
+
+// ── the fetch ─────────────────────────────────────────────────────────────
+
+export type FetchFeed = (url: string, signal?: AbortSignal) => Promise<string>;
+
+// The one request. Identified, bounded in time, and never retried here: a
+// failure is the job's failure, and the operator re-runs it.
+export const fetchFeedText: FetchFeed = async (url, signal) => {
+ const timeout = AbortSignal.timeout(FEED_FETCH_TIMEOUT_MS);
+ const res = await fetch(url, {
+ signal: signal ? AbortSignal.any([signal, timeout]) : timeout,
+ redirect: "follow",
+ headers: {
+ accept:
+ "application/rss+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.1",
+ "user-agent": `${PROJECT_NAME} feed backfill (+${PROJECT_URL})`,
+ },
+ });
+ if (!res.ok) {
+ throw new Error(`the feed answered HTTP ${res.status} ${res.statusText}`.trim());
+ }
+ return res.text();
+};
+
+// ── the run ───────────────────────────────────────────────────────────────
+
+export type FeedBackfillResult = {
+ slug: string;
+ feedUrl: string;
+ dryRun: boolean;
+ feedItems: number;
+ records: number;
+ complete: number;
+ matched: number;
+ written: number;
+ unchanged: number;
+ unmatched: string[];
+ ambiguous: string[];
+ noMetadata: number;
+ failed: { id: string; error: string }[];
+};
+
+export type FeedBackfillOpts = {
+ paths?: Paths;
+ slug: string;
+ // Default: the channel's configured url.
+ feedUrl?: string;
+ dryRun?: boolean;
+ // Who asked, for the history entry ("cli", "ops", …).
+ requestedBy?: string;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // The fetch, injectable so a test never touches the network.
+ fetchFeed?: FetchFeed;
+};
+
+function describePatch(p: FeedMetadataPatch): string {
+ const parts: string[] = [];
+ if (p.title !== undefined) parts.push(`title ${JSON.stringify(p.title)}`);
+ if (p.upload_date !== undefined) parts.push(`date ${p.upload_date}`);
+ if (p.duration !== undefined) parts.push(`duration ${p.duration}s`);
+ if (p.description !== undefined) parts.push(`description (${p.description.length} chars)`);
+ if (p.webpage_url !== undefined) parts.push("webpage_url");
+ return parts.join(", ") || "nothing it lacks";
+}
+
+export async function backfillFeedMetadata(
+ opts: FeedBackfillOpts,
+): Promise<FeedBackfillResult> {
+ const paths = opts.paths ?? getPaths();
+ const { slug } = opts;
+ const log = opts.onLog ?? (() => {});
+ const dryRun = opts.dryRun === true;
+ const config = await readChannelConfig(paths, slug);
+ if (!config) throw new Error(`Channel "${slug}" not found`);
+ const feedUrl = (opts.feedUrl ?? config.url ?? "").trim();
+ if (!/^https?:\/\//i.test(feedUrl)) {
+ throw new Error(
+ feedUrl
+ ? `"${feedUrl}" is not an http(s) feed URL`
+ : `Channel "${slug}" has no url; pass the feed URL`,
+ );
+ }
+ // AN UNREADABLE TEXT TIER IS NOT AN EMPTY CHANNEL (AGENTS.md): the walk below
+ // would find no records and report a clean run.
+ await assertChannelTextReadable(paths, slug, config);
+
+ const dataDir = path.join(paths.channelsDir, slug, "data");
+ const ids = await listChannelVideoIds(paths, slug);
+ const records = await mapConcurrent(ids, READ_CONCURRENCY, async (id) => {
+ const read = await readJsonFile(path.join(dataDir, id, INFO_JSON));
+ const value = read.ok ? read.value : null;
+ const info =
+ value && typeof value === "object" && !Array.isArray(value)
+ ? (value as Info)
+ : null;
+ return { id, info };
+ });
+ const incomplete = records.filter(
+ (r) => r.info && !isRecordComplete(r.id, r.info),
+ ).length;
+ log(
+ `${slug}: ${records.length} record(s), ${incomplete} lacking a title or a date.`,
+ );
+
+ const result: FeedBackfillResult = {
+ slug,
+ feedUrl,
+ dryRun,
+ feedItems: 0,
+ records: records.length,
+ complete: 0,
+ matched: 0,
+ written: 0,
+ unchanged: 0,
+ unmatched: [],
+ ambiguous: [],
+ noMetadata: 0,
+ failed: [],
+ };
+ if (incomplete === 0) {
+ // Nothing to complete is nothing to fetch: the feed is not asked.
+ result.complete = records.filter((r) => r.info).length;
+ result.noMetadata = records.length - result.complete;
+ log("Every record has a title and a date; the feed was not fetched.");
+ return result;
+ }
+
+ log(`Fetching the feed (one request): ${feedUrl}`);
+ const xml = await (opts.fetchFeed ?? fetchFeedText)(feedUrl, opts.signal);
+ const feed = parseRssFeed(xml);
+ if (feed.items.length === 0) {
+ throw new Error("the feed has no <item> — is this URL an RSS feed?");
+ }
+ result.feedItems = feed.items.length;
+ log(
+ `Feed${feed.title ? ` "${feed.title}"` : ""}: ${feed.items.length} item(s).`,
+ );
+
+ const plan = planFeedBackfill(records, feed.items);
+ result.complete = plan.complete.length;
+ result.matched = plan.matched.length;
+ result.unmatched = plan.unmatched;
+ result.ambiguous = plan.ambiguous.map((a) => a.id);
+ result.noMetadata = plan.noMetadata.length;
+
+ for (const m of plan.matched) {
+ if (opts.signal?.aborted) break;
+ const what = describePatch(m.patch);
+ if (dryRun) {
+ log(` would write ${m.id} (by ${m.by}): ${what}`);
+ continue;
+ }
+ if (Object.keys(m.patch).length === 0) {
+ result.unchanged += 1;
+ log(` ${m.id} (by ${m.by}): the item has nothing it lacks`);
+ continue;
+ }
+ try {
+ const { written } = await patchMetadataInfo(
+ path.join(dataDir, m.id),
+ { ...m.patch },
+ {
+ by: "feed-backfill",
+ ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}),
+ onLog: log,
+ },
+ );
+ if (written) {
+ result.written += 1;
+ log(` wrote ${m.id} (by ${m.by}): ${what}`);
+ } else {
+ result.unchanged += 1;
+ }
+ } catch (err) {
+ result.failed.push({ id: m.id, error: (err as Error).message });
+ log(` FAILED ${m.id}: ${(err as Error).message}`);
+ }
+ }
+ for (const a of plan.ambiguous) {
+ log(` ambiguous ${a.id}: ${a.candidates} items share its ${a.rule}`);
+ }
+ for (const id of plan.unmatched) log(` unmatched ${id}`);
+ return result;
+}
+
+export function feedBackfillSummary(r: FeedBackfillResult): string {
+ const head = r.dryRun ? "Dry run" : "Done";
+ const parts = [
+ `${r.matched} matched`,
+ `${r.unmatched.length} unmatched`,
+ `${r.complete} already complete`,
+ ];
+ if (r.ambiguous.length) parts.push(`${r.ambiguous.length} ambiguous`);
+ if (r.noMetadata) parts.push(`${r.noMetadata} without metadata`);
+ if (!r.dryRun) {
+ parts.push(`${r.written} written`);
+ if (r.unchanged) parts.push(`${r.unchanged} unchanged`);
+ if (r.failed.length) parts.push(`${r.failed.length} failed`);
+ }
+ let line = `${head}: ${r.slug} — ${parts.join(", ")} (of ${r.records} record(s)`;
+ line += r.feedItems ? `, ${r.feedItems} feed item(s)).` : ").";
+ if (!r.dryRun && r.written > 0) {
+ line += " Rebuild the index for the new titles and dates to reach search and the sites.";
+ }
+ return line;
+}
diff --git a/common/controller/feedMetadataJob.ts b/common/controller/feedMetadataJob.ts
@@ -0,0 +1,61 @@
+// THE FEED METADATA BACKFILL AS A JOB. The work is
+// controller/feedMetadataBackfill.ts, which the CLI runs offline; this is what
+// the editor's action and `/api/ops/feed-metadata` enqueue. Its own module so
+// the CLI never loads the job registry.
+
+import { getPaths, type Paths } from "../lib/paths";
+import { downloadQueueKey } from "../lib/queueKeys";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "../jobs/streamCommand";
+import { readChannelConfig } from "./channels";
+import {
+ backfillFeedMetadata,
+ feedBackfillSummary,
+} from "./feedMetadataBackfill";
+
+export const FEED_METADATA_JOB_KIND = "feed-metadata";
+
+// One job per run, on the channel's DOWNLOAD queue: the fetch contends with
+// that queue's own requests to the same host, and the writes must not land
+// beside a download rewriting the same metadata.info.json. `needsText` (the
+// kind's entry in jobs/jobKinds.ts) refuses it for a channel whose text cannot
+// be read. Not replayable: a re-run is the same click.
+export async function runFeedMetadataJob(opts: {
+ paths?: Paths;
+ slug: string;
+ feedUrl?: string;
+ dryRun?: boolean;
+ requestedBy?: string;
+ afterRun?: () => void | Promise<void>;
+}): Promise<StreamActionResult> {
+ const paths = opts.paths ?? getPaths();
+ const config = await readChannelConfig(paths, opts.slug);
+ if (!config) return { ok: false, error: `Channel "${opts.slug}" not found` };
+ return runManagedFunction({
+ kind: FEED_METADATA_JOB_KIND,
+ queueKey: downloadQueueKey(config),
+ paths,
+ channelSlug: opts.slug,
+ fn: async (onLog, signal) => {
+ onLog(
+ `${opts.dryRun ? "Previewing" : "Backfilling"} feed metadata for ${opts.slug}.`,
+ );
+ const result = await backfillFeedMetadata({
+ paths,
+ slug: opts.slug,
+ ...(opts.feedUrl ? { feedUrl: opts.feedUrl } : {}),
+ ...(opts.dryRun ? { dryRun: true } : {}),
+ ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}),
+ onLog,
+ signal,
+ });
+ onLog(feedBackfillSummary(result));
+ if (result.failed.length > 0) {
+ throw new Error(`${result.failed.length} record(s) could not be written`);
+ }
+ await opts.afterRun?.();
+ },
+ });
+}
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -201,3 +201,11 @@ test("release 17: every media kind keeps needsMedia, and none is also a text kin
assert.equal(kindNeedsText(k), false, k);
}
});
+
+test("the feed metadata backfill reads and writes the text tier only", () => {
+ assert.ok(getJobKind("feed-metadata"), "registered");
+ assert.equal(kindNeedsMedia("feed-metadata"), false);
+ assert.equal(kindNeedsText("feed-metadata"), true);
+ assert.equal(jobKindLabel("feed-metadata"), "Backfill feed metadata");
+ assert.equal(isDrainableKind("feed-metadata"), false);
+});
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -577,6 +577,25 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
needsMedia: false,
needsText: true,
},
+ // A podcast channel's records completed from its RSS feed
+ // (controller/feedMetadataBackfill.ts): one fetch of the feed, then a write
+ // of title, date, description and duration into each metadata.info.json that
+ // lacks them. On the platform download queue, like the metadata scan, so the
+ // fetch takes its turn with the host and the writes never land beside a
+ // download rewriting the same file. `needsText`: it reads and writes the
+ // text tier only, and against an unreadable one would report a clean run
+ // over zero records. Not drainable (one request and a few small writes); not
+ // replayable (a re-run is the same click, and finds the completed records
+ // complete).
+ "feed-metadata": {
+ kind: "feed-metadata",
+ label: "Backfill feed metadata",
+ drainable: false,
+ replayable: false,
+ queueKeyStrategy: "platform",
+ needsMedia: false,
+ needsText: true,
+ },
// THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They
// ran from /sites — the hub since release 7, the homepage since release 11 —
// with no entry here, so /jobs showed their raw machine kinds. The labels
diff --git a/common/lib/__fixtures__/demo-podcast.rss b/common/lib/__fixtures__/demo-podcast.rss
@@ -0,0 +1,54 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/">
+ <channel>
+ <title>Demo & Friends Podcast</title>
+ <atom:link href="https://feeds.example.com/demo.rss" rel="self" type="application/rss+xml"/>
+ <link>https://example.com</link>
+ <description>A demo feed for tests.</description>
+ <image>
+ <url>https://example.com/cover.jpg</url>
+ <title>Not the channel title</title>
+ </image>
+ <!-- a commented-out <item><title>Never an item</title></item> -->
+ <item>
+ <title><![CDATA[Episode 1: Cats & <Dogs>]]></title>
+ <link>https://example.com/episodes/one</link>
+ <guid isPermaLink="false">guid-0001</guid>
+ <pubDate>Tue, 05 Mar 2024 10:00:00 GMT</pubDate>
+ <enclosure url="https://cdn.example.com/audio/abc123.mp3?key=a&updated=1" length="1000" type="audio/mpeg"/>
+ <itunes:duration>01:02:03</itunes:duration>
+ <description><![CDATA[<p>First line & more.</p><p>Second line<br/>third</p>]]></description>
+ </item>
+ <item>
+ <title>Two & a half ’quotes’ "here"</title>
+ <itunes:title>Not the item title</itunes:title>
+ <guid isPermaLink="false">guid-0002</guid>
+ <pubDate>Mon, 04 Mar 2024 22:30:00 -0500</pubDate>
+ <enclosure url="https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2" type="audio/mpeg" length="2000"></enclosure>
+ <itunes:duration>45:30</itunes:duration>
+ <description><p>Hello &amp; goodbye</p></description>
+ </item>
+ <item>
+ <!-- <title>A trap</title> -->
+ <title>Episode three</title>
+ <link>https://example.com/episodes/three</link>
+ <guid isPermaLink="true">https://example.com/?p=3</guid>
+ <pubDate>Wed, 06 Mar 2024 08:00:00 +0000</pubDate>
+ <enclosure url='https://cdn.example.com/audio/ghi789.mp3' type='audio/mpeg' length='3000' />
+ <itunes:duration>3725</itunes:duration>
+ <itunes:summary>Summary only.</itunes:summary>
+ </item>
+ <item>
+ <title>Episode four</title>
+ <guid isPermaLink="false">guid-0004</guid>
+ <pubDate>Thu, 07 Mar 2024 12:00:00 GMT</pubDate>
+ <enclosure url="https://cdn.example.com/audio/jkl012.mp3" type="audio/mpeg" length="4000"/>
+ <description><![CDATA[A CDATA body that mentions </item> and ]] without ending.]]></description>
+ </item>
+ <item>
+ <title>Trailer</title>
+ <guid isPermaLink="false">guid-trailer</guid>
+ <pubDate>Fri, 01 Mar 2024 09:00:00 GMT</pubDate>
+ </item>
+ </channel>
+</rss>
diff --git a/common/lib/metadataHistory-server.test.ts b/common/lib/metadataHistory-server.test.ts
@@ -6,6 +6,7 @@ import path from "node:path";
import {
loadMetadataHistory,
metadataHistoryPath,
+ patchMetadataInfo,
withMetadataHistory,
} from "./metadataHistory-server";
@@ -126,3 +127,51 @@ test("a failure to record is logged, never thrown", async () => {
assert.match(log, /Could not record the metadata history/);
});
});
+
+test("patchMetadataInfo merges in place, appends new keys in order, and records the rewrite", async () => {
+ await withDir(async (dir) => {
+ const original = { id: "v1", title: "v1", formats: [{ url: "u" }], _version: { v: 1 } };
+ await writeFile(path.join(dir, INFO), JSON.stringify(original));
+ const out = await patchMetadataInfo(
+ dir,
+ { title: "Real", description: "About", upload_date: "20240305" },
+ { by: "feed-backfill", requestedBy: "test" },
+ );
+ assert.deepEqual(out, { written: true });
+ const text = await readFile(path.join(dir, INFO), "utf8");
+ const after = JSON.parse(text);
+ assert.deepEqual(Object.keys(after), ["id", "title", "formats", "_version", "description", "upload_date"]);
+ assert.equal(after.title, "Real");
+ assert.deepEqual(after.formats, original.formats);
+ assert.equal(text.endsWith("\n"), false, "compact, no trailing newline");
+ const h = await loadMetadataHistory(dir);
+ assert.equal(h?.entries.length, 1);
+ assert.equal(h!.entries[0].by, "feed-backfill");
+ assert.equal(h!.entries[0].requestedBy, "test");
+ assert.deepEqual(h!.entries[0].changed.title, { from: "v1", to: "Real" });
+ assert.deepEqual(h!.entries[0].added, { description: "About", upload_date: "20240305" });
+ // No temp file is left behind.
+ assert.deepEqual((await readdir(dir)).sort(), [INFO, "metadata.history.json"].sort());
+ });
+});
+
+test("patchMetadataInfo: a patch that changes nothing writes nothing", async () => {
+ await withDir(async (dir) => {
+ const text = JSON.stringify({ id: "v1", title: "Same" });
+ await writeFile(path.join(dir, INFO), text);
+ const out = await patchMetadataInfo(dir, { title: "Same", description: undefined }, { by: "feed-backfill" });
+ assert.deepEqual(out, { written: false });
+ assert.equal(await readFile(path.join(dir, INFO), "utf8"), text);
+ assert.equal(await loadMetadataHistory(dir), null);
+ });
+});
+
+test("patchMetadataInfo refuses a missing file or one that is not a JSON object", async () => {
+ await withDir(async (dir) => {
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /ENOENT/);
+ await writeFile(path.join(dir, INFO), "[1,2]");
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not a JSON object/);
+ await writeFile(path.join(dir, INFO), "{torn");
+ await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not JSON/);
+ });
+});
diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts
@@ -26,7 +26,7 @@
import path from "node:path";
import { readFile, stat } from "node:fs/promises";
-import { withJsonFileLock } from "./jsonFile-server";
+import { withJsonFileLock, writeJsonAtomic } from "./jsonFile-server";
import { sidecar, sidecarField } from "./sidecar-server";
import {
METADATA_HISTORY_FILENAME,
@@ -129,3 +129,56 @@ export async function withMetadataHistory<T>(
}
}
}
+
+// THE ONE WRITER OF metadata.info.json THAT IS NOT yt-dlp: merge `patch` into
+// the file on disk, inside `withMetadataHistory`, so the rewrite is recorded
+// exactly as a yt-dlp one is (by `ctx.by`, with what moved).
+//
+// A MERGE, NEVER A REPLACEMENT. Every key the file has stays where it is and
+// keeps its value unless `patch` names it; a key the file lacks is appended in
+// `patch`'s order. Order is not cosmetic: two readers scan the file's BYTES
+// rather than parse it — the title from a 16 KB head (videoTitles.ts) and the
+// upload_date from an 8 KB tail (recencyIndex.ts) — so a caller adding a long
+// description and a date names the description first.
+//
+// Compact JSON, no trailing newline — yt-dlp's own shape, near enough for
+// every reader (they all allow whitespace after a colon). A patch that changes
+// nothing writes nothing. A missing file, or one that is not a JSON object,
+// throws: there is no record to complete, and a write would invent one.
+//
+// Under the file's lock, so two patches of one record never drop each other.
+// It does NOT serialise against a yt-dlp spawn rewriting the same file; the
+// caller's job runs on the channel's download queue for that reason.
+export async function patchMetadataInfo(
+ videoDir: string,
+ patch: Record<string, unknown>,
+ ctx: MetadataRewriteContext & { onLog?: (line: string) => void },
+): Promise<{ written: boolean }> {
+ const file = path.join(videoDir, INFO_JSON);
+ return withJsonFileLock(file, async () => {
+ const text = await readFile(file, "utf8");
+ let current: unknown;
+ try {
+ current = JSON.parse(text);
+ } catch {
+ throw new Error(`${file} is not JSON`);
+ }
+ if (typeof current !== "object" || current === null || Array.isArray(current)) {
+ throw new Error(`${file} is not a JSON object`);
+ }
+ const before = current as Record<string, unknown>;
+ const next: Record<string, unknown> = { ...before };
+ let changed = false;
+ for (const [key, value] of Object.entries(patch)) {
+ if (value === undefined) continue;
+ if (JSON.stringify(before[key]) === JSON.stringify(value)) continue;
+ next[key] = value;
+ changed = true;
+ }
+ if (!changed) return { written: false };
+ await withMetadataHistory(videoDir, ctx, () =>
+ writeJsonAtomic(file, next, { indent: 0, newline: false }),
+ );
+ return { written: true };
+ });
+}
diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts
@@ -46,7 +46,8 @@ export const METADATA_HISTORY_CAP = 200;
// A stored value whose JSON is larger than this is replaced by its digest.
export const METADATA_HISTORY_MAX_VALUE_BYTES = 16 * 1024;
-// Who rewrote the file: the three yt-dlp spawns that write it.
+// Who rewrote the file: the three yt-dlp spawns that write it, and the one
+// writer that is not yt-dlp (`patchMetadataInfo`, metadataHistory-server.ts).
export const METADATA_HISTORY_WRITERS = [
// downloadOneManaged's attempt 0, on every managed download.
"prefetch",
@@ -55,6 +56,10 @@ export const METADATA_HISTORY_WRITERS = [
"audio-check",
// runYtdlp's legacy single-video `download-one-audio` job.
"download-one",
+ // The podcast feed backfill (controller/feedMetadataBackfill.ts): title,
+ // date, description and duration from the channel's RSS feed, for a record
+ // imported by its enclosure URL, which carries none of them.
+ "feed-backfill",
] as const;
export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];
diff --git a/common/lib/rssFeed.test.ts b/common/lib/rssFeed.test.ts
@@ -0,0 +1,120 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ feedDateToTimestamp,
+ feedDateToUploadDate,
+ htmlToPlainText,
+ parseItunesDuration,
+ parseRssFeed,
+} from "./rssFeed";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/rssFeed.test.ts
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FIXTURE = readFileSync(
+ path.join(HERE, "__fixtures__", "demo-podcast.rss"),
+ "utf8",
+);
+
+test("the fixture feed: channel title, and every item in order", () => {
+ const feed = parseRssFeed(FIXTURE);
+ assert.equal(feed.title, "Demo & Friends Podcast");
+ assert.deepEqual(
+ feed.items.map((i) => i.guid),
+ ["guid-0001", "guid-0002", "https://example.com/?p=3", "guid-0004", "guid-trailer"],
+ "a commented-out item is not an item, and a CDATA </item> ends nothing",
+ );
+});
+
+test("CDATA is verbatim; entities are decoded once outside it", () => {
+ const [one, two] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.title, "Episode 1: Cats & <Dogs>");
+ assert.equal(two.title, "Two & a half ’quotes’ \"here\"");
+ // An attribute's & is a URL's &.
+ assert.equal(
+ one.enclosureUrl,
+ "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1",
+ );
+});
+
+test("the item's own <title> wins over itunes:title, and a comment is no title", () => {
+ const items = parseRssFeed(FIXTURE).items;
+ assert.equal(items[1].title?.startsWith("Two"), true);
+ assert.equal(items[2].title, "Episode three");
+});
+
+test("descriptions become plain text: HTML in CDATA, escaped HTML, itunes:summary", () => {
+ const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.description, "First line & more.\nSecond line\nthird");
+ assert.equal(two.description, "Hello & goodbye");
+ assert.equal(three.description, "Summary only.");
+ assert.equal(
+ four.description,
+ "A CDATA body that mentions and ]] without ending.",
+ "the CDATA's literal </item> is markup to the plain-text pass, not the end of the item",
+ );
+ assert.equal(trailer.description, null);
+});
+
+test("itunes tags, links, enclosures in either quote style", () => {
+ const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.durationSeconds, 3723);
+ assert.equal(two.durationSeconds, 2730);
+ assert.equal(three.durationSeconds, 3725);
+ assert.equal(four.durationSeconds, null);
+ assert.equal(one.link, "https://example.com/episodes/one");
+ assert.equal(two.link, null);
+ assert.equal(three.enclosureUrl, "https://cdn.example.com/audio/ghi789.mp3");
+ assert.equal(
+ two.enclosureUrl,
+ "https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2",
+ );
+ assert.equal(trailer.enclosureUrl, null);
+ assert.equal(one.pubDate, "Tue, 05 Mar 2024 10:00:00 GMT");
+});
+
+test("parseItunesDuration: seconds, MM:SS, HH:MM:SS; junk is null", () => {
+ assert.equal(parseItunesDuration("3725"), 3725);
+ assert.equal(parseItunesDuration("45:30"), 2730);
+ assert.equal(parseItunesDuration("1:02:03"), 3723);
+ assert.equal(parseItunesDuration(" 90.6 "), 91);
+ assert.equal(parseItunesDuration("1 hour"), null);
+ assert.equal(parseItunesDuration(""), null);
+ assert.equal(parseItunesDuration(null), null);
+});
+
+test("upload_date is the UTC day of the publication instant", () => {
+ assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 10:00:00 GMT"), "20240305");
+ // 22:30 at -0500 is 03:30 the next day in UTC.
+ assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 -0500"), "20240305");
+ assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 EST"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05T23:30:00+02:00"), "20240305");
+ // No zone at all reads as UTC, never as this machine's zone.
+ assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 23:30:00"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05T23:30:00"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05"), "20240305");
+ assert.equal(feedDateToUploadDate("not a date"), null);
+ assert.equal(feedDateToUploadDate(null), null);
+ assert.equal(feedDateToTimestamp("Tue, 05 Mar 2024 10:00:00 GMT"), 1709632800);
+});
+
+test("htmlToPlainText: list items, blank-line runs, numeric entities", () => {
+ assert.equal(
+ htmlToPlainText("<ul><li>one</li><li>two</li></ul><p></p><p></p><p>© x</p>"),
+ "- one\n- two\n\n© x",
+ );
+ assert.equal(htmlToPlainText("plain text"), "plain text");
+});
+
+test("a self-closing Atom-style link is read from its href; a BOM is ignored", () => {
+ const feed = parseRssFeed(
+ "<rss><channel><title>T</title><item><title>A</title>" +
+ '<link href="https://example.com/a"/></item></channel></rss>',
+ );
+ assert.equal(feed.title, "T");
+ assert.equal(feed.items[0].link, "https://example.com/a");
+});
diff --git a/common/lib/rssFeed.ts b/common/lib/rssFeed.ts
@@ -0,0 +1,237 @@
+// A PODCAST RSS FEED, READ: the channel title and, per <item>, the six things
+// an episode's record wants from it — title, publication date, enclosure URL,
+// guid, itunes:duration and description (plus the item's <link>).
+//
+// PURE: text in, values out. The one fetch of a feed lives in
+// controller/feedMetadataBackfill.ts; everything here is testable with a
+// fixture string.
+//
+// A SMALL PARSER, NOT A DEPENDENCY. The repo has no XML parser and an RSS
+// item is a flat list of named children, so this reads exactly that and no
+// more: CDATA sections are lifted out first (they may hold `<`, `</item>` or
+// `]]` look-alikes that a tag scan would misread), comments are dropped, and an
+// element is found by its exact qualified name (`title` never matches
+// `itunes:title`). Entities are decoded once in text and attributes — the five
+// XML ones, numeric references, and the HTML names feeds use in practice.
+// What it will NOT do is validate: a feed that a strict parser would reject but
+// a podcast app would play is read the way the app reads it.
+
+import { decodeEntities } from "../social/xArticle";
+
+export type FeedItem = {
+ title: string | null;
+ // The item's own page, when the feed gives one.
+ link: string | null;
+ guid: string | null;
+ // As written in the feed (RFC 822 in RSS 2.0; some feeds write ISO 8601).
+ pubDate: string | null;
+ enclosureUrl: string | null;
+ durationSeconds: number | null;
+ // Plain text: a feed's description is usually HTML, and a record's
+ // description is read as text.
+ description: string | null;
+};
+
+export type ParsedFeed = {
+ title: string | null;
+ items: FeedItem[];
+};
+
+// U+0000 never appears in well-formed XML, so it is a safe placeholder fence.
+const CDATA_MARK = "\u0000";
+
+type Lifted = { text: string; cdata: string[] };
+
+// Replace every CDATA section with a numbered placeholder, then drop comments
+// (a comment INSIDE a CDATA section is text, which is why this order).
+function liftCdata(xml: string): Lifted {
+ const cdata: string[] = [];
+ const text = xml
+ .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, (_whole, body: string) => {
+ cdata.push(body);
+ return `${CDATA_MARK}${cdata.length - 1}${CDATA_MARK}`;
+ })
+ .replace(/<!--[\s\S]*?-->/g, "");
+ return { text, cdata };
+}
+
+function escapeRe(s: string): string {
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
+}
+
+// The raw inner text of the first `<name …>…</name>` in `block`, or "" for a
+// self-closing one, or null when there is none.
+function elementInner(block: string, name: string): string | null {
+ const re = new RegExp(
+ `<${escapeRe(name)}(?=[\\s/>])([^>]*?)(/?)>`,
+ "g",
+ );
+ const open = re.exec(block);
+ if (!open) return null;
+ if (open[2] === "/") return "";
+ const start = open.index + open[0].length;
+ const close = block.indexOf(`</${name}>`, start);
+ // An unclosed element is read to the end of the block rather than dropped.
+ return close < 0 ? block.slice(start) : block.slice(start, close);
+}
+
+// The attributes of the first `<name …>` in `block`, decoded.
+function elementAttrs(
+ block: string,
+ name: string,
+ cdata: string[],
+): Record<string, string> | null {
+ const open = new RegExp(`<${escapeRe(name)}(?=[\\s/>])([^>]*)>`).exec(block);
+ if (!open) return null;
+ const attrs: Record<string, string> = {};
+ const attrRe = /([^\s"'<>/=]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g;
+ for (const m of open[1].matchAll(attrRe)) {
+ attrs[m[1]] = decodeEntities(restoreCdata(m[2] ?? m[3] ?? "", cdata));
+ }
+ return attrs;
+}
+
+function restoreCdata(s: string, cdata: string[]): string {
+ return s.replace(
+ new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`, "g"),
+ (_w, i: string) => cdata[Number(i)] ?? "",
+ );
+}
+
+// An element's text: entity-decoded outside CDATA, verbatim inside it, with
+// any stray markup outside CDATA removed. Trimmed; "" reads as null.
+function textOf(inner: string | null, cdata: string[]): string | null {
+ if (inner === null) return null;
+ const parts = inner.split(new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`));
+ let out = "";
+ for (let i = 0; i < parts.length; i++) {
+ out +=
+ i % 2 === 1
+ ? (cdata[Number(parts[i])] ?? "")
+ : decodeEntities(parts[i].replace(/<[^>]*>/g, ""));
+ }
+ const trimmed = out.trim();
+ return trimmed ? trimmed : null;
+}
+
+function firstText(
+ block: string,
+ names: readonly string[],
+ cdata: string[],
+): string | null {
+ for (const name of names) {
+ const t = textOf(elementInner(block, name), cdata);
+ if (t) return t;
+ }
+ return null;
+}
+
+// itunes:duration: plain seconds ("3725"), "MM:SS" or "HH:MM:SS", fractions
+// allowed. Anything else is null, never a guess.
+export function parseItunesDuration(raw: string | null | undefined): number | null {
+ if (!raw) return null;
+ const s = raw.trim();
+ if (!/^\d+(?:\.\d+)?(?::\d{1,2}(?:\.\d+)?){0,2}$/.test(s)) return null;
+ const parts = s.split(":").map(Number);
+ let total = 0;
+ for (const p of parts) total = total * 60 + p;
+ return Number.isFinite(total) ? Math.round(total) : null;
+}
+
+// A publication date as a Date, or null. RFC 822 ("Tue, 05 Mar 2024 10:00:00
+// GMT", "+0000", "EST") and ISO 8601 are both what Date.parse reads; a date
+// with no zone at all is read as UTC rather than as this machine's zone.
+export function parseFeedDate(raw: string | null | undefined): Date | null {
+ if (!raw) return null;
+ let s = raw.trim().replace(/\s+/g, " ");
+ if (!s) return null;
+ const hasZone =
+ /(?:Z|[+-]\d{2}:?\d{2}|\b(?:GMT|UTC|UT|[ECMP][SD]T)\b)\s*$/i.test(s);
+ if (!hasZone) {
+ // ISO with a time takes a "Z"; a bare ISO date is already UTC to
+ // Date.parse; anything else (RFC 822 without a zone) takes " GMT".
+ if (/^\d{4}-\d{2}-\d{2}[T ]/.test(s)) s = `${s.slice(0, 10)}T${s.slice(11)}Z`;
+ else if (!/^\d{4}-\d{2}-\d{2}$/.test(s)) s = `${s} GMT`;
+ }
+ const ms = Date.parse(s);
+ return Number.isFinite(ms) ? new Date(ms) : null;
+}
+
+// yt-dlp's `upload_date`: YYYYMMDD of the publication instant, in UTC (as
+// yt-dlp derives it from a timestamp).
+export function feedDateToUploadDate(raw: string | null | undefined): string | null {
+ const d = parseFeedDate(raw);
+ if (!d) return null;
+ const y = d.getUTCFullYear();
+ if (y < 1900 || y > 9999) return null;
+ const mm = String(d.getUTCMonth() + 1).padStart(2, "0");
+ const dd = String(d.getUTCDate()).padStart(2, "0");
+ return `${y}${mm}${dd}`;
+}
+
+// yt-dlp's `timestamp`: whole seconds since the epoch.
+export function feedDateToTimestamp(raw: string | null | undefined): number | null {
+ const d = parseFeedDate(raw);
+ return d ? Math.floor(d.getTime() / 1000) : null;
+}
+
+// A feed description as plain text: block ends become line breaks, tags go,
+// entities are decoded (the HTML ones a description carries once its own XML
+// escaping is gone), and runs of blank lines collapse to one.
+export function htmlToPlainText(html: string): string {
+ return decodeEntities(
+ html
+ .replace(/<br\s*\/?>/gi, "\n")
+ .replace(/<\/(?:p|div|li|h[1-6]|blockquote|tr)>/gi, "\n")
+ .replace(/<li[^>]*>/gi, "- ")
+ .replace(/<[^>]*>/g, ""),
+ )
+ .replace(/ /g, " ")
+ .split("\n")
+ .map((line) => line.replace(/[ \t]+/g, " ").trim())
+ .join("\n")
+ .replace(/\n{3,}/g, "\n\n")
+ .trim();
+}
+
+function parseItem(block: string, cdata: string[]): FeedItem {
+ const enclosure =
+ elementAttrs(block, "enclosure", cdata)?.url ??
+ elementAttrs(block, "media:content", cdata)?.url ??
+ null;
+ const rawDescription = firstText(
+ block,
+ ["description", "itunes:summary", "content:encoded"],
+ cdata,
+ );
+ const description = rawDescription ? htmlToPlainText(rawDescription) : "";
+ return {
+ title: firstText(block, ["title", "itunes:title"], cdata),
+ link:
+ firstText(block, ["link"], cdata) ??
+ elementAttrs(block, "link", cdata)?.href ??
+ null,
+ guid: firstText(block, ["guid"], cdata),
+ pubDate: firstText(block, ["pubDate", "dc:date", "published"], cdata),
+ enclosureUrl: enclosure?.trim() ? enclosure.trim() : null,
+ durationSeconds: parseItunesDuration(
+ firstText(block, ["itunes:duration"], cdata),
+ ),
+ description: description || null,
+ };
+}
+
+export function parseRssFeed(xml: string): ParsedFeed {
+ const { text, cdata } = liftCdata(xml.replace(/^/, ""));
+ const items: FeedItem[] = [];
+ const itemRe = /<item(?=[\s>])[^>]*>([\s\S]*?)<\/item>/g;
+ let firstItemAt = -1;
+ for (const m of text.matchAll(itemRe)) {
+ if (firstItemAt < 0) firstItemAt = m.index ?? 0;
+ items.push(parseItem(m[1], cdata));
+ }
+ // The channel's own title is the first <title> before the first item (an
+ // <image> block's title comes after the channel's in every feed seen).
+ const head = firstItemAt < 0 ? text : text.slice(0, firstItemAt);
+ return { title: firstText(head, ["title"], cdata), items };
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -65,6 +65,7 @@
- **`archilyzer storage migrate-tier` brings a channel moved the old way onto the media tier: its text comes home to this disk, its big files stay on the drive.** Run it with the editor **stopped** — a real run refuses while the editor answers on its port (`ARCHILYZER_EDITOR_URL`, default `http://localhost:3001`) or while a job on disk belongs to a live process; a dry run only notes it. `archilyzer storage migrate-tier <channel>` does one channel, `--all` every channel still shown as "Media layout retired". Start with `--dry-run`: it writes nothing and prints, per channel, how much text, clip windows, scratch and source video would be copied to this disk, how many audio files and live-chat replays stay on the drive, and whether it fits. A real run copies everything but the audio, the raw live-chat replays and any dead `*.temp.*` left by a download's postprocessor into the channel's `data/` on this disk, checks the copy file by file and kind by kind, links each audio file and replay back under its old name with its own time, renames the drive's `<root>/<channel>/data` to `<root>/<channel>/media`, and records `mediaDir` in the channel's `config.json` in place of `dataDir`; then it prints the bytes copied by kind and the links made. Nothing needs re-indexing: every file keeps its time. It needs the copy plus the resume margin free above the disk gate's floor on this disk, and refuses before writing anything when that is not there. `--all` takes the smallest channels first and **stops before omnibased, rekietalaw and the-quartering-rumble**, printing the free space and what each of them would copy; go on with the channel's name, or `--all --include-large`. A run that is stopped or fails partway carries on where it left off when run again, and a channel already done is left alone. The drive keeps its copy of the text, and those dead temps, until you run it again with `--reclaim`, which deletes them and keeps the media. `--reclaim` takes only channels this command migrated, and deletes a copy only where the same file, at the same size, is in the channel's `data/` on this disk; anything else stays on the drive and is listed. Leave it until the editor has run on the migrated channels for a while: until then the drive's copy is a second one. **Superseded auto-subtitles are purged after the migration, not before:** a channel on the retired layout refuses `purge-superseded-auto-subs` like every text job, so a channel's superseded `en-orig` subtitles (7.6 GB on omnibased) are copied to this disk first and purged from there.
- **A video whose YouTube subtitles answer "Too Many Requests" (HTTP 429) is downloaded anyway, and YouTube is not put in a cooldown for it.** YouTube refuses a subtitle file per video while the video itself downloads fine; every one of the day's 429s on 2026-10-01 was a subtitle fetch, and each failed its download, put all of YouTube in a cooldown that reached 30 minutes, and deferred the video for 6 hours to fail the same way again. Now that refusal is noted and the download goes on to the audio, as for a video with no captions, so the transcription lane transcribes it; nothing platform-wide is backed off. The video's subtitles are deferred: **Download missing subs** skips them for 6 hours, and from the third time they are refused, for 7 days; a download from the video's own page still fetches them. The download lane's page lists them under **Deferred subtitles**, and the video page says how many times and when. **Download missing subs** also goes on to the next video when one video's subtitles are refused, instead of stopping. Any other subtitle failure (a 403, a missing file, a chat replay that fails) still fails the download, as before. Needs a rebuild and restart of the editor.
- **The download pace adapts to rate limits, a rate limit that outlasts the cooldown holds the platform, and the auto-download lane waits between downloads.** Every yt-dlp run against a platform now waits its platform's current pace between requests: 1 second for YouTube and Rumble, doubled by each real rate limit (up to 16 seconds) and eased back one step after every 5 clean downloads, and one step for every hour with no rate limit; a subtitle-only refusal never raises it. When a platform has failed three times in a row at the 30-minute cooldown, it is **held**: auto-download tries it once an hour instead of every 30 minutes, and until that try is due a manual **Sync**, download or metadata scan on it is refused with a sentence giving its time. A clean try lifts the hold — the lane's, or a manual Sync, download or scan once the try is due, which is how a hold ends while auto-download is off or has nothing to fetch on that platform — and so does a try whose video came down although its subtitles were refused. The lane page keeps a held platform listed until then (saying when the lane is off), with a **Clear hold** button that drops the hold, the cooldown and the raised pace at once and says so in a job log. The auto-download lane now waits **Sleep between downloads** between two downloads on one platform, as a channel's batch downloads always did, plus whatever the pace was raised by; batch downloads add that too. The lane page's **Rate-limit cooldown** box shows held platforms, the raised paces and the deferred subtitles, the lane says when it is idle because a platform is held or it is pausing between downloads, and `archilyzer doctor` warns about a platform in a cooldown or held, and about a raised pace. **Download missing subs** also waits that gap between videos. The four numbers are the new `pacing` block in `settings.json` (SETTINGS.md). **After the restart, YouTube may be held at its first real failure:** its cooldown count from before the update (the subtitle refusals) still stands, so one failure puts it straight past the cap — **Clear hold** on the download lane's page resets it, and a clean download does too. Needs a rebuild and restart of the editor.
+- **A podcast channel's episodes imported by their enclosure URL get their titles and dates back from the feed.** Such a record carried only its file name as its title and no upload date, so the index skipped it and every surface showed the file id. `archilyzer feeds backfill-metadata <slug> [--feed <url>] [--dry-run]`, `POST /api/ops/feed-metadata {slug, dryRun?}` and `pnpm ops feed-metadata` (the new `feed-metadata` job, on the channel's download queue) fetch the channel's RSS feed once — no media — and match each record lacking a date or a real title to an item by guid, enclosure URL or enclosure file name; a record more than one item fits is reported, not guessed. A match fills in what the record lacks: the title, the description as plain text, the duration (a measured one is kept), the upload date and timestamp in UTC, and `webpage_url` only when the new URL keeps the record's id. Every change is a `feed-backfill` entry in the video's metadata history; a dry run counts matched, unmatched and already complete and writes nothing. Rebuild the index afterwards for the titles and dates to reach search and the sites.
## [0.11.0] - 2026-09-30
- **Transcripts that arrived after a video was first seen are counted.** The stats behind the homepage, the hub and every site's charts were cached per video and refreshed only when the video's metadata changed, so a transcript that came later — a Whisper run days after the download, or a video downloaded after the last index build — never reached them, and a video with YouTube captions alone had no transcription date. Counts and charts were low; the homepage could show a site with 0 transcripts, 0 channels and 0 hours while it served its videos. A stat is now also redone whenever the index re-reads the video, every transcript has a date, and a captioned video is dated by when its captions arrived rather than by a later Normalize run, so its place on "Transcribed over time" can move. **After updating, rebuild and restart the editor before anything else:** until then, **Build stats dataset** runs the old code and would undo the new stats, while a site, hub or homepage build already runs the new code — and the first stats build of any kind re-reads every video once (about 10–30 minutes on a large archive; it can be stopped and picks up where it stopped). Then build the index, the stats, the homepage, the hub, and the sites.
diff --git a/editor/app/api/ops/feed-metadata/route.ts b/editor/app/api/ops/feed-metadata/route.ts
@@ -0,0 +1,21 @@
+import { feedMetadataBackfillAction } from "../../../channels/[slug]/pipelineActions";
+import { jobResponse, ops, optBool, reqSlug } from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug: string, dryRun?: boolean } -> { ok: true, jobId }
+//
+// A podcast channel's records completed from its RSS feed: one fetch of the
+// channel's url, then title, date, description and duration into each record
+// that lacks them. `dryRun` logs matched / unmatched / already complete and
+// writes nothing. Follow the job's log for the counts.
+export async function POST(request: Request) {
+ return ops(request, ["slug", "dryRun"], async (body) =>
+ jobResponse(
+ await feedMetadataBackfillAction(
+ reqSlug(body, "slug"),
+ optBool(body, "dryRun"),
+ ),
+ ),
+ );
+}
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -28,6 +28,7 @@ import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdl
import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob";
+import { runFeedMetadataJob } from "yt-dlp-transcript-common/controller/feedMetadataJob";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates";
import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy";
@@ -555,3 +556,50 @@ export async function importVideoAction(
},
});
}
+
+// A PODCAST CHANNEL'S RECORDS COMPLETED FROM ITS RSS FEED (the
+// `feed-metadata` job, controller/feedMetadataBackfill.ts): one fetch of the
+// channel's url, then title, date, description and duration written into each
+// record that lacks them, through the metadata history. `dryRun` counts
+// matched / unmatched / already complete and writes nothing.
+//
+// The feed is one request to the channel's host, so a held host or one in its
+// rate-limit cooldown is answered with a sentence, as the metadata scan is.
+export async function feedMetadataBackfillAction(
+ slug: string,
+ dryRun?: boolean,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const channelConfig = await readChannelConfig(paths, slug);
+ if (!channelConfig) {
+ return { ok: false, error: `Channel "${slug}" not found` };
+ }
+ if (!channelConfig.url) {
+ return { ok: false, error: "Channel has no `url` configured" };
+ }
+ const platform = detectPlatform(channelConfig.url) ?? "unknown";
+ const held = await heldPlatformRefusal(platform, "The feed backfill", paths);
+ if (held) return { ok: false, info: true, error: held };
+ const remainingMs = await platformCooldownRemainingMs(platform, paths);
+ if (remainingMs > 0) {
+ const secs = Math.ceil(remainingMs / 1000);
+ return {
+ ok: false,
+ info: true,
+ error: `${platform} is in a rate-limit cooldown (${secs}s remaining). Run the feed backfill once it lapses.`,
+ };
+ }
+ return runFeedMetadataJob({
+ paths,
+ slug,
+ ...(dryRun ? { dryRun: true } : {}),
+ requestedBy: "editor",
+ afterRun: () => {
+ safeRevalidate([
+ `/channels/${slug}`,
+ "/channels",
+ ["/channels/[slug]/videos/[id]", "page"],
+ ]);
+ },
+ });
+}
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -26,6 +26,7 @@
//
// pnpm ops sync --json '{"slug":"the-quartering"}' --wait
// pnpm ops metadata-scan --json '{"slug":"the-quartering"}'
+// pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait
// pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}'
// pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}'
// pnpm ops lane --json '{"lane":"download","held":true}'
@@ -95,6 +96,9 @@ const ACTIONS = [
"channel-config",
"metadata-scan",
"import-video",
+ // A podcast channel's records completed from its RSS feed ({slug, dryRun?}):
+ // one fetch of the feed, no media.
+ "feed-metadata",
"refresh-report",
"sync",
"download-missing",
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -436,3 +436,10 @@ test("persist-videos is a POST to its route, named in the usage", () => {
assert.equal(parseArgs(["persist-videos", "--file", "list.json"]).bodyFile, "list.json");
assert.match(usage(), /persist-videos/);
});
+
+test("feed-metadata posts {slug, dryRun} to /api/ops/feed-metadata", () => {
+ const p = parseArgs(["feed-metadata", "--json", '{"slug":"demo-channel","dryRun":true}']);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/feed-metadata");
+ assert.deepEqual(p.body, { slug: "demo-channel", dryRun: true });
+});