import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import { readFileSync } from "node:fs"; import { tmpdir } from "node:os"; import path from "node:path"; import { fileURLToPath } from "node:url"; import type { Paths } from "../lib/paths"; import type { FeedItem } from "../lib/rssFeed"; import { loadMetadataHistory } from "../lib/metadataHistory-server"; import { backfillFeedMetadata, feedBackfillSummary, feedPatchFor, isPlaceholderTitle, isRecordComplete, planFeedBackfill, type FetchFeed, } from "./feedMetadataBackfill"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test controller/feedMetadataBackfill.test.ts // // No network: the fetch is a fake that serves the fixture feed and counts. const HERE = path.dirname(fileURLToPath(import.meta.url)); const FEED_XML = readFileSync( path.join(HERE, "..", "lib", "__fixtures__", "demo-podcast.rss"), "utf8", ); const FEED_URL = "https://feeds.example.com/demo.rss"; const SLUG = "demo-channel"; // What the generic extractor writes for a direct .mp3 URL: the file name as // the title, no date. Key order as yt-dlp writes it. function genericInfo(fileName: string, extra: Record = {}) { return { id: `uuid-${fileName}`, title: fileName.replace(/\.mp3$/, ""), direct: true, formats: [{ format_id: "mpeg", url: `https://cdn.example.com/audio/${fileName}?key=k` }], webpage_url: `https://cdn.example.com/audio/${fileName}?key=k`, webpage_url_basename: fileName, extractor: "generic", _version: { version: "2026.01.01" }, ...extra, }; } type Seed = Record | null>; async function withCorpus( seed: Seed, fn: (paths: Paths, dataDir: string) => Promise, ): Promise { const dir = await mkdtemp(path.join(tmpdir(), "ttb-feed-")); const transcriptsDir = path.join(dir, "corpus"); const paths = { transcriptsDir, channelsDir: path.join(transcriptsDir, "channels"), } as Paths; const channelDir = path.join(paths.channelsDir, SLUG); const dataDir = path.join(channelDir, "data"); await mkdir(dataDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ name: "Demo", url: FEED_URL, handling: "transcribe" }), ); for (const [id, info] of Object.entries(seed)) { await mkdir(path.join(dataDir, id), { recursive: true }); if (info) { // yt-dlp's own separators, so a byte comparison means something. await writeFile( path.join(dataDir, id, "metadata.info.json"), JSON.stringify(info), ); } } try { await fn(paths, dataDir); } finally { await rm(dir, { recursive: true, force: true }); } } function fakeFetch(): FetchFeed & { calls: string[] } { const calls: string[] = []; const f = (async (url: string) => { calls.push(url); return FEED_XML; }) as FetchFeed & { calls: string[] }; f.calls = calls; return f; } const readInfo = async (dataDir: string, id: string) => JSON.parse(await readFile(path.join(dataDir, id, "metadata.info.json"), "utf8")) as Record< string, unknown >; const SEED: Seed = { // Matched by its enclosure URL: same path, a different signed query. "abc123.mp3": genericInfo("abc123.mp3"), // Matched by guid: yt-dlp forced the feed's guid as the id. "def456.mp3": genericInfo("def456.mp3", { id: "guid-0002", webpage_url: "https://cdn2.example.org/def456.mp3?key=q", }), // Matched by file name only, and already carries a measured duration. "ghi789.mp3": genericInfo("ghi789.mp3", { webpage_url: "https://mirror.example.org/x/ghi789.mp3", duration: 3700, }), "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" }, "zzz999.mp3": genericInfo("zzz999.mp3"), "nometa.mp3": null, }; test("a run completes every matched record, through the history, with one fetch", async () => { await withCorpus(SEED, async (paths, dataDir) => { const fetchFeed = fakeFetch(); const lines: string[] = []; const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed, requestedBy: "test", onLog: (l) => lines.push(l), }); assert.deepEqual(fetchFeed.calls, [FEED_URL], "the channel's url, once"); assert.equal(r.feedItems, 5); assert.equal(r.matched, 3); assert.equal(r.written, 3); assert.equal(r.complete, 1); assert.deepEqual(r.unmatched, ["zzz999.mp3"]); assert.equal(r.noMetadata, 1); assert.deepEqual(r.failed, []); const one = await readInfo(dataDir, "abc123.mp3"); assert.equal(one.title, "Episode 1: Cats & "); assert.equal(one.upload_date, "20240305"); assert.equal(one.timestamp, 1709632800); assert.equal(one.duration, 3723); assert.equal(one.description, "First line & more.\nSecond line\nthird"); // The item's would rename the dir (its id is "one"), so the // enclosure — whose file name IS the dir name — is the webpage_url. assert.equal( one.webpage_url, "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1", ); // Everything else is untouched, in place; new keys are appended with the // date LAST (the tail reader's 8 KB). const keys = Object.keys(one); assert.deepEqual(keys.slice(0, 8), Object.keys(genericInfo("abc123.mp3"))); assert.equal(keys.at(-1), "upload_date"); assert.deepEqual(one.formats, genericInfo("abc123.mp3").formats); const two = await readInfo(dataDir, "def456.mp3"); assert.equal(two.title, "Two & a half ’quotes’ \"here\""); assert.equal(two.upload_date, "20240305", "22:30 at -0500 is the next UTC day"); assert.equal(two.description, "Hello & goodbye"); const three = await readInfo(dataDir, "ghi789.mp3"); assert.equal(three.title, "Episode three"); assert.equal(three.duration, 3700, "a measured duration is kept"); assert.equal(three.upload_date, "20240306"); // Through the history: one entry per written record, by feed-backfill. const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3")); assert.equal(h?.entries.length, 1); const e = h!.entries[0]; assert.equal(e.by, "feed-backfill"); assert.equal(e.requestedBy, "test"); assert.deepEqual(e.changed.title, { from: "abc123", to: "Episode 1: Cats & " }); assert.equal(e.added.upload_date, "20240305"); // The complete and the unmatched records are not touched. assert.equal(await loadMetadataHistory(path.join(dataDir, "done.mp3")), null); assert.equal(await loadMetadataHistory(path.join(dataDir, "zzz999.mp3")), null); assert.match(feedBackfillSummary(r), /3 matched, 1 unmatched, 1 already complete/); assert.ok(lines.some((l) => l.includes("unmatched zzz999.mp3"))); }); }); test("a second run finds the records complete and still asks the feed only for the rest", async () => { await withCorpus(SEED, async (paths, dataDir) => { await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: fakeFetch() }); const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); const fetchFeed = fakeFetch(); const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed }); assert.equal(r.complete, 4); assert.equal(r.written, 0); assert.deepEqual(r.unmatched, ["zzz999.mp3"]); const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); assert.deepEqual(after, before, "a complete record is not rewritten"); const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3")); assert.equal(h?.entries.length, 1, "and no second history entry"); }); }); test("nothing incomplete: the feed is not fetched at all", async () => { await withCorpus( { "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" } }, async (paths) => { const fetchFeed = fakeFetch(); const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed }); assert.deepEqual(fetchFeed.calls, []); assert.equal(r.complete, 1); assert.equal(r.matched, 0); }, ); }); test("a dry run reports matched / unmatched / complete and writes nothing", async () => { await withCorpus(SEED, async (paths, dataDir) => { const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); const lines: string[] = []; const r = await backfillFeedMetadata({ paths, slug: SLUG, dryRun: true, fetchFeed: fakeFetch(), onLog: (l) => lines.push(l), }); assert.equal(r.matched, 3); assert.equal(r.unmatched.length, 1); assert.equal(r.complete, 1); assert.equal(r.written, 0); const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); assert.deepEqual(after, before); assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null); assert.ok(lines.some((l) => l.startsWith(" would write abc123.mp3 (by enclosure)"))); assert.match(feedBackfillSummary(r), /^Dry run: demo-channel — 3 matched, 1 unmatched, 1 already complete/); }); }); test("--feed overrides the channel's url; a non-http url and a feed with no items refuse", async () => { await withCorpus(SEED, async (paths, dataDir) => { const fetchFeed = fakeFetch(); await backfillFeedMetadata({ paths, slug: SLUG, feedUrl: "https://mirror.example.net/feed.xml", dryRun: true, fetchFeed, }); assert.deepEqual(fetchFeed.calls, ["https://mirror.example.net/feed.xml"]); await assert.rejects( backfillFeedMetadata({ paths, slug: SLUG, feedUrl: "file:///etc/passwd", fetchFeed }), /not an http\(s\) feed URL/, ); await assert.rejects( backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: async () => "not a feed", }), /no /, ); await assert.rejects( backfillFeedMetadata({ paths, slug: "no-such-channel", fetchFeed }), /not found/, ); // A failed fetch writes nothing. await assert.rejects( backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: async () => { throw new Error("the feed answered HTTP 503"); }, }), /503/, ); assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null); }); }); // ── the pure rules ────────────────────────────────────────────────────────── const item = (over: Partial): FeedItem => ({ title: "T", link: null, guid: null, pubDate: "Tue, 05 Mar 2024 10:00:00 GMT", enclosureUrl: null, durationSeconds: null, description: null, ...over, }); test("placeholder titles: the dir name, its stem, the extractor's ids, or none", () => { assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc" }), true); assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc.mp3" }), true); assert.equal(isPlaceholderTitle("abc.mp3", { title: "uuid-1", id: "uuid-1" }), true); assert.equal(isPlaceholderTitle("abc.mp3", { title: " " }), true); assert.equal(isPlaceholderTitle("abc.mp3", {}), true); assert.equal(isPlaceholderTitle("abc.mp3", { title: "A real title" }), false); assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "20240101" }), true); assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "2024" }), false); assert.equal(isRecordComplete("abc.mp3", { title: "abc", upload_date: "20240101" }), false); }); test("a rule that finds several items decides nothing; the next unique rule still can", () => { const dupA = item({ guid: "g-a", enclosureUrl: "https://a.example.com/x/same.mp3" }); const dupB = item({ guid: "g-b", enclosureUrl: "https://b.example.com/y/same.mp3" }); const plan = planFeedBackfill( [ { id: "same.mp3", info: { id: "nope", title: "same" } }, { id: "same.mp3", info: { id: "g-b", title: "same" } }, ], [dupA, dupB], ); assert.deepEqual(plan.ambiguous, [{ id: "same.mp3", rule: "file-name", candidates: 2 }]); assert.equal(plan.matched.length, 1); assert.equal(plan.matched[0].by, "guid"); assert.equal(plan.matched[0].item, dupB); }); test("a URL is compared without its fragment, then without its query", () => { const it = item({ enclosureUrl: "https://cdn.example.com/a/ep.mp3?updated=1" }); const plan = planFeedBackfill( [ { id: "x1", info: { webpage_url: "https://cdn.example.com/a/ep.mp3?updated=1#__youtubedl_smuggle=1" } }, { id: "x2", info: { url: "https://cdn.example.com/a/ep.mp3?key=signed" } }, { id: "x3", info: { webpage_url: "https://cdn.example.com/b/ep.mp3" } }, ], [it], ); assert.deepEqual(plan.matched.map((m) => [m.id, m.by]), [ ["x1", "enclosure"], ["x2", "enclosure"], ]); assert.deepEqual(plan.unmatched, ["x3"]); }); test("the patch fills only what is missing, and a URL only when it keeps the record's id", () => { const it = item({ title: "Real", description: "About it", durationSeconds: 60, link: "https://example.com/episodes/ep.mp3", enclosureUrl: "https://cdn.example.com/ep.mp3", }); // The link's id is the dir name: the link wins. assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, it).webpage_url, "https://example.com/episodes/ep.mp3"); // A show-page link would rename the dir: the enclosure is used instead. const showLink = { ...it, link: "https://example.com/show" }; assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, showLink).webpage_url, "https://cdn.example.com/ep.mp3"); // Neither keeps the id: webpage_url is left alone. assert.equal("webpage_url" in feedPatchFor("other.mp3", { title: "other" }, showLink), false); // A record with a real title and description keeps them; only the date goes in. assert.deepEqual( Object.keys( feedPatchFor("ep.mp3", { title: "Mine", description: "Mine too", duration: 59, webpage_url: "https://example.com/episodes/ep.mp3" }, it), ), ["timestamp", "upload_date"], ); // An item with an unreadable date adds no date. assert.equal("upload_date" in feedPatchFor("ep.mp3", { title: "ep" }, { ...it, pubDate: "soon" }), false); });