Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 3751f19e76a1a186483fb18482cc9dc819c59b6b
parent c8668fe58c84cd97129c41d13f02918e7b444e6f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 05:22:44 -0400

common: backfill a podcast channel's record metadata from its RSS feed

One fetch of the feed, then each record lacking a date or a real title is
matched to an item by guid, enclosure URL or enclosure file name (the first
rule that finds exactly one item wins) and completed where it lacks a title,
description, duration, upload_date and timestamp. Written through
patchMetadataInfo, the one writer of metadata.info.json that is not yt-dlp,
which merges in place and records the rewrite in metadata.history.json as
feed-backfill. webpage_url is only set to a URL whose id is the record's dir
name, since the snapshot reconciles every dir to that id. A new job kind,
feed-metadata, on the channel's download queue.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/feedMetadataBackfill.test.ts | 363+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/feedMetadataBackfill.ts | 505+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/feedMetadataJob.ts | 61+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.test.ts | 8++++++++
Mcommon/jobs/jobKinds.ts | 19+++++++++++++++++++
Mcommon/lib/metadataHistory-server.test.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/metadataHistory-server.ts | 55++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/lib/metadataHistory.ts | 7++++++-
8 files changed, 1065 insertions(+), 2 deletions(-)

diff --git a/common/controller/feedMetadataBackfill.test.ts b/common/controller/feedMetadataBackfill.test.ts @@ -0,0 +1,363 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; +import { readFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import type { Paths } from "../lib/paths"; +import type { FeedItem } from "../lib/rssFeed"; +import { loadMetadataHistory } from "../lib/metadataHistory-server"; +import { + backfillFeedMetadata, + feedBackfillSummary, + feedPatchFor, + isPlaceholderTitle, + isRecordComplete, + planFeedBackfill, + type FetchFeed, +} from "./feedMetadataBackfill"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/feedMetadataBackfill.test.ts +// +// No network: the fetch is a fake that serves the fixture feed and counts. + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FEED_XML = readFileSync( + path.join(HERE, "..", "lib", "__fixtures__", "demo-podcast.rss"), + "utf8", +); +const FEED_URL = "https://feeds.example.com/demo.rss"; +const SLUG = "demo-channel"; + +// What the generic extractor writes for a direct .mp3 URL: the file name as +// the title, no date. Key order as yt-dlp writes it. +function genericInfo(fileName: string, extra: Record<string, unknown> = {}) { + return { + id: `uuid-${fileName}`, + title: fileName.replace(/\.mp3$/, ""), + direct: true, + formats: [{ format_id: "mpeg", url: `https://cdn.example.com/audio/${fileName}?key=k` }], + webpage_url: `https://cdn.example.com/audio/${fileName}?key=k`, + webpage_url_basename: fileName, + extractor: "generic", + _version: { version: "2026.01.01" }, + ...extra, + }; +} + +type Seed = Record<string, Record<string, unknown> | null>; + +async function withCorpus( + seed: Seed, + fn: (paths: Paths, dataDir: string) => Promise<void>, +): Promise<void> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-feed-")); + const transcriptsDir = path.join(dir, "corpus"); + const paths = { + transcriptsDir, + channelsDir: path.join(transcriptsDir, "channels"), + } as Paths; + const channelDir = path.join(paths.channelsDir, SLUG); + const dataDir = path.join(channelDir, "data"); + await mkdir(dataDir, { recursive: true }); + await writeFile( + path.join(channelDir, "config.json"), + JSON.stringify({ name: "Demo", url: FEED_URL, handling: "transcribe" }), + ); + for (const [id, info] of Object.entries(seed)) { + await mkdir(path.join(dataDir, id), { recursive: true }); + if (info) { + // yt-dlp's own separators, so a byte comparison means something. + await writeFile( + path.join(dataDir, id, "metadata.info.json"), + JSON.stringify(info), + ); + } + } + try { + await fn(paths, dataDir); + } finally { + await rm(dir, { recursive: true, force: true }); + } +} + +function fakeFetch(): FetchFeed & { calls: string[] } { + const calls: string[] = []; + const f = (async (url: string) => { + calls.push(url); + return FEED_XML; + }) as FetchFeed & { calls: string[] }; + f.calls = calls; + return f; +} + +const readInfo = async (dataDir: string, id: string) => + JSON.parse(await readFile(path.join(dataDir, id, "metadata.info.json"), "utf8")) as Record< + string, + unknown + >; + +const SEED: Seed = { + // Matched by its enclosure URL: same path, a different signed query. + "abc123.mp3": genericInfo("abc123.mp3"), + // Matched by guid: yt-dlp forced the feed's guid as the id. + "def456.mp3": genericInfo("def456.mp3", { + id: "guid-0002", + webpage_url: "https://cdn2.example.org/def456.mp3?key=q", + }), + // Matched by file name only, and already carries a measured duration. + "ghi789.mp3": genericInfo("ghi789.mp3", { + webpage_url: "https://mirror.example.org/x/ghi789.mp3", + duration: 3700, + }), + "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" }, + "zzz999.mp3": genericInfo("zzz999.mp3"), + "nometa.mp3": null, +}; + +test("a run completes every matched record, through the history, with one fetch", async () => { + await withCorpus(SEED, async (paths, dataDir) => { + const fetchFeed = fakeFetch(); + const lines: string[] = []; + const r = await backfillFeedMetadata({ + paths, + slug: SLUG, + fetchFeed, + requestedBy: "test", + onLog: (l) => lines.push(l), + }); + assert.deepEqual(fetchFeed.calls, [FEED_URL], "the channel's url, once"); + assert.equal(r.feedItems, 5); + assert.equal(r.matched, 3); + assert.equal(r.written, 3); + assert.equal(r.complete, 1); + assert.deepEqual(r.unmatched, ["zzz999.mp3"]); + assert.equal(r.noMetadata, 1); + assert.deepEqual(r.failed, []); + + const one = await readInfo(dataDir, "abc123.mp3"); + assert.equal(one.title, "Episode 1: Cats & <Dogs>"); + assert.equal(one.upload_date, "20240305"); + assert.equal(one.timestamp, 1709632800); + assert.equal(one.duration, 3723); + assert.equal(one.description, "First line & more.\nSecond line\nthird"); + // The item's <link> would rename the dir (its id is "one"), so the + // enclosure — whose file name IS the dir name — is the webpage_url. + assert.equal( + one.webpage_url, + "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1", + ); + // Everything else is untouched, in place; new keys are appended with the + // date LAST (the tail reader's 8 KB). + const keys = Object.keys(one); + assert.deepEqual(keys.slice(0, 8), Object.keys(genericInfo("abc123.mp3"))); + assert.equal(keys.at(-1), "upload_date"); + assert.deepEqual(one.formats, genericInfo("abc123.mp3").formats); + + const two = await readInfo(dataDir, "def456.mp3"); + assert.equal(two.title, "Two & a half ’quotes’ \"here\""); + assert.equal(two.upload_date, "20240305", "22:30 at -0500 is the next UTC day"); + assert.equal(two.description, "Hello & goodbye"); + + const three = await readInfo(dataDir, "ghi789.mp3"); + assert.equal(three.title, "Episode three"); + assert.equal(three.duration, 3700, "a measured duration is kept"); + assert.equal(three.upload_date, "20240306"); + + // Through the history: one entry per written record, by feed-backfill. + const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3")); + assert.equal(h?.entries.length, 1); + const e = h!.entries[0]; + assert.equal(e.by, "feed-backfill"); + assert.equal(e.requestedBy, "test"); + assert.deepEqual(e.changed.title, { from: "abc123", to: "Episode 1: Cats & <Dogs>" }); + assert.equal(e.added.upload_date, "20240305"); + + // The complete and the unmatched records are not touched. + assert.equal(await loadMetadataHistory(path.join(dataDir, "done.mp3")), null); + assert.equal(await loadMetadataHistory(path.join(dataDir, "zzz999.mp3")), null); + assert.match(feedBackfillSummary(r), /3 matched, 1 unmatched, 1 already complete/); + assert.ok(lines.some((l) => l.includes("unmatched zzz999.mp3"))); + }); +}); + +test("a second run finds the records complete and still asks the feed only for the rest", async () => { + await withCorpus(SEED, async (paths, dataDir) => { + await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed: fakeFetch() }); + const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); + const fetchFeed = fakeFetch(); + const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed }); + assert.equal(r.complete, 4); + assert.equal(r.written, 0); + assert.deepEqual(r.unmatched, ["zzz999.mp3"]); + const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); + assert.deepEqual(after, before, "a complete record is not rewritten"); + const h = await loadMetadataHistory(path.join(dataDir, "abc123.mp3")); + assert.equal(h?.entries.length, 1, "and no second history entry"); + }); +}); + +test("nothing incomplete: the feed is not fetched at all", async () => { + await withCorpus( + { "done.mp3": { id: "done", title: "A real title", upload_date: "20240101" } }, + async (paths) => { + const fetchFeed = fakeFetch(); + const r = await backfillFeedMetadata({ paths, slug: SLUG, fetchFeed }); + assert.deepEqual(fetchFeed.calls, []); + assert.equal(r.complete, 1); + assert.equal(r.matched, 0); + }, + ); +}); + +test("a dry run reports matched / unmatched / complete and writes nothing", async () => { + await withCorpus(SEED, async (paths, dataDir) => { + const before = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); + const lines: string[] = []; + const r = await backfillFeedMetadata({ + paths, + slug: SLUG, + dryRun: true, + fetchFeed: fakeFetch(), + onLog: (l) => lines.push(l), + }); + assert.equal(r.matched, 3); + assert.equal(r.unmatched.length, 1); + assert.equal(r.complete, 1); + assert.equal(r.written, 0); + const after = await readFile(path.join(dataDir, "abc123.mp3", "metadata.info.json")); + assert.deepEqual(after, before); + assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null); + assert.ok(lines.some((l) => l.startsWith(" would write abc123.mp3 (by enclosure)"))); + assert.match(feedBackfillSummary(r), /^Dry run: demo-channel — 3 matched, 1 unmatched, 1 already complete/); + }); +}); + +test("--feed overrides the channel's url; a non-http url and a feed with no items refuse", async () => { + await withCorpus(SEED, async (paths, dataDir) => { + const fetchFeed = fakeFetch(); + await backfillFeedMetadata({ + paths, + slug: SLUG, + feedUrl: "https://mirror.example.net/feed.xml", + dryRun: true, + fetchFeed, + }); + assert.deepEqual(fetchFeed.calls, ["https://mirror.example.net/feed.xml"]); + + await assert.rejects( + backfillFeedMetadata({ paths, slug: SLUG, feedUrl: "file:///etc/passwd", fetchFeed }), + /not an http\(s\) feed URL/, + ); + await assert.rejects( + backfillFeedMetadata({ + paths, + slug: SLUG, + fetchFeed: async () => "<html><body>not a feed</body></html>", + }), + /no <item>/, + ); + await assert.rejects( + backfillFeedMetadata({ paths, slug: "no-such-channel", fetchFeed }), + /not found/, + ); + // A failed fetch writes nothing. + await assert.rejects( + backfillFeedMetadata({ + paths, + slug: SLUG, + fetchFeed: async () => { + throw new Error("the feed answered HTTP 503"); + }, + }), + /503/, + ); + assert.equal(await loadMetadataHistory(path.join(dataDir, "abc123.mp3")), null); + }); +}); + +// ── the pure rules ────────────────────────────────────────────────────────── + +const item = (over: Partial<FeedItem>): FeedItem => ({ + title: "T", + link: null, + guid: null, + pubDate: "Tue, 05 Mar 2024 10:00:00 GMT", + enclosureUrl: null, + durationSeconds: null, + description: null, + ...over, +}); + +test("placeholder titles: the dir name, its stem, the extractor's ids, or none", () => { + assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc" }), true); + assert.equal(isPlaceholderTitle("abc.mp3", { title: "abc.mp3" }), true); + assert.equal(isPlaceholderTitle("abc.mp3", { title: "uuid-1", id: "uuid-1" }), true); + assert.equal(isPlaceholderTitle("abc.mp3", { title: " " }), true); + assert.equal(isPlaceholderTitle("abc.mp3", {}), true); + assert.equal(isPlaceholderTitle("abc.mp3", { title: "A real title" }), false); + assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "20240101" }), true); + assert.equal(isRecordComplete("abc.mp3", { title: "A real title", upload_date: "2024" }), false); + assert.equal(isRecordComplete("abc.mp3", { title: "abc", upload_date: "20240101" }), false); +}); + +test("a rule that finds several items decides nothing; the next unique rule still can", () => { + const dupA = item({ guid: "g-a", enclosureUrl: "https://a.example.com/x/same.mp3" }); + const dupB = item({ guid: "g-b", enclosureUrl: "https://b.example.com/y/same.mp3" }); + const plan = planFeedBackfill( + [ + { id: "same.mp3", info: { id: "nope", title: "same" } }, + { id: "same.mp3", info: { id: "g-b", title: "same" } }, + ], + [dupA, dupB], + ); + assert.deepEqual(plan.ambiguous, [{ id: "same.mp3", rule: "file-name", candidates: 2 }]); + assert.equal(plan.matched.length, 1); + assert.equal(plan.matched[0].by, "guid"); + assert.equal(plan.matched[0].item, dupB); +}); + +test("a URL is compared without its fragment, then without its query", () => { + const it = item({ enclosureUrl: "https://cdn.example.com/a/ep.mp3?updated=1" }); + const plan = planFeedBackfill( + [ + { id: "x1", info: { webpage_url: "https://cdn.example.com/a/ep.mp3?updated=1#__youtubedl_smuggle=1" } }, + { id: "x2", info: { url: "https://cdn.example.com/a/ep.mp3?key=signed" } }, + { id: "x3", info: { webpage_url: "https://cdn.example.com/b/ep.mp3" } }, + ], + [it], + ); + assert.deepEqual(plan.matched.map((m) => [m.id, m.by]), [ + ["x1", "enclosure"], + ["x2", "enclosure"], + ]); + assert.deepEqual(plan.unmatched, ["x3"]); +}); + +test("the patch fills only what is missing, and a URL only when it keeps the record's id", () => { + const it = item({ + title: "Real", + description: "About it", + durationSeconds: 60, + link: "https://example.com/episodes/ep.mp3", + enclosureUrl: "https://cdn.example.com/ep.mp3", + }); + // The link's id is the dir name: the link wins. + assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, it).webpage_url, "https://example.com/episodes/ep.mp3"); + // A show-page link would rename the dir: the enclosure is used instead. + const showLink = { ...it, link: "https://example.com/show" }; + assert.equal(feedPatchFor("ep.mp3", { title: "ep" }, showLink).webpage_url, "https://cdn.example.com/ep.mp3"); + // Neither keeps the id: webpage_url is left alone. + assert.equal("webpage_url" in feedPatchFor("other.mp3", { title: "other" }, showLink), false); + // A record with a real title and description keeps them; only the date goes in. + assert.deepEqual( + Object.keys( + feedPatchFor("ep.mp3", { title: "Mine", description: "Mine too", duration: 59, webpage_url: "https://example.com/episodes/ep.mp3" }, it), + ), + ["timestamp", "upload_date"], + ); + // An item with an unreadable date adds no date. + assert.equal("upload_date" in feedPatchFor("ep.mp3", { title: "ep" }, { ...it, pubDate: "soon" }), false); +}); diff --git a/common/controller/feedMetadataBackfill.ts b/common/controller/feedMetadataBackfill.ts @@ -0,0 +1,505 @@ +// A PODCAST CHANNEL'S RECORDS, COMPLETED FROM ITS RSS FEED. +// +// An episode imported one by one by its enclosure URL (`import-video` with a +// direct .mp3 link) goes through yt-dlp's generic extractor, which knows +// nothing but the file: the record's title is the file name, and it has no +// upload_date, no description and no duration. The index skips a record with +// no upload_date, and every surface shows the file id as its title. The feed +// the episode came from has all four; this reads them back. +// +// ONE FETCH, OF ONE URL — the channel's configured url, or the one the caller +// names — and nothing else on the network: no enclosure is requested, no media +// is downloaded. The editor's job takes its turn on the channel's download +// queue (`downloadQueueKey`), behind whatever else is fetching from that host. +// +// WHICH RECORDS. A record is COMPLETE — and left alone — when it has an +// 8-digit upload_date and a title that is not a placeholder (the dir name, its +// stem, or yt-dlp's id). Every other record is matched to a feed item: +// +// guid the record's yt-dlp id (or display_id) is the item's guid — +// what yt-dlp itself records when it reads an episode out of the +// feed (it forces the guid as the id); +// enclosure one of the record's URLs (webpage_url, original_url, url) is the +// item's enclosure URL, compared without its fragment, then +// without its query; +// file name the record's dir name is the enclosure's file name, as the +// import named it (`extractVideoId` of the URL). +// +// The first rule that finds exactly ONE item wins. A rule that finds several +// (a feed listing the same file twice) decides nothing; a record no rule +// settles is reported as ambiguous or unmatched, never guessed. +// +// WHAT IS WRITTEN, and only where the record lacks it: the title, the +// description, the duration (a value measured from the media beats the +// feed's declared one, so an existing duration stays), the upload_date and +// timestamp (UTC, from pubDate) — and webpage_url, under one rule below. +// Through `patchMetadataInfo`, the one non-yt-dlp writer of +// metadata.info.json, so every change is in `metadata.history.json` as +// `feed-backfill`. +// +// WEBPAGE_URL IS THE VIDEO'S ID, NOT A LINK. The snapshot reconciles every +// video dir to `extractVideoId(webpage_url)` (reconcileVideoDirs.ts) and a +// re-download fetches it (undownloadedVideos.ts). So the item's <link> — an +// episode page, sometimes the show's page shared by every item — is written +// only when its id IS the record's dir name, else the enclosure URL under the +// same test, else nothing: a page link that renamed or merged the dir would +// cost the record, not decorate it. + +import path from "node:path"; +import { getPaths, type Paths } from "../lib/paths"; +import { readJsonFile } from "../lib/jsonFile-server"; +import { mapConcurrent } from "../lib/concurrency"; +import { assertChannelTextReadable } from "../lib/channelMedia"; +import { extractVideoId } from "../lib/videoId"; +import { PROJECT_NAME, PROJECT_URL } from "../lib/project"; +import { patchMetadataInfo } from "../lib/metadataHistory-server"; +import { + feedDateToTimestamp, + feedDateToUploadDate, + parseRssFeed, + type FeedItem, +} from "../lib/rssFeed"; +import { readChannelConfig } from "./channels"; +import { listChannelVideoIds } from "./keptVideos"; + +const INFO_JSON = "metadata.info.json"; +const READ_CONCURRENCY = 16; +// One request, but not one that may hang a queue: a feed that has not +// answered in this long is not going to. +const FEED_FETCH_TIMEOUT_MS = 60_000; + +type Info = Record<string, unknown>; + +export type FeedMatchRule = "guid" | "enclosure" | "file-name"; + +export type FeedMetadataPatch = { + title?: string; + description?: string; + duration?: number; + webpage_url?: string; + timestamp?: number; + upload_date?: string; +}; + +export type FeedRecord = { id: string; info: Info | null }; + +export type FeedBackfillMatch = { + id: string; + by: FeedMatchRule; + item: FeedItem; + patch: FeedMetadataPatch; +}; + +export type FeedBackfillPlan = { + complete: string[]; + matched: FeedBackfillMatch[]; + unmatched: string[]; + ambiguous: { id: string; rule: FeedMatchRule; candidates: number }[]; + // A video dir with no readable metadata.info.json: nothing to complete. + noMetadata: string[]; +}; + +// ── pure: what a record lacks ───────────────────────────────────────────── + +function str(v: unknown): string | null { + return typeof v === "string" && v.trim() ? v.trim() : null; +} + +function stem(name: string): string { + const dot = name.lastIndexOf("."); + return dot > 0 ? name.slice(0, dot) : name; +} + +// A title that only restates the file: what the generic extractor writes for a +// direct media URL. +export function isPlaceholderTitle(id: string, info: Info): boolean { + const title = str(info.title); + if (!title) return true; + const fileNames = [id, str(info.webpage_url_basename)].filter( + (s): s is string => s !== null, + ); + const placeholders = new Set<string>([ + ...fileNames, + ...fileNames.map(stem), + ...[str(info.id), str(info.display_id)].filter((s): s is string => s !== null), + ]); + return placeholders.has(title); +} + +function hasUploadDate(info: Info): boolean { + return typeof info.upload_date === "string" && /^\d{8}$/.test(info.upload_date); +} + +export function isRecordComplete(id: string, info: Info): boolean { + return hasUploadDate(info) && !isPlaceholderTitle(id, info); +} + +// ── pure: matching ──────────────────────────────────────────────────────── + +function urlKeys(raw: string | null): string[] { + if (!raw) return []; + try { + const u = new URL(raw); + u.hash = ""; + const full = u.href; + u.search = ""; + return full === u.href ? [full] : [full, u.href]; + } catch { + return []; + } +} + +function safeDecode(s: string): string { + try { + return decodeURIComponent(s); + } catch { + return s; + } +} + +function fileNameKeys(raw: string | null): string[] { + if (!raw) return []; + const keys = new Set([raw, safeDecode(raw)]); + return [...keys]; +} + +type FeedIndex = Record<FeedMatchRule, Map<string, FeedItem[]>>; + +function add(map: Map<string, FeedItem[]>, key: string, item: FeedItem): void { + const list = map.get(key); + if (!list) map.set(key, [item]); + else if (!list.includes(item)) list.push(item); +} + +export function indexFeedItems(items: readonly FeedItem[]): FeedIndex { + const index: FeedIndex = { + guid: new Map(), + enclosure: new Map(), + "file-name": new Map(), + }; + for (const item of items) { + if (item.guid) add(index.guid, item.guid, item); + for (const k of urlKeys(item.enclosureUrl)) add(index.enclosure, k, item); + const fileName = item.enclosureUrl ? extractVideoId(item.enclosureUrl) : null; + for (const k of fileNameKeys(fileName)) add(index["file-name"], k, item); + } + return index; +} + +const RULES: readonly FeedMatchRule[] = ["guid", "enclosure", "file-name"]; + +function recordKeys(id: string, info: Info, rule: FeedMatchRule): string[] { + switch (rule) { + case "guid": + return [str(info.id), str(info.display_id)].filter( + (s): s is string => s !== null, + ); + case "enclosure": + return [info.webpage_url, info.original_url, info.url].flatMap((u) => + urlKeys(str(u)), + ); + case "file-name": + return [id, str(info.webpage_url_basename)].flatMap((n) => fileNameKeys(n)); + } +} + +export type RecordMatch = + | { kind: "matched"; item: FeedItem; by: FeedMatchRule } + | { kind: "ambiguous"; rule: FeedMatchRule; candidates: number } + | { kind: "unmatched" }; + +export function matchRecord(id: string, info: Info, index: FeedIndex): RecordMatch { + let ambiguous: RecordMatch | null = null; + for (const rule of RULES) { + const found = new Set<FeedItem>(); + for (const key of recordKeys(id, info, rule)) { + for (const item of index[rule].get(key) ?? []) found.add(item); + } + if (found.size === 1) return { kind: "matched", item: [...found][0], by: rule }; + if (found.size > 1 && !ambiguous) { + ambiguous = { kind: "ambiguous", rule, candidates: found.size }; + } + } + return ambiguous ?? { kind: "unmatched" }; +} + +// ── pure: what to write ─────────────────────────────────────────────────── + +// The fields a matched record lacks, from its item. Key order is the order +// they are appended to a file that lacks them: the long description before the +// date, so the date stays in the tail recencyIndex.ts reads. +export function feedPatchFor( + id: string, + info: Info, + item: FeedItem, +): FeedMetadataPatch { + const patch: FeedMetadataPatch = {}; + if (item.title && isPlaceholderTitle(id, info)) patch.title = item.title; + if (item.description && !str(info.description)) { + patch.description = item.description; + } + const duration = info.duration; + if ( + item.durationSeconds !== null && + item.durationSeconds > 0 && + !(typeof duration === "number" && duration > 0) + ) { + patch.duration = item.durationSeconds; + } + // See the header: a URL here must keep the record's id. + const webpage = [item.link, item.enclosureUrl].find( + (u): u is string => !!u && extractVideoId(u) === id, + ); + if (webpage && webpage !== info.webpage_url) patch.webpage_url = webpage; + if (!hasUploadDate(info)) { + const uploadDate = feedDateToUploadDate(item.pubDate); + const timestamp = feedDateToTimestamp(item.pubDate); + if (uploadDate) { + if (timestamp !== null && typeof info.timestamp !== "number") { + patch.timestamp = timestamp; + } + patch.upload_date = uploadDate; + } + } + return patch; +} + +export function planFeedBackfill( + records: readonly FeedRecord[], + items: readonly FeedItem[], +): FeedBackfillPlan { + const index = indexFeedItems(items); + const plan: FeedBackfillPlan = { + complete: [], + matched: [], + unmatched: [], + ambiguous: [], + noMetadata: [], + }; + for (const { id, info } of records) { + if (!info) { + plan.noMetadata.push(id); + continue; + } + if (isRecordComplete(id, info)) { + plan.complete.push(id); + continue; + } + const m = matchRecord(id, info, index); + if (m.kind === "matched") { + plan.matched.push({ id, by: m.by, item: m.item, patch: feedPatchFor(id, info, m.item) }); + } else if (m.kind === "ambiguous") { + plan.ambiguous.push({ id, rule: m.rule, candidates: m.candidates }); + } else { + plan.unmatched.push(id); + } + } + return plan; +} + +// ── the fetch ───────────────────────────────────────────────────────────── + +export type FetchFeed = (url: string, signal?: AbortSignal) => Promise<string>; + +// The one request. Identified, bounded in time, and never retried here: a +// failure is the job's failure, and the operator re-runs it. +export const fetchFeedText: FetchFeed = async (url, signal) => { + const timeout = AbortSignal.timeout(FEED_FETCH_TIMEOUT_MS); + const res = await fetch(url, { + signal: signal ? AbortSignal.any([signal, timeout]) : timeout, + redirect: "follow", + headers: { + accept: + "application/rss+xml, application/xml;q=0.9, text/xml;q=0.8, */*;q=0.1", + "user-agent": `${PROJECT_NAME} feed backfill (+${PROJECT_URL})`, + }, + }); + if (!res.ok) { + throw new Error(`the feed answered HTTP ${res.status} ${res.statusText}`.trim()); + } + return res.text(); +}; + +// ── the run ─────────────────────────────────────────────────────────────── + +export type FeedBackfillResult = { + slug: string; + feedUrl: string; + dryRun: boolean; + feedItems: number; + records: number; + complete: number; + matched: number; + written: number; + unchanged: number; + unmatched: string[]; + ambiguous: string[]; + noMetadata: number; + failed: { id: string; error: string }[]; +}; + +export type FeedBackfillOpts = { + paths?: Paths; + slug: string; + // Default: the channel's configured url. + feedUrl?: string; + dryRun?: boolean; + // Who asked, for the history entry ("cli", "ops", …). + requestedBy?: string; + onLog?: (line: string) => void; + signal?: AbortSignal; + // The fetch, injectable so a test never touches the network. + fetchFeed?: FetchFeed; +}; + +function describePatch(p: FeedMetadataPatch): string { + const parts: string[] = []; + if (p.title !== undefined) parts.push(`title ${JSON.stringify(p.title)}`); + if (p.upload_date !== undefined) parts.push(`date ${p.upload_date}`); + if (p.duration !== undefined) parts.push(`duration ${p.duration}s`); + if (p.description !== undefined) parts.push(`description (${p.description.length} chars)`); + if (p.webpage_url !== undefined) parts.push("webpage_url"); + return parts.join(", ") || "nothing it lacks"; +} + +export async function backfillFeedMetadata( + opts: FeedBackfillOpts, +): Promise<FeedBackfillResult> { + const paths = opts.paths ?? getPaths(); + const { slug } = opts; + const log = opts.onLog ?? (() => {}); + const dryRun = opts.dryRun === true; + const config = await readChannelConfig(paths, slug); + if (!config) throw new Error(`Channel "${slug}" not found`); + const feedUrl = (opts.feedUrl ?? config.url ?? "").trim(); + if (!/^https?:\/\//i.test(feedUrl)) { + throw new Error( + feedUrl + ? `"${feedUrl}" is not an http(s) feed URL` + : `Channel "${slug}" has no url; pass the feed URL`, + ); + } + // AN UNREADABLE TEXT TIER IS NOT AN EMPTY CHANNEL (AGENTS.md): the walk below + // would find no records and report a clean run. + await assertChannelTextReadable(paths, slug, config); + + const dataDir = path.join(paths.channelsDir, slug, "data"); + const ids = await listChannelVideoIds(paths, slug); + const records = await mapConcurrent(ids, READ_CONCURRENCY, async (id) => { + const read = await readJsonFile(path.join(dataDir, id, INFO_JSON)); + const value = read.ok ? read.value : null; + const info = + value && typeof value === "object" && !Array.isArray(value) + ? (value as Info) + : null; + return { id, info }; + }); + const incomplete = records.filter( + (r) => r.info && !isRecordComplete(r.id, r.info), + ).length; + log( + `${slug}: ${records.length} record(s), ${incomplete} lacking a title or a date.`, + ); + + const result: FeedBackfillResult = { + slug, + feedUrl, + dryRun, + feedItems: 0, + records: records.length, + complete: 0, + matched: 0, + written: 0, + unchanged: 0, + unmatched: [], + ambiguous: [], + noMetadata: 0, + failed: [], + }; + if (incomplete === 0) { + // Nothing to complete is nothing to fetch: the feed is not asked. + result.complete = records.filter((r) => r.info).length; + result.noMetadata = records.length - result.complete; + log("Every record has a title and a date; the feed was not fetched."); + return result; + } + + log(`Fetching the feed (one request): ${feedUrl}`); + const xml = await (opts.fetchFeed ?? fetchFeedText)(feedUrl, opts.signal); + const feed = parseRssFeed(xml); + if (feed.items.length === 0) { + throw new Error("the feed has no <item> — is this URL an RSS feed?"); + } + result.feedItems = feed.items.length; + log( + `Feed${feed.title ? ` "${feed.title}"` : ""}: ${feed.items.length} item(s).`, + ); + + const plan = planFeedBackfill(records, feed.items); + result.complete = plan.complete.length; + result.matched = plan.matched.length; + result.unmatched = plan.unmatched; + result.ambiguous = plan.ambiguous.map((a) => a.id); + result.noMetadata = plan.noMetadata.length; + + for (const m of plan.matched) { + if (opts.signal?.aborted) break; + const what = describePatch(m.patch); + if (dryRun) { + log(` would write ${m.id} (by ${m.by}): ${what}`); + continue; + } + if (Object.keys(m.patch).length === 0) { + result.unchanged += 1; + log(` ${m.id} (by ${m.by}): the item has nothing it lacks`); + continue; + } + try { + const { written } = await patchMetadataInfo( + path.join(dataDir, m.id), + { ...m.patch }, + { + by: "feed-backfill", + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onLog: log, + }, + ); + if (written) { + result.written += 1; + log(` wrote ${m.id} (by ${m.by}): ${what}`); + } else { + result.unchanged += 1; + } + } catch (err) { + result.failed.push({ id: m.id, error: (err as Error).message }); + log(` FAILED ${m.id}: ${(err as Error).message}`); + } + } + for (const a of plan.ambiguous) { + log(` ambiguous ${a.id}: ${a.candidates} items share its ${a.rule}`); + } + for (const id of plan.unmatched) log(` unmatched ${id}`); + return result; +} + +export function feedBackfillSummary(r: FeedBackfillResult): string { + const head = r.dryRun ? "Dry run" : "Done"; + const parts = [ + `${r.matched} matched`, + `${r.unmatched.length} unmatched`, + `${r.complete} already complete`, + ]; + if (r.ambiguous.length) parts.push(`${r.ambiguous.length} ambiguous`); + if (r.noMetadata) parts.push(`${r.noMetadata} without metadata`); + if (!r.dryRun) { + parts.push(`${r.written} written`); + if (r.unchanged) parts.push(`${r.unchanged} unchanged`); + if (r.failed.length) parts.push(`${r.failed.length} failed`); + } + let line = `${head}: ${r.slug} — ${parts.join(", ")} (of ${r.records} record(s)`; + line += r.feedItems ? `, ${r.feedItems} feed item(s)).` : ")."; + if (!r.dryRun && r.written > 0) { + line += " Rebuild the index for the new titles and dates to reach search and the sites."; + } + return line; +} diff --git a/common/controller/feedMetadataJob.ts b/common/controller/feedMetadataJob.ts @@ -0,0 +1,61 @@ +// THE FEED METADATA BACKFILL AS A JOB. The work is +// controller/feedMetadataBackfill.ts, which the CLI runs offline; this is what +// the editor's action and `/api/ops/feed-metadata` enqueue. Its own module so +// the CLI never loads the job registry. + +import { getPaths, type Paths } from "../lib/paths"; +import { downloadQueueKey } from "../lib/queueKeys"; +import { + runManagedFunction, + type StreamActionResult, +} from "../jobs/streamCommand"; +import { readChannelConfig } from "./channels"; +import { + backfillFeedMetadata, + feedBackfillSummary, +} from "./feedMetadataBackfill"; + +export const FEED_METADATA_JOB_KIND = "feed-metadata"; + +// One job per run, on the channel's DOWNLOAD queue: the fetch contends with +// that queue's own requests to the same host, and the writes must not land +// beside a download rewriting the same metadata.info.json. `needsText` (the +// kind's entry in jobs/jobKinds.ts) refuses it for a channel whose text cannot +// be read. Not replayable: a re-run is the same click. +export async function runFeedMetadataJob(opts: { + paths?: Paths; + slug: string; + feedUrl?: string; + dryRun?: boolean; + requestedBy?: string; + afterRun?: () => void | Promise<void>; +}): Promise<StreamActionResult> { + const paths = opts.paths ?? getPaths(); + const config = await readChannelConfig(paths, opts.slug); + if (!config) return { ok: false, error: `Channel "${opts.slug}" not found` }; + return runManagedFunction({ + kind: FEED_METADATA_JOB_KIND, + queueKey: downloadQueueKey(config), + paths, + channelSlug: opts.slug, + fn: async (onLog, signal) => { + onLog( + `${opts.dryRun ? "Previewing" : "Backfilling"} feed metadata for ${opts.slug}.`, + ); + const result = await backfillFeedMetadata({ + paths, + slug: opts.slug, + ...(opts.feedUrl ? { feedUrl: opts.feedUrl } : {}), + ...(opts.dryRun ? { dryRun: true } : {}), + ...(opts.requestedBy ? { requestedBy: opts.requestedBy } : {}), + onLog, + signal, + }); + onLog(feedBackfillSummary(result)); + if (result.failed.length > 0) { + throw new Error(`${result.failed.length} record(s) could not be written`); + } + await opts.afterRun?.(); + }, + }); +} diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -201,3 +201,11 @@ test("release 17: every media kind keeps needsMedia, and none is also a text kin assert.equal(kindNeedsText(k), false, k); } }); + +test("the feed metadata backfill reads and writes the text tier only", () => { + assert.ok(getJobKind("feed-metadata"), "registered"); + assert.equal(kindNeedsMedia("feed-metadata"), false); + assert.equal(kindNeedsText("feed-metadata"), true); + assert.equal(jobKindLabel("feed-metadata"), "Backfill feed metadata"); + assert.equal(isDrainableKind("feed-metadata"), false); +}); diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -577,6 +577,25 @@ const JOB_KINDS: Record<string, JobKindMeta> = { needsMedia: false, needsText: true, }, + // A podcast channel's records completed from its RSS feed + // (controller/feedMetadataBackfill.ts): one fetch of the feed, then a write + // of title, date, description and duration into each metadata.info.json that + // lacks them. On the platform download queue, like the metadata scan, so the + // fetch takes its turn with the host and the writes never land beside a + // download rewriting the same file. `needsText`: it reads and writes the + // text tier only, and against an unreadable one would report a clean run + // over zero records. Not drainable (one request and a few small writes); not + // replayable (a re-run is the same click, and finds the completed records + // complete). + "feed-metadata": { + kind: "feed-metadata", + label: "Backfill feed metadata", + drainable: false, + replayable: false, + queueKeyStrategy: "platform", + needsMedia: false, + needsText: true, + }, // THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They // ran from /sites — the hub since release 7, the homepage since release 11 — // with no entry here, so /jobs showed their raw machine kinds. The labels diff --git a/common/lib/metadataHistory-server.test.ts b/common/lib/metadataHistory-server.test.ts @@ -6,6 +6,7 @@ import path from "node:path"; import { loadMetadataHistory, metadataHistoryPath, + patchMetadataInfo, withMetadataHistory, } from "./metadataHistory-server"; @@ -126,3 +127,51 @@ test("a failure to record is logged, never thrown", async () => { assert.match(log, /Could not record the metadata history/); }); }); + +test("patchMetadataInfo merges in place, appends new keys in order, and records the rewrite", async () => { + await withDir(async (dir) => { + const original = { id: "v1", title: "v1", formats: [{ url: "u" }], _version: { v: 1 } }; + await writeFile(path.join(dir, INFO), JSON.stringify(original)); + const out = await patchMetadataInfo( + dir, + { title: "Real", description: "About", upload_date: "20240305" }, + { by: "feed-backfill", requestedBy: "test" }, + ); + assert.deepEqual(out, { written: true }); + const text = await readFile(path.join(dir, INFO), "utf8"); + const after = JSON.parse(text); + assert.deepEqual(Object.keys(after), ["id", "title", "formats", "_version", "description", "upload_date"]); + assert.equal(after.title, "Real"); + assert.deepEqual(after.formats, original.formats); + assert.equal(text.endsWith("\n"), false, "compact, no trailing newline"); + const h = await loadMetadataHistory(dir); + assert.equal(h?.entries.length, 1); + assert.equal(h!.entries[0].by, "feed-backfill"); + assert.equal(h!.entries[0].requestedBy, "test"); + assert.deepEqual(h!.entries[0].changed.title, { from: "v1", to: "Real" }); + assert.deepEqual(h!.entries[0].added, { description: "About", upload_date: "20240305" }); + // No temp file is left behind. + assert.deepEqual((await readdir(dir)).sort(), [INFO, "metadata.history.json"].sort()); + }); +}); + +test("patchMetadataInfo: a patch that changes nothing writes nothing", async () => { + await withDir(async (dir) => { + const text = JSON.stringify({ id: "v1", title: "Same" }); + await writeFile(path.join(dir, INFO), text); + const out = await patchMetadataInfo(dir, { title: "Same", description: undefined }, { by: "feed-backfill" }); + assert.deepEqual(out, { written: false }); + assert.equal(await readFile(path.join(dir, INFO), "utf8"), text); + assert.equal(await loadMetadataHistory(dir), null); + }); +}); + +test("patchMetadataInfo refuses a missing file or one that is not a JSON object", async () => { + await withDir(async (dir) => { + await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /ENOENT/); + await writeFile(path.join(dir, INFO), "[1,2]"); + await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not a JSON object/); + await writeFile(path.join(dir, INFO), "{torn"); + await assert.rejects(patchMetadataInfo(dir, { title: "x" }, { by: "feed-backfill" }), /not JSON/); + }); +}); diff --git a/common/lib/metadataHistory-server.ts b/common/lib/metadataHistory-server.ts @@ -26,7 +26,7 @@ import path from "node:path"; import { readFile, stat } from "node:fs/promises"; -import { withJsonFileLock } from "./jsonFile-server"; +import { withJsonFileLock, writeJsonAtomic } from "./jsonFile-server"; import { sidecar, sidecarField } from "./sidecar-server"; import { METADATA_HISTORY_FILENAME, @@ -129,3 +129,56 @@ export async function withMetadataHistory<T>( } } } + +// THE ONE WRITER OF metadata.info.json THAT IS NOT yt-dlp: merge `patch` into +// the file on disk, inside `withMetadataHistory`, so the rewrite is recorded +// exactly as a yt-dlp one is (by `ctx.by`, with what moved). +// +// A MERGE, NEVER A REPLACEMENT. Every key the file has stays where it is and +// keeps its value unless `patch` names it; a key the file lacks is appended in +// `patch`'s order. Order is not cosmetic: two readers scan the file's BYTES +// rather than parse it — the title from a 16 KB head (videoTitles.ts) and the +// upload_date from an 8 KB tail (recencyIndex.ts) — so a caller adding a long +// description and a date names the description first. +// +// Compact JSON, no trailing newline — yt-dlp's own shape, near enough for +// every reader (they all allow whitespace after a colon). A patch that changes +// nothing writes nothing. A missing file, or one that is not a JSON object, +// throws: there is no record to complete, and a write would invent one. +// +// Under the file's lock, so two patches of one record never drop each other. +// It does NOT serialise against a yt-dlp spawn rewriting the same file; the +// caller's job runs on the channel's download queue for that reason. +export async function patchMetadataInfo( + videoDir: string, + patch: Record<string, unknown>, + ctx: MetadataRewriteContext & { onLog?: (line: string) => void }, +): Promise<{ written: boolean }> { + const file = path.join(videoDir, INFO_JSON); + return withJsonFileLock(file, async () => { + const text = await readFile(file, "utf8"); + let current: unknown; + try { + current = JSON.parse(text); + } catch { + throw new Error(`${file} is not JSON`); + } + if (typeof current !== "object" || current === null || Array.isArray(current)) { + throw new Error(`${file} is not a JSON object`); + } + const before = current as Record<string, unknown>; + const next: Record<string, unknown> = { ...before }; + let changed = false; + for (const [key, value] of Object.entries(patch)) { + if (value === undefined) continue; + if (JSON.stringify(before[key]) === JSON.stringify(value)) continue; + next[key] = value; + changed = true; + } + if (!changed) return { written: false }; + await withMetadataHistory(videoDir, ctx, () => + writeJsonAtomic(file, next, { indent: 0, newline: false }), + ); + return { written: true }; + }); +} diff --git a/common/lib/metadataHistory.ts b/common/lib/metadataHistory.ts @@ -46,7 +46,8 @@ export const METADATA_HISTORY_CAP = 200; // A stored value whose JSON is larger than this is replaced by its digest. export const METADATA_HISTORY_MAX_VALUE_BYTES = 16 * 1024; -// Who rewrote the file: the three yt-dlp spawns that write it. +// Who rewrote the file: the three yt-dlp spawns that write it, and the one +// writer that is not yt-dlp (`patchMetadataInfo`, metadataHistory-server.ts). export const METADATA_HISTORY_WRITERS = [ // downloadOneManaged's attempt 0, on every managed download. "prefetch", @@ -55,6 +56,10 @@ export const METADATA_HISTORY_WRITERS = [ "audio-check", // runYtdlp's legacy single-video `download-one-audio` job. "download-one", + // The podcast feed backfill (controller/feedMetadataBackfill.ts): title, + // date, description and duration from the channel's RSS feed, for a record + // imported by its enclosure URL, which carries none of them. + "feed-backfill", ] as const; export type MetadataHistoryWriter = (typeof METADATA_HISTORY_WRITERS)[number];