commit c8668fe58c84cd97129c41d13f02918e7b444e6f
parent 1362716c5c8b15293d2542cf549602f5c2c151ec
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 05:18:26 -0400
common: a small RSS feed reader (items: title, pubDate, enclosure, guid, itunes:duration, description) with a fixture feed
CDATA is lifted out before the tag scan, comments are dropped, elements are
found by their exact qualified name, and entities are decoded once. A
description is turned into plain text; a publication date with no zone reads
as UTC.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 411 insertions(+), 0 deletions(-)
diff --git a/common/lib/__fixtures__/demo-podcast.rss b/common/lib/__fixtures__/demo-podcast.rss
@@ -0,0 +1,54 @@
+<?xml version="1.0" encoding="UTF-8"?>
+<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/">
+ <channel>
+ <title>Demo & Friends Podcast</title>
+ <atom:link href="https://feeds.example.com/demo.rss" rel="self" type="application/rss+xml"/>
+ <link>https://example.com</link>
+ <description>A demo feed for tests.</description>
+ <image>
+ <url>https://example.com/cover.jpg</url>
+ <title>Not the channel title</title>
+ </image>
+ <!-- a commented-out <item><title>Never an item</title></item> -->
+ <item>
+ <title><![CDATA[Episode 1: Cats & <Dogs>]]></title>
+ <link>https://example.com/episodes/one</link>
+ <guid isPermaLink="false">guid-0001</guid>
+ <pubDate>Tue, 05 Mar 2024 10:00:00 GMT</pubDate>
+ <enclosure url="https://cdn.example.com/audio/abc123.mp3?key=a&updated=1" length="1000" type="audio/mpeg"/>
+ <itunes:duration>01:02:03</itunes:duration>
+ <description><![CDATA[<p>First line & more.</p><p>Second line<br/>third</p>]]></description>
+ </item>
+ <item>
+ <title>Two & a half ’quotes’ "here"</title>
+ <itunes:title>Not the item title</itunes:title>
+ <guid isPermaLink="false">guid-0002</guid>
+ <pubDate>Mon, 04 Mar 2024 22:30:00 -0500</pubDate>
+ <enclosure url="https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2" type="audio/mpeg" length="2000"></enclosure>
+ <itunes:duration>45:30</itunes:duration>
+ <description><p>Hello &amp; goodbye</p></description>
+ </item>
+ <item>
+ <!-- <title>A trap</title> -->
+ <title>Episode three</title>
+ <link>https://example.com/episodes/three</link>
+ <guid isPermaLink="true">https://example.com/?p=3</guid>
+ <pubDate>Wed, 06 Mar 2024 08:00:00 +0000</pubDate>
+ <enclosure url='https://cdn.example.com/audio/ghi789.mp3' type='audio/mpeg' length='3000' />
+ <itunes:duration>3725</itunes:duration>
+ <itunes:summary>Summary only.</itunes:summary>
+ </item>
+ <item>
+ <title>Episode four</title>
+ <guid isPermaLink="false">guid-0004</guid>
+ <pubDate>Thu, 07 Mar 2024 12:00:00 GMT</pubDate>
+ <enclosure url="https://cdn.example.com/audio/jkl012.mp3" type="audio/mpeg" length="4000"/>
+ <description><![CDATA[A CDATA body that mentions </item> and ]] without ending.]]></description>
+ </item>
+ <item>
+ <title>Trailer</title>
+ <guid isPermaLink="false">guid-trailer</guid>
+ <pubDate>Fri, 01 Mar 2024 09:00:00 GMT</pubDate>
+ </item>
+ </channel>
+</rss>
diff --git a/common/lib/rssFeed.test.ts b/common/lib/rssFeed.test.ts
@@ -0,0 +1,120 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readFileSync } from "node:fs";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ feedDateToTimestamp,
+ feedDateToUploadDate,
+ htmlToPlainText,
+ parseItunesDuration,
+ parseRssFeed,
+} from "./rssFeed";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/rssFeed.test.ts
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const FIXTURE = readFileSync(
+ path.join(HERE, "__fixtures__", "demo-podcast.rss"),
+ "utf8",
+);
+
+test("the fixture feed: channel title, and every item in order", () => {
+ const feed = parseRssFeed(FIXTURE);
+ assert.equal(feed.title, "Demo & Friends Podcast");
+ assert.deepEqual(
+ feed.items.map((i) => i.guid),
+ ["guid-0001", "guid-0002", "https://example.com/?p=3", "guid-0004", "guid-trailer"],
+ "a commented-out item is not an item, and a CDATA </item> ends nothing",
+ );
+});
+
+test("CDATA is verbatim; entities are decoded once outside it", () => {
+ const [one, two] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.title, "Episode 1: Cats & <Dogs>");
+ assert.equal(two.title, "Two & a half ’quotes’ \"here\"");
+ // An attribute's & is a URL's &.
+ assert.equal(
+ one.enclosureUrl,
+ "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1",
+ );
+});
+
+test("the item's own <title> wins over itunes:title, and a comment is no title", () => {
+ const items = parseRssFeed(FIXTURE).items;
+ assert.equal(items[1].title?.startsWith("Two"), true);
+ assert.equal(items[2].title, "Episode three");
+});
+
+test("descriptions become plain text: HTML in CDATA, escaped HTML, itunes:summary", () => {
+ const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.description, "First line & more.\nSecond line\nthird");
+ assert.equal(two.description, "Hello & goodbye");
+ assert.equal(three.description, "Summary only.");
+ assert.equal(
+ four.description,
+ "A CDATA body that mentions and ]] without ending.",
+ "the CDATA's literal </item> is markup to the plain-text pass, not the end of the item",
+ );
+ assert.equal(trailer.description, null);
+});
+
+test("itunes tags, links, enclosures in either quote style", () => {
+ const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items;
+ assert.equal(one.durationSeconds, 3723);
+ assert.equal(two.durationSeconds, 2730);
+ assert.equal(three.durationSeconds, 3725);
+ assert.equal(four.durationSeconds, null);
+ assert.equal(one.link, "https://example.com/episodes/one");
+ assert.equal(two.link, null);
+ assert.equal(three.enclosureUrl, "https://cdn.example.com/audio/ghi789.mp3");
+ assert.equal(
+ two.enclosureUrl,
+ "https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2",
+ );
+ assert.equal(trailer.enclosureUrl, null);
+ assert.equal(one.pubDate, "Tue, 05 Mar 2024 10:00:00 GMT");
+});
+
+test("parseItunesDuration: seconds, MM:SS, HH:MM:SS; junk is null", () => {
+ assert.equal(parseItunesDuration("3725"), 3725);
+ assert.equal(parseItunesDuration("45:30"), 2730);
+ assert.equal(parseItunesDuration("1:02:03"), 3723);
+ assert.equal(parseItunesDuration(" 90.6 "), 91);
+ assert.equal(parseItunesDuration("1 hour"), null);
+ assert.equal(parseItunesDuration(""), null);
+ assert.equal(parseItunesDuration(null), null);
+});
+
+test("upload_date is the UTC day of the publication instant", () => {
+ assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 10:00:00 GMT"), "20240305");
+ // 22:30 at -0500 is 03:30 the next day in UTC.
+ assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 -0500"), "20240305");
+ assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 EST"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05T23:30:00+02:00"), "20240305");
+ // No zone at all reads as UTC, never as this machine's zone.
+ assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 23:30:00"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05T23:30:00"), "20240305");
+ assert.equal(feedDateToUploadDate("2024-03-05"), "20240305");
+ assert.equal(feedDateToUploadDate("not a date"), null);
+ assert.equal(feedDateToUploadDate(null), null);
+ assert.equal(feedDateToTimestamp("Tue, 05 Mar 2024 10:00:00 GMT"), 1709632800);
+});
+
+test("htmlToPlainText: list items, blank-line runs, numeric entities", () => {
+ assert.equal(
+ htmlToPlainText("<ul><li>one</li><li>two</li></ul><p></p><p></p><p>© x</p>"),
+ "- one\n- two\n\n© x",
+ );
+ assert.equal(htmlToPlainText("plain text"), "plain text");
+});
+
+test("a self-closing Atom-style link is read from its href; a BOM is ignored", () => {
+ const feed = parseRssFeed(
+ "<rss><channel><title>T</title><item><title>A</title>" +
+ '<link href="https://example.com/a"/></item></channel></rss>',
+ );
+ assert.equal(feed.title, "T");
+ assert.equal(feed.items[0].link, "https://example.com/a");
+});
diff --git a/common/lib/rssFeed.ts b/common/lib/rssFeed.ts
@@ -0,0 +1,237 @@
+// A PODCAST RSS FEED, READ: the channel title and, per <item>, the six things
+// an episode's record wants from it — title, publication date, enclosure URL,
+// guid, itunes:duration and description (plus the item's <link>).
+//
+// PURE: text in, values out. The one fetch of a feed lives in
+// controller/feedMetadataBackfill.ts; everything here is testable with a
+// fixture string.
+//
+// A SMALL PARSER, NOT A DEPENDENCY. The repo has no XML parser and an RSS
+// item is a flat list of named children, so this reads exactly that and no
+// more: CDATA sections are lifted out first (they may hold `<`, `</item>` or
+// `]]` look-alikes that a tag scan would misread), comments are dropped, and an
+// element is found by its exact qualified name (`title` never matches
+// `itunes:title`). Entities are decoded once in text and attributes — the five
+// XML ones, numeric references, and the HTML names feeds use in practice.
+// What it will NOT do is validate: a feed that a strict parser would reject but
+// a podcast app would play is read the way the app reads it.
+
+import { decodeEntities } from "../social/xArticle";
+
+export type FeedItem = {
+ title: string | null;
+ // The item's own page, when the feed gives one.
+ link: string | null;
+ guid: string | null;
+ // As written in the feed (RFC 822 in RSS 2.0; some feeds write ISO 8601).
+ pubDate: string | null;
+ enclosureUrl: string | null;
+ durationSeconds: number | null;
+ // Plain text: a feed's description is usually HTML, and a record's
+ // description is read as text.
+ description: string | null;
+};
+
+export type ParsedFeed = {
+ title: string | null;
+ items: FeedItem[];
+};
+
+// U+0000 never appears in well-formed XML, so it is a safe placeholder fence.
+const CDATA_MARK = "\u0000";
+
+type Lifted = { text: string; cdata: string[] };
+
+// Replace every CDATA section with a numbered placeholder, then drop comments
+// (a comment INSIDE a CDATA section is text, which is why this order).
+function liftCdata(xml: string): Lifted {
+ const cdata: string[] = [];
+ const text = xml
+ .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, (_whole, body: string) => {
+ cdata.push(body);
+ return `${CDATA_MARK}${cdata.length - 1}${CDATA_MARK}`;
+ })
+ .replace(/<!--[\s\S]*?-->/g, "");
+ return { text, cdata };
+}
+
+function escapeRe(s: string): string {
+ return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
+}
+
+// The raw inner text of the first `<name …>…</name>` in `block`, or "" for a
+// self-closing one, or null when there is none.
+function elementInner(block: string, name: string): string | null {
+ const re = new RegExp(
+ `<${escapeRe(name)}(?=[\\s/>])([^>]*?)(/?)>`,
+ "g",
+ );
+ const open = re.exec(block);
+ if (!open) return null;
+ if (open[2] === "/") return "";
+ const start = open.index + open[0].length;
+ const close = block.indexOf(`</${name}>`, start);
+ // An unclosed element is read to the end of the block rather than dropped.
+ return close < 0 ? block.slice(start) : block.slice(start, close);
+}
+
+// The attributes of the first `<name …>` in `block`, decoded.
+function elementAttrs(
+ block: string,
+ name: string,
+ cdata: string[],
+): Record<string, string> | null {
+ const open = new RegExp(`<${escapeRe(name)}(?=[\\s/>])([^>]*)>`).exec(block);
+ if (!open) return null;
+ const attrs: Record<string, string> = {};
+ const attrRe = /([^\s"'<>/=]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g;
+ for (const m of open[1].matchAll(attrRe)) {
+ attrs[m[1]] = decodeEntities(restoreCdata(m[2] ?? m[3] ?? "", cdata));
+ }
+ return attrs;
+}
+
+function restoreCdata(s: string, cdata: string[]): string {
+ return s.replace(
+ new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`, "g"),
+ (_w, i: string) => cdata[Number(i)] ?? "",
+ );
+}
+
+// An element's text: entity-decoded outside CDATA, verbatim inside it, with
+// any stray markup outside CDATA removed. Trimmed; "" reads as null.
+function textOf(inner: string | null, cdata: string[]): string | null {
+ if (inner === null) return null;
+ const parts = inner.split(new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`));
+ let out = "";
+ for (let i = 0; i < parts.length; i++) {
+ out +=
+ i % 2 === 1
+ ? (cdata[Number(parts[i])] ?? "")
+ : decodeEntities(parts[i].replace(/<[^>]*>/g, ""));
+ }
+ const trimmed = out.trim();
+ return trimmed ? trimmed : null;
+}
+
+function firstText(
+ block: string,
+ names: readonly string[],
+ cdata: string[],
+): string | null {
+ for (const name of names) {
+ const t = textOf(elementInner(block, name), cdata);
+ if (t) return t;
+ }
+ return null;
+}
+
+// itunes:duration: plain seconds ("3725"), "MM:SS" or "HH:MM:SS", fractions
+// allowed. Anything else is null, never a guess.
+export function parseItunesDuration(raw: string | null | undefined): number | null {
+ if (!raw) return null;
+ const s = raw.trim();
+ if (!/^\d+(?:\.\d+)?(?::\d{1,2}(?:\.\d+)?){0,2}$/.test(s)) return null;
+ const parts = s.split(":").map(Number);
+ let total = 0;
+ for (const p of parts) total = total * 60 + p;
+ return Number.isFinite(total) ? Math.round(total) : null;
+}
+
+// A publication date as a Date, or null. RFC 822 ("Tue, 05 Mar 2024 10:00:00
+// GMT", "+0000", "EST") and ISO 8601 are both what Date.parse reads; a date
+// with no zone at all is read as UTC rather than as this machine's zone.
+export function parseFeedDate(raw: string | null | undefined): Date | null {
+ if (!raw) return null;
+ let s = raw.trim().replace(/\s+/g, " ");
+ if (!s) return null;
+ const hasZone =
+ /(?:Z|[+-]\d{2}:?\d{2}|\b(?:GMT|UTC|UT|[ECMP][SD]T)\b)\s*$/i.test(s);
+ if (!hasZone) {
+ // ISO with a time takes a "Z"; a bare ISO date is already UTC to
+ // Date.parse; anything else (RFC 822 without a zone) takes " GMT".
+ if (/^\d{4}-\d{2}-\d{2}[T ]/.test(s)) s = `${s.slice(0, 10)}T${s.slice(11)}Z`;
+ else if (!/^\d{4}-\d{2}-\d{2}$/.test(s)) s = `${s} GMT`;
+ }
+ const ms = Date.parse(s);
+ return Number.isFinite(ms) ? new Date(ms) : null;
+}
+
+// yt-dlp's `upload_date`: YYYYMMDD of the publication instant, in UTC (as
+// yt-dlp derives it from a timestamp).
+export function feedDateToUploadDate(raw: string | null | undefined): string | null {
+ const d = parseFeedDate(raw);
+ if (!d) return null;
+ const y = d.getUTCFullYear();
+ if (y < 1900 || y > 9999) return null;
+ const mm = String(d.getUTCMonth() + 1).padStart(2, "0");
+ const dd = String(d.getUTCDate()).padStart(2, "0");
+ return `${y}${mm}${dd}`;
+}
+
+// yt-dlp's `timestamp`: whole seconds since the epoch.
+export function feedDateToTimestamp(raw: string | null | undefined): number | null {
+ const d = parseFeedDate(raw);
+ return d ? Math.floor(d.getTime() / 1000) : null;
+}
+
+// A feed description as plain text: block ends become line breaks, tags go,
+// entities are decoded (the HTML ones a description carries once its own XML
+// escaping is gone), and runs of blank lines collapse to one.
+export function htmlToPlainText(html: string): string {
+ return decodeEntities(
+ html
+ .replace(/<br\s*\/?>/gi, "\n")
+ .replace(/<\/(?:p|div|li|h[1-6]|blockquote|tr)>/gi, "\n")
+ .replace(/<li[^>]*>/gi, "- ")
+ .replace(/<[^>]*>/g, ""),
+ )
+ .replace(/ /g, " ")
+ .split("\n")
+ .map((line) => line.replace(/[ \t]+/g, " ").trim())
+ .join("\n")
+ .replace(/\n{3,}/g, "\n\n")
+ .trim();
+}
+
+function parseItem(block: string, cdata: string[]): FeedItem {
+ const enclosure =
+ elementAttrs(block, "enclosure", cdata)?.url ??
+ elementAttrs(block, "media:content", cdata)?.url ??
+ null;
+ const rawDescription = firstText(
+ block,
+ ["description", "itunes:summary", "content:encoded"],
+ cdata,
+ );
+ const description = rawDescription ? htmlToPlainText(rawDescription) : "";
+ return {
+ title: firstText(block, ["title", "itunes:title"], cdata),
+ link:
+ firstText(block, ["link"], cdata) ??
+ elementAttrs(block, "link", cdata)?.href ??
+ null,
+ guid: firstText(block, ["guid"], cdata),
+ pubDate: firstText(block, ["pubDate", "dc:date", "published"], cdata),
+ enclosureUrl: enclosure?.trim() ? enclosure.trim() : null,
+ durationSeconds: parseItunesDuration(
+ firstText(block, ["itunes:duration"], cdata),
+ ),
+ description: description || null,
+ };
+}
+
+export function parseRssFeed(xml: string): ParsedFeed {
+ const { text, cdata } = liftCdata(xml.replace(/^/, ""));
+ const items: FeedItem[] = [];
+ const itemRe = /<item(?=[\s>])[^>]*>([\s\S]*?)<\/item>/g;
+ let firstItemAt = -1;
+ for (const m of text.matchAll(itemRe)) {
+ if (firstItemAt < 0) firstItemAt = m.index ?? 0;
+ items.push(parseItem(m[1], cdata));
+ }
+ // The channel's own title is the first <title> before the first item (an
+ // <image> block's title comes after the channel's in every feed seen).
+ const head = firstItemAt < 0 ? text : text.slice(0, firstItemAt);
+ return { title: firstText(head, ["title"], cdata), items };
+}