// A PODCAST RSS FEED, READ: the channel title and, per , the six things
// an episode's record wants from it — title, publication date, enclosure URL,
// guid, itunes:duration and description (plus the item's ).
//
// PURE: text in, values out. The one fetch of a feed lives in
// controller/feedMetadataBackfill.ts; everything here is testable with a
// fixture string.
//
// A SMALL PARSER, NOT A DEPENDENCY. The repo has no XML parser and an RSS
// item is a flat list of named children, so this reads exactly that and no
// more: CDATA sections are lifted out first (they may hold `<`, `` or
// `]]` look-alikes that a tag scan would misread), comments are dropped, and an
// element is found by its exact qualified name (`title` never matches
// `itunes:title`). Entities are decoded once in text and attributes — the five
// XML ones, numeric references, and the HTML names feeds use in practice.
// What it will NOT do is validate: a feed that a strict parser would reject but
// a podcast app would play is read the way the app reads it.
import { decodeEntities } from "../social/xArticle";
export type FeedItem = {
title: string | null;
// The item's own page, when the feed gives one.
link: string | null;
guid: string | null;
// As written in the feed (RFC 822 in RSS 2.0; some feeds write ISO 8601).
pubDate: string | null;
enclosureUrl: string | null;
durationSeconds: number | null;
// Plain text: a feed's description is usually HTML, and a record's
// description is read as text.
description: string | null;
};
export type ParsedFeed = {
title: string | null;
items: FeedItem[];
};
// U+0000 never appears in well-formed XML, so it is a safe placeholder fence.
const CDATA_MARK = "\u0000";
type Lifted = { text: string; cdata: string[] };
// Replace every CDATA section with a numbered placeholder, then drop comments
// (a comment INSIDE a CDATA section is text, which is why this order).
function liftCdata(xml: string): Lifted {
const cdata: string[] = [];
const text = xml
.replace(//g, (_whole, body: string) => {
cdata.push(body);
return `${CDATA_MARK}${cdata.length - 1}${CDATA_MARK}`;
})
.replace(//g, "");
return { text, cdata };
}
function escapeRe(s: string): string {
return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
}
// The raw inner text of the first `…` in `block`, or "" for a
// self-closing one, or null when there is none.
function elementInner(block: string, name: string): string | null {
const re = new RegExp(
`<${escapeRe(name)}(?=[\\s/>])([^>]*?)(/?)>`,
"g",
);
const open = re.exec(block);
if (!open) return null;
if (open[2] === "/") return "";
const start = open.index + open[0].length;
const close = block.indexOf(`${name}>`, start);
// An unclosed element is read to the end of the block rather than dropped.
return close < 0 ? block.slice(start) : block.slice(start, close);
}
// The attributes of the first `` in `block`, decoded.
function elementAttrs(
block: string,
name: string,
cdata: string[],
): Record | null {
const open = new RegExp(`<${escapeRe(name)}(?=[\\s/>])([^>]*)>`).exec(block);
if (!open) return null;
const attrs: Record = {};
const attrRe = /([^\s"'<>/=]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g;
for (const m of open[1].matchAll(attrRe)) {
attrs[m[1]] = decodeEntities(restoreCdata(m[2] ?? m[3] ?? "", cdata));
}
return attrs;
}
function restoreCdata(s: string, cdata: string[]): string {
return s.replace(
new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`, "g"),
(_w, i: string) => cdata[Number(i)] ?? "",
);
}
// An element's text: entity-decoded outside CDATA, verbatim inside it, with
// any stray markup outside CDATA removed. Trimmed; "" reads as null.
function textOf(inner: string | null, cdata: string[]): string | null {
if (inner === null) return null;
const parts = inner.split(new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`));
let out = "";
for (let i = 0; i < parts.length; i++) {
out +=
i % 2 === 1
? (cdata[Number(parts[i])] ?? "")
: decodeEntities(parts[i].replace(/<[^>]*>/g, ""));
}
const trimmed = out.trim();
return trimmed ? trimmed : null;
}
function firstText(
block: string,
names: readonly string[],
cdata: string[],
): string | null {
for (const name of names) {
const t = textOf(elementInner(block, name), cdata);
if (t) return t;
}
return null;
}
// itunes:duration: plain seconds ("3725"), "MM:SS" or "HH:MM:SS", fractions
// allowed. Anything else is null, never a guess.
export function parseItunesDuration(raw: string | null | undefined): number | null {
if (!raw) return null;
const s = raw.trim();
if (!/^\d+(?:\.\d+)?(?::\d{1,2}(?:\.\d+)?){0,2}$/.test(s)) return null;
const parts = s.split(":").map(Number);
let total = 0;
for (const p of parts) total = total * 60 + p;
return Number.isFinite(total) ? Math.round(total) : null;
}
// A publication date as a Date, or null. RFC 822 ("Tue, 05 Mar 2024 10:00:00
// GMT", "+0000", "EST") and ISO 8601 are both what Date.parse reads; a date
// with no zone at all is read as UTC rather than as this machine's zone.
export function parseFeedDate(raw: string | null | undefined): Date | null {
if (!raw) return null;
let s = raw.trim().replace(/\s+/g, " ");
if (!s) return null;
const hasZone =
/(?:Z|[+-]\d{2}:?\d{2}|\b(?:GMT|UTC|UT|[ECMP][SD]T)\b)\s*$/i.test(s);
if (!hasZone) {
// ISO with a time takes a "Z"; a bare ISO date is already UTC to
// Date.parse; anything else (RFC 822 without a zone) takes " GMT".
if (/^\d{4}-\d{2}-\d{2}[T ]/.test(s)) s = `${s.slice(0, 10)}T${s.slice(11)}Z`;
else if (!/^\d{4}-\d{2}-\d{2}$/.test(s)) s = `${s} GMT`;
}
const ms = Date.parse(s);
return Number.isFinite(ms) ? new Date(ms) : null;
}
// yt-dlp's `upload_date`: YYYYMMDD of the publication instant, in UTC (as
// yt-dlp derives it from a timestamp).
export function feedDateToUploadDate(raw: string | null | undefined): string | null {
const d = parseFeedDate(raw);
if (!d) return null;
const y = d.getUTCFullYear();
if (y < 1900 || y > 9999) return null;
const mm = String(d.getUTCMonth() + 1).padStart(2, "0");
const dd = String(d.getUTCDate()).padStart(2, "0");
return `${y}${mm}${dd}`;
}
// yt-dlp's `timestamp`: whole seconds since the epoch.
export function feedDateToTimestamp(raw: string | null | undefined): number | null {
const d = parseFeedDate(raw);
return d ? Math.floor(d.getTime() / 1000) : null;
}
// A feed description as plain text: block ends become line breaks, tags go,
// entities are decoded (the HTML ones a description carries once its own XML
// escaping is gone), and runs of blank lines collapse to one.
export function htmlToPlainText(html: string): string {
return decodeEntities(
html
.replace(/ /gi, "\n")
.replace(/<\/(?:p|div|li|h[1-6]|blockquote|tr)>/gi, "\n")
.replace(/
]*>/gi, "- ")
.replace(/<[^>]*>/g, ""),
)
.replace(/ /g, " ")
.split("\n")
.map((line) => line.replace(/[ \t]+/g, " ").trim())
.join("\n")
.replace(/\n{3,}/g, "\n\n")
.trim();
}
function parseItem(block: string, cdata: string[]): FeedItem {
const enclosure =
elementAttrs(block, "enclosure", cdata)?.url ??
elementAttrs(block, "media:content", cdata)?.url ??
null;
const rawDescription = firstText(
block,
["description", "itunes:summary", "content:encoded"],
cdata,
);
const description = rawDescription ? htmlToPlainText(rawDescription) : "";
return {
title: firstText(block, ["title", "itunes:title"], cdata),
link:
firstText(block, ["link"], cdata) ??
elementAttrs(block, "link", cdata)?.href ??
null,
guid: firstText(block, ["guid"], cdata),
pubDate: firstText(block, ["pubDate", "dc:date", "published"], cdata),
enclosureUrl: enclosure?.trim() ? enclosure.trim() : null,
durationSeconds: parseItunesDuration(
firstText(block, ["itunes:duration"], cdata),
),
description: description || null,
};
}
export function parseRssFeed(xml: string): ParsedFeed {
const { text, cdata } = liftCdata(xml.replace(/^/, ""));
const items: FeedItem[] = [];
const itemRe = /])[^>]*>([\s\S]*?)<\/item>/g;
let firstItemAt = -1;
for (const m of text.matchAll(itemRe)) {
if (firstItemAt < 0) firstItemAt = m.index ?? 0;
items.push(parseItem(m[1], cdata));
}
// The channel's own title is the first before the first item (an
// block's title comes after the channel's in every feed seen).
const head = firstItemAt < 0 ? text : text.slice(0, firstItemAt);
return { title: firstText(head, ["title"], cdata), items };
}