// A PODCAST RSS FEED, READ: the channel title and, per , the six things // an episode's record wants from it — title, publication date, enclosure URL, // guid, itunes:duration and description (plus the item's ). // // PURE: text in, values out. The one fetch of a feed lives in // controller/feedMetadataBackfill.ts; everything here is testable with a // fixture string. // // A SMALL PARSER, NOT A DEPENDENCY. The repo has no XML parser and an RSS // item is a flat list of named children, so this reads exactly that and no // more: CDATA sections are lifted out first (they may hold `<`, `` or // `]]` look-alikes that a tag scan would misread), comments are dropped, and an // element is found by its exact qualified name (`title` never matches // `itunes:title`). Entities are decoded once in text and attributes — the five // XML ones, numeric references, and the HTML names feeds use in practice. // What it will NOT do is validate: a feed that a strict parser would reject but // a podcast app would play is read the way the app reads it. import { decodeEntities } from "../social/xArticle"; export type FeedItem = { title: string | null; // The item's own page, when the feed gives one. link: string | null; guid: string | null; // As written in the feed (RFC 822 in RSS 2.0; some feeds write ISO 8601). pubDate: string | null; enclosureUrl: string | null; durationSeconds: number | null; // Plain text: a feed's description is usually HTML, and a record's // description is read as text. description: string | null; }; export type ParsedFeed = { title: string | null; items: FeedItem[]; }; // U+0000 never appears in well-formed XML, so it is a safe placeholder fence. const CDATA_MARK = "\u0000"; type Lifted = { text: string; cdata: string[] }; // Replace every CDATA section with a numbered placeholder, then drop comments // (a comment INSIDE a CDATA section is text, which is why this order). function liftCdata(xml: string): Lifted { const cdata: string[] = []; const text = xml .replace(//g, (_whole, body: string) => { cdata.push(body); return `${CDATA_MARK}${cdata.length - 1}${CDATA_MARK}`; }) .replace(//g, ""); return { text, cdata }; } function escapeRe(s: string): string { return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); } // The raw inner text of the first `…` in `block`, or "" for a // self-closing one, or null when there is none. function elementInner(block: string, name: string): string | null { const re = new RegExp( `<${escapeRe(name)}(?=[\\s/>])([^>]*?)(/?)>`, "g", ); const open = re.exec(block); if (!open) return null; if (open[2] === "/") return ""; const start = open.index + open[0].length; const close = block.indexOf(``, start); // An unclosed element is read to the end of the block rather than dropped. return close < 0 ? block.slice(start) : block.slice(start, close); } // The attributes of the first `` in `block`, decoded. function elementAttrs( block: string, name: string, cdata: string[], ): Record | null { const open = new RegExp(`<${escapeRe(name)}(?=[\\s/>])([^>]*)>`).exec(block); if (!open) return null; const attrs: Record = {}; const attrRe = /([^\s"'<>/=]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g; for (const m of open[1].matchAll(attrRe)) { attrs[m[1]] = decodeEntities(restoreCdata(m[2] ?? m[3] ?? "", cdata)); } return attrs; } function restoreCdata(s: string, cdata: string[]): string { return s.replace( new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`, "g"), (_w, i: string) => cdata[Number(i)] ?? "", ); } // An element's text: entity-decoded outside CDATA, verbatim inside it, with // any stray markup outside CDATA removed. Trimmed; "" reads as null. function textOf(inner: string | null, cdata: string[]): string | null { if (inner === null) return null; const parts = inner.split(new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`)); let out = ""; for (let i = 0; i < parts.length; i++) { out += i % 2 === 1 ? (cdata[Number(parts[i])] ?? "") : decodeEntities(parts[i].replace(/<[^>]*>/g, "")); } const trimmed = out.trim(); return trimmed ? trimmed : null; } function firstText( block: string, names: readonly string[], cdata: string[], ): string | null { for (const name of names) { const t = textOf(elementInner(block, name), cdata); if (t) return t; } return null; } // itunes:duration: plain seconds ("3725"), "MM:SS" or "HH:MM:SS", fractions // allowed. Anything else is null, never a guess. export function parseItunesDuration(raw: string | null | undefined): number | null { if (!raw) return null; const s = raw.trim(); if (!/^\d+(?:\.\d+)?(?::\d{1,2}(?:\.\d+)?){0,2}$/.test(s)) return null; const parts = s.split(":").map(Number); let total = 0; for (const p of parts) total = total * 60 + p; return Number.isFinite(total) ? Math.round(total) : null; } // A publication date as a Date, or null. RFC 822 ("Tue, 05 Mar 2024 10:00:00 // GMT", "+0000", "EST") and ISO 8601 are both what Date.parse reads; a date // with no zone at all is read as UTC rather than as this machine's zone. export function parseFeedDate(raw: string | null | undefined): Date | null { if (!raw) return null; let s = raw.trim().replace(/\s+/g, " "); if (!s) return null; const hasZone = /(?:Z|[+-]\d{2}:?\d{2}|\b(?:GMT|UTC|UT|[ECMP][SD]T)\b)\s*$/i.test(s); if (!hasZone) { // ISO with a time takes a "Z"; a bare ISO date is already UTC to // Date.parse; anything else (RFC 822 without a zone) takes " GMT". if (/^\d{4}-\d{2}-\d{2}[T ]/.test(s)) s = `${s.slice(0, 10)}T${s.slice(11)}Z`; else if (!/^\d{4}-\d{2}-\d{2}$/.test(s)) s = `${s} GMT`; } const ms = Date.parse(s); return Number.isFinite(ms) ? new Date(ms) : null; } // yt-dlp's `upload_date`: YYYYMMDD of the publication instant, in UTC (as // yt-dlp derives it from a timestamp). export function feedDateToUploadDate(raw: string | null | undefined): string | null { const d = parseFeedDate(raw); if (!d) return null; const y = d.getUTCFullYear(); if (y < 1900 || y > 9999) return null; const mm = String(d.getUTCMonth() + 1).padStart(2, "0"); const dd = String(d.getUTCDate()).padStart(2, "0"); return `${y}${mm}${dd}`; } // yt-dlp's `timestamp`: whole seconds since the epoch. export function feedDateToTimestamp(raw: string | null | undefined): number | null { const d = parseFeedDate(raw); return d ? Math.floor(d.getTime() / 1000) : null; } // A feed description as plain text: block ends become line breaks, tags go, // entities are decoded (the HTML ones a description carries once its own XML // escaping is gone), and runs of blank lines collapse to one. export function htmlToPlainText(html: string): string { return decodeEntities( html .replace(//gi, "\n") .replace(/<\/(?:p|div|li|h[1-6]|blockquote|tr)>/gi, "\n") .replace(/]*>/gi, "- ") .replace(/<[^>]*>/g, ""), ) .replace(/ /g, " ") .split("\n") .map((line) => line.replace(/[ \t]+/g, " ").trim()) .join("\n") .replace(/\n{3,}/g, "\n\n") .trim(); } function parseItem(block: string, cdata: string[]): FeedItem { const enclosure = elementAttrs(block, "enclosure", cdata)?.url ?? elementAttrs(block, "media:content", cdata)?.url ?? null; const rawDescription = firstText( block, ["description", "itunes:summary", "content:encoded"], cdata, ); const description = rawDescription ? htmlToPlainText(rawDescription) : ""; return { title: firstText(block, ["title", "itunes:title"], cdata), link: firstText(block, ["link"], cdata) ?? elementAttrs(block, "link", cdata)?.href ?? null, guid: firstText(block, ["guid"], cdata), pubDate: firstText(block, ["pubDate", "dc:date", "published"], cdata), enclosureUrl: enclosure?.trim() ? enclosure.trim() : null, durationSeconds: parseItunesDuration( firstText(block, ["itunes:duration"], cdata), ), description: description || null, }; } export function parseRssFeed(xml: string): ParsedFeed { const { text, cdata } = liftCdata(xml.replace(/^/, "")); const items: FeedItem[] = []; const itemRe = /])[^>]*>([\s\S]*?)<\/item>/g; let firstItemAt = -1; for (const m of text.matchAll(itemRe)) { if (firstItemAt < 0) firstItemAt = m.index ?? 0; items.push(parseItem(m[1], cdata)); } // The channel's own title is the first before the first item (an // <image> block's title comes after the channel's in every feed seen). const head = firstItemAt < 0 ? text : text.slice(0, firstItemAt); return { title: firstText(head, ["title"], cdata), items }; }