Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c8668fe58c84cd97129c41d13f02918e7b444e6f
parent 1362716c5c8b15293d2542cf549602f5c2c151ec
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  5 Oct 2026 05:18:26 -0400

common: a small RSS feed reader (items: title, pubDate, enclosure, guid, itunes:duration, description) with a fixture feed

CDATA is lifted out before the tag scan, comments are dropped, elements are
found by their exact qualified name, and entities are decoded once. A
description is turned into plain text; a publication date with no zone reads
as UTC.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/lib/__fixtures__/demo-podcast.rss | 54++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/rssFeed.test.ts | 120+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/rssFeed.ts | 237+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 411 insertions(+), 0 deletions(-)

diff --git a/common/lib/__fixtures__/demo-podcast.rss b/common/lib/__fixtures__/demo-podcast.rss @@ -0,0 +1,54 @@ +<?xml version="1.0" encoding="UTF-8"?> +<rss version="2.0" xmlns:itunes="http://www.itunes.com/dtds/podcast-1.0.dtd" xmlns:atom="http://www.w3.org/2005/Atom" xmlns:content="http://purl.org/rss/1.0/modules/content/"> + <channel> + <title>Demo &amp; Friends Podcast</title> + <atom:link href="https://feeds.example.com/demo.rss" rel="self" type="application/rss+xml"/> + <link>https://example.com</link> + <description>A demo feed for tests.</description> + <image> + <url>https://example.com/cover.jpg</url> + <title>Not the channel title</title> + </image> + <!-- a commented-out <item><title>Never an item</title></item> --> + <item> + <title><![CDATA[Episode 1: Cats & <Dogs>]]></title> + <link>https://example.com/episodes/one</link> + <guid isPermaLink="false">guid-0001</guid> + <pubDate>Tue, 05 Mar 2024 10:00:00 GMT</pubDate> + <enclosure url="https://cdn.example.com/audio/abc123.mp3?key=a&amp;updated=1" length="1000" type="audio/mpeg"/> + <itunes:duration>01:02:03</itunes:duration> + <description><![CDATA[<p>First line &amp; more.</p><p>Second&nbsp;line<br/>third</p>]]></description> + </item> + <item> + <title>Two &amp; a half &#8217;quotes&#8217; &quot;here&quot;</title> + <itunes:title>Not the item title</itunes:title> + <guid isPermaLink="false">guid-0002</guid> + <pubDate>Mon, 04 Mar 2024 22:30:00 -0500</pubDate> + <enclosure url="https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2" type="audio/mpeg" length="2000"></enclosure> + <itunes:duration>45:30</itunes:duration> + <description>&lt;p&gt;Hello &amp;amp; goodbye&lt;/p&gt;</description> + </item> + <item> + <!-- <title>A trap</title> --> + <title>Episode three</title> + <link>https://example.com/episodes/three</link> + <guid isPermaLink="true">https://example.com/?p=3</guid> + <pubDate>Wed, 06 Mar 2024 08:00:00 +0000</pubDate> + <enclosure url='https://cdn.example.com/audio/ghi789.mp3' type='audio/mpeg' length='3000' /> + <itunes:duration>3725</itunes:duration> + <itunes:summary>Summary only.</itunes:summary> + </item> + <item> + <title>Episode four</title> + <guid isPermaLink="false">guid-0004</guid> + <pubDate>Thu, 07 Mar 2024 12:00:00 GMT</pubDate> + <enclosure url="https://cdn.example.com/audio/jkl012.mp3" type="audio/mpeg" length="4000"/> + <description><![CDATA[A CDATA body that mentions </item> and ]] without ending.]]></description> + </item> + <item> + <title>Trailer</title> + <guid isPermaLink="false">guid-trailer</guid> + <pubDate>Fri, 01 Mar 2024 09:00:00 GMT</pubDate> + </item> + </channel> +</rss> diff --git a/common/lib/rssFeed.test.ts b/common/lib/rssFeed.test.ts @@ -0,0 +1,120 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + feedDateToTimestamp, + feedDateToUploadDate, + htmlToPlainText, + parseItunesDuration, + parseRssFeed, +} from "./rssFeed"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/rssFeed.test.ts + +const HERE = path.dirname(fileURLToPath(import.meta.url)); +const FIXTURE = readFileSync( + path.join(HERE, "__fixtures__", "demo-podcast.rss"), + "utf8", +); + +test("the fixture feed: channel title, and every item in order", () => { + const feed = parseRssFeed(FIXTURE); + assert.equal(feed.title, "Demo & Friends Podcast"); + assert.deepEqual( + feed.items.map((i) => i.guid), + ["guid-0001", "guid-0002", "https://example.com/?p=3", "guid-0004", "guid-trailer"], + "a commented-out item is not an item, and a CDATA </item> ends nothing", + ); +}); + +test("CDATA is verbatim; entities are decoded once outside it", () => { + const [one, two] = parseRssFeed(FIXTURE).items; + assert.equal(one.title, "Episode 1: Cats & <Dogs>"); + assert.equal(two.title, "Two & a half ’quotes’ \"here\""); + // An attribute's &amp; is a URL's &. + assert.equal( + one.enclosureUrl, + "https://cdn.example.com/audio/abc123.mp3?key=a&updated=1", + ); +}); + +test("the item's own <title> wins over itunes:title, and a comment is no title", () => { + const items = parseRssFeed(FIXTURE).items; + assert.equal(items[1].title?.startsWith("Two"), true); + assert.equal(items[2].title, "Episode three"); +}); + +test("descriptions become plain text: HTML in CDATA, escaped HTML, itunes:summary", () => { + const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items; + assert.equal(one.description, "First line & more.\nSecond line\nthird"); + assert.equal(two.description, "Hello & goodbye"); + assert.equal(three.description, "Summary only."); + assert.equal( + four.description, + "A CDATA body that mentions and ]] without ending.", + "the CDATA's literal </item> is markup to the plain-text pass, not the end of the item", + ); + assert.equal(trailer.description, null); +}); + +test("itunes tags, links, enclosures in either quote style", () => { + const [one, two, three, four, trailer] = parseRssFeed(FIXTURE).items; + assert.equal(one.durationSeconds, 3723); + assert.equal(two.durationSeconds, 2730); + assert.equal(three.durationSeconds, 3725); + assert.equal(four.durationSeconds, null); + assert.equal(one.link, "https://example.com/episodes/one"); + assert.equal(two.link, null); + assert.equal(three.enclosureUrl, "https://cdn.example.com/audio/ghi789.mp3"); + assert.equal( + two.enclosureUrl, + "https://track.example.net/redirect.mp3/cdn.example.com/audio/def456.mp3?updated=2", + ); + assert.equal(trailer.enclosureUrl, null); + assert.equal(one.pubDate, "Tue, 05 Mar 2024 10:00:00 GMT"); +}); + +test("parseItunesDuration: seconds, MM:SS, HH:MM:SS; junk is null", () => { + assert.equal(parseItunesDuration("3725"), 3725); + assert.equal(parseItunesDuration("45:30"), 2730); + assert.equal(parseItunesDuration("1:02:03"), 3723); + assert.equal(parseItunesDuration(" 90.6 "), 91); + assert.equal(parseItunesDuration("1 hour"), null); + assert.equal(parseItunesDuration(""), null); + assert.equal(parseItunesDuration(null), null); +}); + +test("upload_date is the UTC day of the publication instant", () => { + assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 10:00:00 GMT"), "20240305"); + // 22:30 at -0500 is 03:30 the next day in UTC. + assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 -0500"), "20240305"); + assert.equal(feedDateToUploadDate("Mon, 04 Mar 2024 22:30:00 EST"), "20240305"); + assert.equal(feedDateToUploadDate("2024-03-05T23:30:00+02:00"), "20240305"); + // No zone at all reads as UTC, never as this machine's zone. + assert.equal(feedDateToUploadDate("Tue, 05 Mar 2024 23:30:00"), "20240305"); + assert.equal(feedDateToUploadDate("2024-03-05T23:30:00"), "20240305"); + assert.equal(feedDateToUploadDate("2024-03-05"), "20240305"); + assert.equal(feedDateToUploadDate("not a date"), null); + assert.equal(feedDateToUploadDate(null), null); + assert.equal(feedDateToTimestamp("Tue, 05 Mar 2024 10:00:00 GMT"), 1709632800); +}); + +test("htmlToPlainText: list items, blank-line runs, numeric entities", () => { + assert.equal( + htmlToPlainText("<ul><li>one</li><li>two</li></ul><p></p><p></p><p>&#169; x</p>"), + "- one\n- two\n\n© x", + ); + assert.equal(htmlToPlainText("plain text"), "plain text"); +}); + +test("a self-closing Atom-style link is read from its href; a BOM is ignored", () => { + const feed = parseRssFeed( + "<rss><channel><title>T</title><item><title>A</title>" + + '<link href="https://example.com/a"/></item></channel></rss>', + ); + assert.equal(feed.title, "T"); + assert.equal(feed.items[0].link, "https://example.com/a"); +}); diff --git a/common/lib/rssFeed.ts b/common/lib/rssFeed.ts @@ -0,0 +1,237 @@ +// A PODCAST RSS FEED, READ: the channel title and, per <item>, the six things +// an episode's record wants from it — title, publication date, enclosure URL, +// guid, itunes:duration and description (plus the item's <link>). +// +// PURE: text in, values out. The one fetch of a feed lives in +// controller/feedMetadataBackfill.ts; everything here is testable with a +// fixture string. +// +// A SMALL PARSER, NOT A DEPENDENCY. The repo has no XML parser and an RSS +// item is a flat list of named children, so this reads exactly that and no +// more: CDATA sections are lifted out first (they may hold `<`, `</item>` or +// `]]` look-alikes that a tag scan would misread), comments are dropped, and an +// element is found by its exact qualified name (`title` never matches +// `itunes:title`). Entities are decoded once in text and attributes — the five +// XML ones, numeric references, and the HTML names feeds use in practice. +// What it will NOT do is validate: a feed that a strict parser would reject but +// a podcast app would play is read the way the app reads it. + +import { decodeEntities } from "../social/xArticle"; + +export type FeedItem = { + title: string | null; + // The item's own page, when the feed gives one. + link: string | null; + guid: string | null; + // As written in the feed (RFC 822 in RSS 2.0; some feeds write ISO 8601). + pubDate: string | null; + enclosureUrl: string | null; + durationSeconds: number | null; + // Plain text: a feed's description is usually HTML, and a record's + // description is read as text. + description: string | null; +}; + +export type ParsedFeed = { + title: string | null; + items: FeedItem[]; +}; + +// U+0000 never appears in well-formed XML, so it is a safe placeholder fence. +const CDATA_MARK = "\u0000"; + +type Lifted = { text: string; cdata: string[] }; + +// Replace every CDATA section with a numbered placeholder, then drop comments +// (a comment INSIDE a CDATA section is text, which is why this order). +function liftCdata(xml: string): Lifted { + const cdata: string[] = []; + const text = xml + .replace(/<!\[CDATA\[([\s\S]*?)\]\]>/g, (_whole, body: string) => { + cdata.push(body); + return `${CDATA_MARK}${cdata.length - 1}${CDATA_MARK}`; + }) + .replace(/<!--[\s\S]*?-->/g, ""); + return { text, cdata }; +} + +function escapeRe(s: string): string { + return s.replace(/[.*+?^${}()|[\]\\]/g, "\\$&"); +} + +// The raw inner text of the first `<name …>…</name>` in `block`, or "" for a +// self-closing one, or null when there is none. +function elementInner(block: string, name: string): string | null { + const re = new RegExp( + `<${escapeRe(name)}(?=[\\s/>])([^>]*?)(/?)>`, + "g", + ); + const open = re.exec(block); + if (!open) return null; + if (open[2] === "/") return ""; + const start = open.index + open[0].length; + const close = block.indexOf(`</${name}>`, start); + // An unclosed element is read to the end of the block rather than dropped. + return close < 0 ? block.slice(start) : block.slice(start, close); +} + +// The attributes of the first `<name …>` in `block`, decoded. +function elementAttrs( + block: string, + name: string, + cdata: string[], +): Record<string, string> | null { + const open = new RegExp(`<${escapeRe(name)}(?=[\\s/>])([^>]*)>`).exec(block); + if (!open) return null; + const attrs: Record<string, string> = {}; + const attrRe = /([^\s"'<>/=]+)\s*=\s*(?:"([^"]*)"|'([^']*)')/g; + for (const m of open[1].matchAll(attrRe)) { + attrs[m[1]] = decodeEntities(restoreCdata(m[2] ?? m[3] ?? "", cdata)); + } + return attrs; +} + +function restoreCdata(s: string, cdata: string[]): string { + return s.replace( + new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`, "g"), + (_w, i: string) => cdata[Number(i)] ?? "", + ); +} + +// An element's text: entity-decoded outside CDATA, verbatim inside it, with +// any stray markup outside CDATA removed. Trimmed; "" reads as null. +function textOf(inner: string | null, cdata: string[]): string | null { + if (inner === null) return null; + const parts = inner.split(new RegExp(`${CDATA_MARK}(\\d+)${CDATA_MARK}`)); + let out = ""; + for (let i = 0; i < parts.length; i++) { + out += + i % 2 === 1 + ? (cdata[Number(parts[i])] ?? "") + : decodeEntities(parts[i].replace(/<[^>]*>/g, "")); + } + const trimmed = out.trim(); + return trimmed ? trimmed : null; +} + +function firstText( + block: string, + names: readonly string[], + cdata: string[], +): string | null { + for (const name of names) { + const t = textOf(elementInner(block, name), cdata); + if (t) return t; + } + return null; +} + +// itunes:duration: plain seconds ("3725"), "MM:SS" or "HH:MM:SS", fractions +// allowed. Anything else is null, never a guess. +export function parseItunesDuration(raw: string | null | undefined): number | null { + if (!raw) return null; + const s = raw.trim(); + if (!/^\d+(?:\.\d+)?(?::\d{1,2}(?:\.\d+)?){0,2}$/.test(s)) return null; + const parts = s.split(":").map(Number); + let total = 0; + for (const p of parts) total = total * 60 + p; + return Number.isFinite(total) ? Math.round(total) : null; +} + +// A publication date as a Date, or null. RFC 822 ("Tue, 05 Mar 2024 10:00:00 +// GMT", "+0000", "EST") and ISO 8601 are both what Date.parse reads; a date +// with no zone at all is read as UTC rather than as this machine's zone. +export function parseFeedDate(raw: string | null | undefined): Date | null { + if (!raw) return null; + let s = raw.trim().replace(/\s+/g, " "); + if (!s) return null; + const hasZone = + /(?:Z|[+-]\d{2}:?\d{2}|\b(?:GMT|UTC|UT|[ECMP][SD]T)\b)\s*$/i.test(s); + if (!hasZone) { + // ISO with a time takes a "Z"; a bare ISO date is already UTC to + // Date.parse; anything else (RFC 822 without a zone) takes " GMT". + if (/^\d{4}-\d{2}-\d{2}[T ]/.test(s)) s = `${s.slice(0, 10)}T${s.slice(11)}Z`; + else if (!/^\d{4}-\d{2}-\d{2}$/.test(s)) s = `${s} GMT`; + } + const ms = Date.parse(s); + return Number.isFinite(ms) ? new Date(ms) : null; +} + +// yt-dlp's `upload_date`: YYYYMMDD of the publication instant, in UTC (as +// yt-dlp derives it from a timestamp). +export function feedDateToUploadDate(raw: string | null | undefined): string | null { + const d = parseFeedDate(raw); + if (!d) return null; + const y = d.getUTCFullYear(); + if (y < 1900 || y > 9999) return null; + const mm = String(d.getUTCMonth() + 1).padStart(2, "0"); + const dd = String(d.getUTCDate()).padStart(2, "0"); + return `${y}${mm}${dd}`; +} + +// yt-dlp's `timestamp`: whole seconds since the epoch. +export function feedDateToTimestamp(raw: string | null | undefined): number | null { + const d = parseFeedDate(raw); + return d ? Math.floor(d.getTime() / 1000) : null; +} + +// A feed description as plain text: block ends become line breaks, tags go, +// entities are decoded (the HTML ones a description carries once its own XML +// escaping is gone), and runs of blank lines collapse to one. +export function htmlToPlainText(html: string): string { + return decodeEntities( + html + .replace(/<br\s*\/?>/gi, "\n") + .replace(/<\/(?:p|div|li|h[1-6]|blockquote|tr)>/gi, "\n") + .replace(/<li[^>]*>/gi, "- ") + .replace(/<[^>]*>/g, ""), + ) + .replace(/ /g, " ") + .split("\n") + .map((line) => line.replace(/[ \t]+/g, " ").trim()) + .join("\n") + .replace(/\n{3,}/g, "\n\n") + .trim(); +} + +function parseItem(block: string, cdata: string[]): FeedItem { + const enclosure = + elementAttrs(block, "enclosure", cdata)?.url ?? + elementAttrs(block, "media:content", cdata)?.url ?? + null; + const rawDescription = firstText( + block, + ["description", "itunes:summary", "content:encoded"], + cdata, + ); + const description = rawDescription ? htmlToPlainText(rawDescription) : ""; + return { + title: firstText(block, ["title", "itunes:title"], cdata), + link: + firstText(block, ["link"], cdata) ?? + elementAttrs(block, "link", cdata)?.href ?? + null, + guid: firstText(block, ["guid"], cdata), + pubDate: firstText(block, ["pubDate", "dc:date", "published"], cdata), + enclosureUrl: enclosure?.trim() ? enclosure.trim() : null, + durationSeconds: parseItunesDuration( + firstText(block, ["itunes:duration"], cdata), + ), + description: description || null, + }; +} + +export function parseRssFeed(xml: string): ParsedFeed { + const { text, cdata } = liftCdata(xml.replace(/^/, "")); + const items: FeedItem[] = []; + const itemRe = /<item(?=[\s>])[^>]*>([\s\S]*?)<\/item>/g; + let firstItemAt = -1; + for (const m of text.matchAll(itemRe)) { + if (firstItemAt < 0) firstItemAt = m.index ?? 0; + items.push(parseItem(m[1], cdata)); + } + // The channel's own title is the first <title> before the first item (an + // <image> block's title comes after the channel's in every feed seen). + const head = firstItemAt < 0 ? text : text.slice(0, firstItemAt); + return { title: firstText(head, ["title"], cdata), items }; +}