// A small HTML reader, pure: no DOM, no dependency. Shared by the X Article // reader (xArticle.ts) and the XenForo thread parser (xenforoParse.ts). // // For HTML a browser serialised (outerHTML): attributes are double-quoted, void // elements are unclosed, text escapes only & < > and nbsp. It tolerates more // (single quotes, bare values, stray end tags), but it is not a general parser. export type HtmlElement = { tag: string; attrs: Record; children: HtmlNode[]; }; export type HtmlNode = HtmlElement | string; const VOID_TAGS = new Set([ "area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta", "param", "source", "track", "wbr", ]); const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]); const NAMED_ENTITIES: Record = { amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0", hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“", rdquo: "”", copy: "©", reg: "®", trade: "™", }; export function decodeEntities(s: string): string { return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => { if (name[0] === "#") { const code = name[1] === "x" || name[1] === "X" ? parseInt(name.slice(2), 16) : parseInt(name.slice(1), 10); return Number.isFinite(code) && code > 0 && code <= 0x10ffff ? String.fromCodePoint(code) : whole; } return NAMED_ENTITIES[name.toLowerCase()] ?? whole; }); } const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y; const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y; // The document's top-level nodes, under a synthetic root element. export function parseHtml(html: string): HtmlElement { const root: HtmlElement = { tag: "#root", attrs: {}, children: [] }; // Lowered once: a raw-text element's end is searched for in it (a page holds // dozens of scripts, and lowering the whole page per script is quadratic). let lower: string | undefined; const stack: HtmlElement[] = [root]; const top = () => stack[stack.length - 1]; let i = 0; while (i < html.length) { const lt = html.indexOf("<", i); if (lt < 0) { top().children.push(decodeEntities(html.slice(i))); break; } if (lt > i) top().children.push(decodeEntities(html.slice(i, lt))); i = lt; if (html.startsWith("", i + 4); i = end < 0 ? html.length : end + 3; continue; } if (html[i + 1] === "!" || html[i + 1] === "?") { const end = html.indexOf(">", i); i = end < 0 ? html.length : end + 1; continue; } if (html[i + 1] === "/") { const end = html.indexOf(">", i); const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase(); i = end < 0 ? html.length : end + 1; // Close up to the matching open element; a stray end tag is ignored. for (let k = stack.length - 1; k > 0; k--) { if (stack[k].tag === name) { stack.length = k; break; } } continue; } TAG_NAME_RE.lastIndex = i + 1; const nameMatch = TAG_NAME_RE.exec(html); if (!nameMatch) { // A "<" that opens no tag is text. top().children.push("<"); i++; continue; } const tag = nameMatch[0].toLowerCase(); let j = TAG_NAME_RE.lastIndex; const attrs: Record = {}; for (;;) { ATTR_RE.lastIndex = j; const a = ATTR_RE.exec(html); if (!a || a[0].length === 0) break; attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? ""); j = ATTR_RE.lastIndex; } const close = html.indexOf(">", j); const selfClosing = close > 0 && html[close - 1] === "/"; i = close < 0 ? html.length : close + 1; const el: HtmlElement = { tag, attrs, children: [] }; top().children.push(el); if (RAW_TEXT_TAGS.has(tag)) { lower ??= html.toLowerCase(); const end = lower.indexOf(`", end); i = gt < 0 ? html.length : gt + 1; continue; } if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el); } return root; }