// A small HTML reader, pure: no DOM, no dependency. Shared by the X Article
// reader (xArticle.ts) and the XenForo thread parser (xenforoParse.ts).
//
// For HTML a browser serialised (outerHTML): attributes are double-quoted, void
// elements are unclosed, text escapes only & < > and nbsp. It tolerates more
// (single quotes, bare values, stray end tags), but it is not a general parser.
export type HtmlElement = {
tag: string;
attrs: Record;
children: HtmlNode[];
};
export type HtmlNode = HtmlElement | string;
const VOID_TAGS = new Set([
"area", "base", "br", "col", "embed", "hr", "img", "input", "link", "meta",
"param", "source", "track", "wbr",
]);
const RAW_TEXT_TAGS = new Set(["script", "style", "textarea", "title"]);
const NAMED_ENTITIES: Record = {
amp: "&", lt: "<", gt: ">", quot: '"', apos: "'", nbsp: "\u00a0",
hellip: "…", mdash: "—", ndash: "–", lsquo: "‘", rsquo: "’", ldquo: "“",
rdquo: "”", copy: "©", reg: "®", trade: "™",
};
export function decodeEntities(s: string): string {
return s.replace(/&(#x[0-9a-f]+|#\d+|[a-z]+);/gi, (whole, name: string) => {
if (name[0] === "#") {
const code =
name[1] === "x" || name[1] === "X"
? parseInt(name.slice(2), 16)
: parseInt(name.slice(1), 10);
return Number.isFinite(code) && code > 0 && code <= 0x10ffff
? String.fromCodePoint(code)
: whole;
}
return NAMED_ENTITIES[name.toLowerCase()] ?? whole;
});
}
const TAG_NAME_RE = /[A-Za-z][A-Za-z0-9:-]*/y;
const ATTR_RE = /\s*([^\s"'<>\/=]+)(?:\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'=<>`]+)))?/y;
// The document's top-level nodes, under a synthetic root element.
export function parseHtml(html: string): HtmlElement {
const root: HtmlElement = { tag: "#root", attrs: {}, children: [] };
// Lowered once: a raw-text element's end is searched for in it (a page holds
// dozens of scripts, and lowering the whole page per script is quadratic).
let lower: string | undefined;
const stack: HtmlElement[] = [root];
const top = () => stack[stack.length - 1];
let i = 0;
while (i < html.length) {
const lt = html.indexOf("<", i);
if (lt < 0) {
top().children.push(decodeEntities(html.slice(i)));
break;
}
if (lt > i) top().children.push(decodeEntities(html.slice(i, lt)));
i = lt;
if (html.startsWith("", i + 4);
i = end < 0 ? html.length : end + 3;
continue;
}
if (html[i + 1] === "!" || html[i + 1] === "?") {
const end = html.indexOf(">", i);
i = end < 0 ? html.length : end + 1;
continue;
}
if (html[i + 1] === "/") {
const end = html.indexOf(">", i);
const name = html.slice(i + 2, end < 0 ? html.length : end).trim().toLowerCase();
i = end < 0 ? html.length : end + 1;
// Close up to the matching open element; a stray end tag is ignored.
for (let k = stack.length - 1; k > 0; k--) {
if (stack[k].tag === name) {
stack.length = k;
break;
}
}
continue;
}
TAG_NAME_RE.lastIndex = i + 1;
const nameMatch = TAG_NAME_RE.exec(html);
if (!nameMatch) {
// A "<" that opens no tag is text.
top().children.push("<");
i++;
continue;
}
const tag = nameMatch[0].toLowerCase();
let j = TAG_NAME_RE.lastIndex;
const attrs: Record = {};
for (;;) {
ATTR_RE.lastIndex = j;
const a = ATTR_RE.exec(html);
if (!a || a[0].length === 0) break;
attrs[a[1].toLowerCase()] = decodeEntities(a[2] ?? a[3] ?? a[4] ?? "");
j = ATTR_RE.lastIndex;
}
const close = html.indexOf(">", j);
const selfClosing = close > 0 && html[close - 1] === "/";
i = close < 0 ? html.length : close + 1;
const el: HtmlElement = { tag, attrs, children: [] };
top().children.push(el);
if (RAW_TEXT_TAGS.has(tag)) {
lower ??= html.toLowerCase();
const end = lower.indexOf(`${tag}`, i);
const stop = end < 0 ? html.length : end;
if (tag === "title" || tag === "textarea") el.children.push(decodeEntities(html.slice(i, stop)));
const gt = end < 0 ? -1 : html.indexOf(">", end);
i = gt < 0 ? html.length : gt + 1;
continue;
}
if (!selfClosing && !VOID_TAGS.has(tag)) stack.push(el);
}
return root;
}