import { test } from "node:test"; import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import path from "node:path"; import { fileURLToPath } from "node:url"; // Run with: // pnpm --filter homepage test // // public/_headers is what keeps the raw source tree from being served as // code on this origin, and nothing but the Pages edge applies it. This pins // its security property with a small replica of wrangler's `_headers` // handling (pages-shared parseHeaders + generateRulesMatcher + attachHeaders, // read in wrangler 4.88 and re-checked by the review in 4.142): every // matching rule applies in file order, `! Name` detaches a header, and a // header a LATER rule sets again is APPENDED ("a, b"), not replaced. const FILE = path.join(path.dirname(fileURLToPath(import.meta.url)), "..", "..", "public", "_headers"); const TEXT = readFileSync(FILE, "utf8"); type Rule = { path: string; set: Record; unset: string[]; re: RegExp; lines: string[] }; function parse(text: string): Rule[] { const rules: Rule[] = []; let rule: Rule | undefined; for (const raw of text.split("\n")) { const line = raw.trim(); if (!line || line.startsWith("#")) continue; if (/^([^\s]+:\/\/|^\/)/.test(line)) { if (rule) rules.push(rule); const esc = (s: string) => s.replace(/[|\\{}()[\]^$+*?.]/g, "\\$&"); rule = { path: line, set: {}, unset: [], re: new RegExp(`^${line.split("*").map(esc).join("(?.*)")}$`), lines: [] }; continue; } rule!.lines.push(line); if (line.startsWith("! ")) { rule!.unset.push(line.slice(2).toLowerCase()); continue; } const i = line.indexOf(":"); const name = line.slice(0, i).trim().toLowerCase(); const value = line.slice(i + 1).trim(); rule!.set[name] = rule!.set[name] ? `${rule!.set[name]}, ${value}` : value; } if (rule) rules.push(rule); return rules; } function served(rules: Rule[], pathname: string, contentType: string): Headers { const h = new Headers({ "content-type": contentType }); const setOnce = new Set(); for (const r of rules.filter((x) => x.re.test(pathname))) { for (const k of r.unset) h.delete(k); for (const [k, v] of Object.entries(r.set)) { if (setOnce.has(k)) h.append(k, v); else { h.set(k, v); setOnce.add(k); } } } return h; } const RULES = parse(TEXT); test("wrangler's limits: at most 100 rules, no line over 2,000 characters, one splat a rule", () => { assert.ok(RULES.length <= 100, `${RULES.length} rules`); for (const line of TEXT.split("\n")) assert.ok(line.length <= 2000, line.slice(0, 60)); for (const r of RULES) assert.ok((r.path.match(/\*/g) ?? []).length <= 1, r.path); }); test("every /source/tree override detaches Content-Type before it sets its own", () => { const overrides = RULES.filter((r) => r.path.startsWith("/source/tree/") && r.path !== "/source/tree/*" && "content-type" in r.set); assert.ok(overrides.length >= 7, overrides.map((r) => r.path).join(" ")); for (const r of overrides) { const unset = r.lines.findIndex((l) => l.toLowerCase() === "! content-type"); const set = r.lines.findIndex((l) => /^content-type:/i.test(l)); assert.ok(unset !== -1 && unset < set, `${r.path} must say "! Content-Type" before it sets one`); } }); test("the raw tree is text, its pages HTML, its binaries their own type — each ONE value", () => { const cases: Array<[string, string, string]> = [ ["/source/tree/common/lib/paths.ts", "video/mp2t", "text/plain; charset=utf-8"], ["/source/tree/README.md", "text/markdown", "text/plain; charset=utf-8"], ["/source/tree/", "text/html", "text/html; charset=utf-8"], ["/source/tree/common/", "text/html", "text/html; charset=utf-8"], ["/source/tree/umtool/report-to-video/fonts/Archivo%5Bwdth%2Cwght%5D.ttf", "font/ttf", "font/ttf"], ["/source/tree/homepage/app/docs/%5Bslug%5D/page.tsx", "application/octet-stream", "text/plain; charset=utf-8"], ["/source/tree/umtool/models/yunet.onnx", "application/octet-stream", "application/octet-stream"], ["/source/tree/homepage/public/icon.svg", "image/svg+xml", "image/svg+xml"], // A missing directory: Pages answers with a 404 page, on this rule. ["/source/tree/no/such/dir/", "text/html", "text/html; charset=utf-8"], ]; for (const [p, served0, want] of cases) { const h = served(RULES, p, served0); assert.equal(h.get("content-type"), want, p); assert.equal(h.get("x-content-type-options"), "nosniff", p); } assert.match(served(RULES, "/source/tree/a.svg", "image/svg+xml").get("content-security-policy")!, /default-src 'none'/); assert.equal(served(RULES, "/source/manifest.json", "application/json").get("content-type"), "application/json"); }); test("the history pages (release 15 slice SG) are not indexed, like the raw tree, and keep their own types", () => { const cases: Array<[string, string]> = [ ["/source/git/log.html", "text/html"], ["/source/git/commit/0123456789abcdef0123456789abcdef01234567.html", "text/html"], ["/source/git/atom.xml", "application/xml"], ["/source/git/style.css", "text/css"], ]; for (const [p, type] of cases) { const h = served(RULES, p, type); assert.equal(h.get("x-robots-tag"), "noindex", p); assert.equal(h.get("content-type"), type, `${p} keeps its type: no /source/tree rule applies`); } assert.equal(served(RULES, "/source/tree/README.md", "text/markdown").get("x-robots-tag"), "noindex"); assert.equal(served(RULES, "/source/", "text/html").get("x-robots-tag"), null, "the /source/ page itself is indexed"); });