Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit aab32fe5f987522b39bce42934849eea556bc30d
parent df5107ec4a64ba18e9377953e5302ea96597d9e2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 28 Sep 2026 14:22:47 -0400

common: the audit report names a literal by where it was written and prints no object bytes; a tracked 404.html is refused

Review Q3/L4/L5: a label is `denylist line 3 (len 5)`, `scrub line 2 lhs (len
11)` or `built-in home rule (len 11)` — no character of the literal (a first
letter and a length all but spell a short first name). A hit is listed by
kind, object id (the blob's path in history once known), byte offset and, for a
commit or tag, the header field it sits in (author, committer, tagger, …) or
`message`; a tree hit by its entry number. The ±24-byte context is gone: it
printed whatever sat beside a denied name. `operatorLines` drops a byte-order
mark and CRLF endings from the denylist. Review L1: writeTreeIndexes refuses a
tracked `404.html` anywhere — Pages would serve it, as HTML on this origin, for
every missing path below it.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/publish/sourceAudit.test.ts | 74+++++++++++++++++++++++++++++++++++++++++++++++---------------------------
Mcommon/publish/sourceAudit.ts | 172++++++++++++++++++++++++++++++++++++++++++++++---------------------------------
Mcommon/publish/sourceTree.test.ts | 12+++++++++++-
Mcommon/publish/sourceTree.ts | 12++++++++++--
4 files changed, 168 insertions(+), 102 deletions(-)

diff --git a/common/publish/sourceAudit.test.ts b/common/publish/sourceAudit.test.ts @@ -13,8 +13,8 @@ import { formatAuditReport, literalLabel, maskLiterals, + objectField, parseDenylist, - redactHit, runGitleaks, scanBuffer, treeEntryNames, @@ -38,7 +38,8 @@ after(() => rmSync(TMP, { recursive: true, force: true })); // The planted literal every test hunts for. Never a real name. const PLANTED = "plantedhome"; -const lits = (...xs: string[]): Literal[] => xs.map((x) => ({ bytes: Buffer.from(x), ci: false })); +const lits = (...xs: string[]): Literal[] => + xs.map((x, i) => ({ bytes: Buffer.from(x), ci: false, from: `denylist line ${i + 1}` })); let n = 0; function repo(): string { @@ -64,11 +65,17 @@ function commit(dir: string, files: Record<string, string>, message: string, aut test("the denylist: one literal a line, i: for any case, comments and blanks skipped, duplicates once", () => { const list = parseDenylist(`# a comment\n\n${PLANTED}\ni:MiXeD\n ${PLANTED} \r\ni:\n`); assert.equal(list.length, 2); - assert.deepEqual(list[0], { bytes: Buffer.from(PLANTED), ci: false }); - assert.deepEqual(list[1], { bytes: Buffer.from("mixed"), ci: true }); - // Named by position, first character and length — never by its bytes. - assert.equal(literalLabel(list, 0), `#1 (p…, len ${PLANTED.length})`); - assert.equal(literalLabel(list, 1), "#2 (m…, len 5, any case)"); + assert.deepEqual(list[0], { bytes: Buffer.from(PLANTED), ci: false, from: "denylist line 3" }); + assert.deepEqual(list[1], { bytes: Buffer.from("mixed"), ci: true, from: "denylist line 4" }); + // Named by where it was written and its length — none of its characters. + assert.equal(literalLabel(list, 0), `denylist line 3 (len ${PLANTED.length})`); + assert.equal(literalLabel(list, 1), "denylist line 4 (len 5, any case)"); +}); + +test("a byte-order mark and CRLF endings never become part of a literal", () => { + const list = parseDenylist(`\uFEFF${PLANTED}\r\ni:Other\r\n`); + assert.deepEqual(list.map((l) => l.bytes.toString()), [PLANTED, "other"]); + assert.equal(list[1].from, "denylist line 2"); }); test("scanBuffer finds every occurrence; an i: literal matches any ASCII case", () => { @@ -80,20 +87,25 @@ test("scanBuffer finds every occurrence; an i: literal matches any ASCII case", ); }); -test("redaction masks every occurrence in the window, a half-shown one included, and dots the unprintable", () => { - const list = lits(PLANTED); - // The second occurrence starts 20 bytes after the first ends: the ±24-byte - // window shows only its first four bytes, which must be masked too. - const buf = Buffer.from(`\u0001st ${PLANTED}${"-".repeat(20)}${PLANTED} tail`); - const hits = scanBuffer(buf, list); - assert.equal(hits.length, 2); - const shown = redactHit(buf, hits[0], hits); - assert.equal(shown, `.st [REDACTED]${"-".repeat(20)}[REDACTED]`); - assert.ok(!shown.includes("plan"), shown); +test("maskLiterals merges overlapping occurrences into one mark and leaves the rest", () => { + const list = lits(PLANTED, "homeplanted"); assert.equal(maskLiterals(`/srv/${PLANTED}/x ${PLANTED}`, list), "/srv/[REDACTED]/x [REDACTED]"); + assert.equal(maskLiterals(`a plantedhomeplanted b`, list), "a [REDACTED] b"); assert.equal(maskLiterals("nothing here", list), "nothing here"); }); +test("objectField names a commit's header field or its message, never its bytes", () => { + const commit = Buffer.from( + "tree 1234\nparent 5678\nauthor A <a@x> 1 +0000\ncommitter C <c@x> 1 +0000\ngpgsig -----BEGIN\n sig line\n -----END\n\nthe message\n", + ); + const at = (s: string) => commit.indexOf(s); + assert.equal(objectField(commit, at("a@x")), "author"); + assert.equal(objectField(commit, at("c@x")), "committer"); + assert.equal(objectField(commit, at("sig line")), "gpgsig"); + assert.equal(objectField(commit, at("message")), "message"); + assert.equal(objectField(commit, at("5678")), "parent"); +}); + test("a tree's entry names are what is scanned, not its binary ids", () => { const id = Buffer.alloc(20, 0x70); // 'p' x20: would match "ppp" if ids were read const tree = Buffer.concat([ @@ -112,20 +124,28 @@ test("the object walk finds a literal in a blob, a commit message, an author lin commit(dir, { "README.md": "clean 3\n" }, "an identity", `Planted <${PLANTED}@example.invalid>`); commit(dir, { [`dir-${PLANTED}/x.txt`]: "x\n" }, "a name"); const result = await auditObjects(path.join(dir, ".git"), lits(PLANTED)); - const kinds = result.hits.map((h) => h.kind).sort(); + const kinds = result.hits.map((h) => `${h.kind}:${h.field ?? ""}`).sort(); // The message and the author line are one commit object each. - assert.deepEqual(kinds, ["blob", "commit", "commit", "tree"]); + // The tree's hit is its second entry: README.md, dir-…, notes.txt. + assert.deepEqual(kinds, ["blob:", "commit:author", "commit:message", "tree:entry 2"]); assert.equal(result.commits, 5); const objects = execFileSync("git", ["count-objects", "-v"], { cwd: dir }).toString(); assert.equal(result.objects, Number(/^count: (\d+)/m.exec(objects)![1]), "every object was read"); - for (const h of result.hits) assert.ok(!h.context?.includes(PLANTED), h.context); - // The report names the literal by number and never prints it. + // The report names the literal by where it was written, and prints no + // byte of any object: not the literal, not what sits beside it. const report = formatAuditReport(result, lits(PLANTED), { scrubFile: path.join(os.homedir(), ".config", "archilyzer", "source-scrub.txt"), }).join("\n"); assert.ok(!report.includes(PLANTED), report); + for (const beside of ["/srv/", "/data", "example.invalid", "Planted <", "the message says"]) { + assert.ok(!report.includes(beside), `the report carries "${beside}" from an object`); + } assert.match(report, /AUDIT REFUSED: 4 hits/); - assert.match(report, /#1 \(p…, len 11\): 1 in blob, 2 in commits, 1 in tree/); + assert.match(report, /denylist line 1 \(len 11\): 1 in blob, 2 in commits, 1 in tree/); + assert.match(report, /commit [0-9a-f]{12} \(author, byte \d+\): denylist line 1 \(len 11\)/); + assert.match(report, /commit [0-9a-f]{12} \(message, byte \d+\): denylist line 1 \(len 11\)/); + assert.match(report, /blob [0-9a-f]{12} \(byte 10\): denylist line 1 \(len 11\)/); + assert.match(report, /tree [0-9a-f]{12} \(entry 2\): denylist line 1 \(len 11\)/); assert.match(report, /add a rule to ~\/\.config\/archilyzer\/source-scrub\.txt or drop the file from history, then re-run\.$/); }); @@ -163,11 +183,11 @@ test("the staged-file sweep reads contents, names and gzip'd bytes decompressed, writeFileSync(path.join(dir, "t.tar.gz"), gzipSync(Buffer.from(`inside ${PLANTED}`))); writeFileSync(path.join(dir, "pack-1.pack"), PLANTED); // the object walk's job const result = await auditFiles(dir, lits(PLANTED)); - assert.deepEqual(result.hits.map((h) => h.where).sort(), [ - "t.tar.gz", - "tree/bad.txt", - "tree/d-[REDACTED]", - "tree/d-[REDACTED]/f.txt", + assert.deepEqual(result.hits.map((h) => `${h.where} ${h.field}`).sort(), [ + "t.tar.gz decompressed", + "tree/bad.txt contents", + "tree/d-[REDACTED] path", + "tree/d-[REDACTED]/f.txt path", ]); assert.equal(result.files, 5); symlinkSync("/etc/hostname", path.join(dir, "tree", "link")); diff --git a/common/publish/sourceAudit.ts b/common/publish/sourceAudit.ts @@ -14,11 +14,15 @@ // - runGitleaks: the secret scanner over the mirror's history, when it is // installed (a WARNING, not a refusal, when it is not). // -// THE REPORT NEVER PRINTS A LITERAL. A literal is named `#n (x…, len L)` — its -// position in the list, its first character, its length — and every byte of -// every occurrence of every literal inside a context window is replaced by -// `[REDACTED]`, including an occurrence the window only half covers. Paths and -// subjects printed from the repo go through the same mask. +// THE REPORT NEVER PRINTS A LITERAL, AND NO BYTES OF THE OBJECT AROUND ONE. A +// literal is named by where the operator wrote it — `denylist line 3 (len 5)`, +// `scrub line 2 lhs (len 11)`, `built-in home rule (len 11)` — never by any of +// its characters. A hit is named by its object (kind, id, the path when one +// is known), the byte offset, and for a commit or tag the field it sits in +// (author, committer, tagger, message): a context window would print what sits +// NEXT to a denied name (a surname, the rest of an address), which is exactly +// as private. Paths and child lines quoted in the log are masked +// (`[REDACTED]`). import { spawn } from "node:child_process"; import { existsSync, statSync, accessSync, constants } from "node:fs"; @@ -40,9 +44,11 @@ export class SourceRefusal extends Error { /** * One denied literal. `ci` literals match ASCII case-insensitively: their - * `bytes` are stored folded, and are searched for in a folded copy. + * `bytes` are stored folded, and are searched for in a folded copy. `from` + * says where the operator wrote it (`denylist line 3`), which is how a report + * names it. */ -export type Literal = { bytes: Buffer; ci: boolean }; +export type Literal = { bytes: Buffer; ci: boolean; from?: string }; const LOWER_A = 0x61; const UPPER_A = 0x41; @@ -59,22 +65,33 @@ export function foldAscii(buf: Uint8Array): Buffer { } /** + * An operator file's text as lines: a UTF-8 byte-order mark at the start is + * dropped and CRLF endings become LF. An editor that writes either must not + * turn the first rule (and the denial it implies) into one that never matches. + */ +export function operatorLines(text: string): string[] { + return text.replace(/^\uFEFF/, "").split("\n").map((l) => l.replace(/\r$/, "")); +} + +/** * The denylist file: one literal per line. `i:` in front makes it ASCII * case-insensitive. Blank lines and lines starting with `#` are skipped; - * surrounding whitespace is trimmed. Duplicates collapse. + * surrounding whitespace is trimmed. Duplicates collapse (the first line + * names it). */ export function parseDenylist(text: string): Literal[] { const out: Literal[] = []; - for (const raw of text.split("\n")) { + operatorLines(text).forEach((raw, i) => { const line = raw.trim(); - if (line === "" || line.startsWith("#")) continue; + if (line === "" || line.startsWith("#")) return; + const from = `denylist line ${i + 1}`; if (line.startsWith("i:")) { const lit = line.slice(2).trim(); - if (lit) out.push({ bytes: foldAscii(Buffer.from(lit, "utf8")), ci: true }); + if (lit) out.push({ bytes: foldAscii(Buffer.from(lit, "utf8")), ci: true, from }); } else { - out.push({ bytes: Buffer.from(line, "utf8"), ci: false }); + out.push({ bytes: Buffer.from(line, "utf8"), ci: false, from }); } - } + }); return dedupeLiterals(out); } @@ -90,12 +107,14 @@ export function dedupeLiterals(list: Literal[]): Literal[] { return out; } -/** How a literal is named in a report: never its bytes. */ +/** + * How a literal is named in a report: where it was written and its length — + * never any of its characters (a first letter and a length all but spell a + * short first name). + */ export function literalLabel(literals: readonly Literal[], index: number): string { const l = literals[index]; - const first = l.bytes.subarray(0, 4).toString("utf8").slice(0, 1); - const shown = /^[\x21-\x7e]$/.test(first) ? first : "?"; - return `#${index + 1} (${shown}…, len ${l.bytes.length}${l.ci ? ", any case" : ""})`; + return `${l.from ?? `literal ${index + 1}`} (len ${l.bytes.length}${l.ci ? ", any case" : ""})`; } // ── scanning ──────────────────────────────────────────────────────────────── @@ -121,37 +140,6 @@ export function scanBuffer(buf: Buffer, literals: readonly Literal[]): Hit[] { return hits.sort((a, b) => a.offset - b.offset || a.lit - b.lit); } -const RADIUS = 24; - -/** - * ±24 bytes around `hit`, printable: every byte any hit covers becomes one - * `[REDACTED]` per run (so a second literal the window only half shows is - * masked too), anything outside 0x20–0x7e becomes `.`. - */ -export function redactHit(buf: Buffer, hit: Hit, allHits: readonly Hit[]): string { - const from = Math.max(0, hit.offset - RADIUS); - const to = Math.min(buf.length, hit.offset + hit.length + RADIUS); - const covered = new Uint8Array(to - from); - for (const h of allHits) { - const a = Math.max(from, h.offset); - const b = Math.min(to, h.offset + h.length); - for (let i = a; i < b; i++) covered[i - from] = 1; - } - let out = ""; - let inRun = false; - for (let i = from; i < to; i++) { - if (covered[i - from]) { - if (!inRun) out += "[REDACTED]"; - inRun = true; - continue; - } - inRun = false; - const b = buf[i]; - out += b >= 0x20 && b <= 0x7e ? String.fromCharCode(b) : "."; - } - return out; -} - /** `text` with every occurrence of every literal replaced by `[REDACTED]`. */ export function maskLiterals(text: string, literals: readonly Literal[]): string { const buf = Buffer.from(text, "utf8"); @@ -181,13 +169,17 @@ const KIND_ORDER: readonly HitKind[] = ["blob", "commit", "tree", "tag", "file", export type AuditHit = { kind: HitKind; - // An object id (blob/commit/tree/tag), a path relative to the staged dir - // (file) or the finding's rule (gitleaks). + // An object id (blob/commit/tree/tag) — with the blob's path in history + // once known — a path relative to the staged dir (file), or the finding + // (gitleaks). Paths are masked. where: string; // A literal's index, or -1 for a gitleaks finding. lit: number; - // The redacted context (only the first CONTEXTS hits carry one). - context?: string; + // The byte offset of the hit in the object or file (-1 for gitleaks). + offset: number; + // Where in the object, without its bytes: a commit's or tag's header field + // (`author`, `committer`, `tagger`, …) or `message`; a tree's `entry N`. + field?: string; }; export type AuditResult = { @@ -199,9 +191,9 @@ export type AuditResult = { hits: AuditHit[]; }; -// Only this many hits keep a context line: a literal every commit carries +// Only this many hits get a line of their own: a literal every commit carries // (the gate's own planted `Co-Authored-By`) would otherwise print thousands. -const CONTEXTS = 20; +const LISTED = 20; export function emptyAudit(literals: number): AuditResult { return { literals, objects: 0, commits: 0, files: 0, gitleaks: "not run", hits: [] }; @@ -211,20 +203,50 @@ function record( result: AuditResult, kind: HitKind, where: string, - buf: Buffer, hits: Hit[], + fieldOf?: (offset: number) => string, + // A tree's hits are offsets into its joined entry NAMES, not the object: + // the entry number says where, and the offset is left out. + withOffset = true, ): void { for (const h of hits) { - const withContext = result.hits.length < CONTEXTS; result.hits.push({ kind, where, lit: h.lit, - ...(withContext ? { context: redactHit(buf, h, hits) } : {}), + offset: withOffset ? h.offset : -1, + ...(fieldOf ? { field: fieldOf(h.offset) } : {}), }); } } +/** + * Which part of a commit or tag object `offset` falls in: `message` past the + * blank line that ends the headers, else the header line's keyword (`author`, + * `committer`, `tagger`, `tree`, `parent`, …) — a word git wrote, never the + * operator's bytes. + */ +export function objectField(data: Buffer, offset: number): string { + const end = data.indexOf("\n\n"); + if (end !== -1 && offset > end) return "message"; + let lineStart = data.lastIndexOf(0x0a, offset - 1) + 1; + // A continuation line (a signature, a mergetag) begins with a space: walk + // back to the header it continues. + while (lineStart > 0 && data[lineStart] === 0x20) { + lineStart = data.lastIndexOf(0x0a, lineStart - 2) + 1; + } + const sp = data.indexOf(0x20, lineStart); + const word = data.subarray(lineStart, sp === -1 ? lineStart : sp).toString("latin1"); + return /^[a-z][a-z-]{0,15}$/.test(word) ? word : "header"; +} + +/** A tree hit's entry number (1-based) in the NUL-separated names buffer. */ +function treeEntry(names: Buffer, offset: number): string { + let n = 1; + for (let i = names.indexOf(0, 0); i !== -1 && i < offset; i = names.indexOf(0, i + 1)) n++; + return `entry ${n}`; +} + // ── the object walk ───────────────────────────────────────────────────────── /** The git environment a child must not inherit: it would point git elsewhere. */ @@ -292,12 +314,16 @@ export async function auditObjects( if (type === "tree") { const names = treeEntryNames(body, oid.length / 2); const hits = scanBuffer(names, literals); - if (hits.length) record(result, "tree", oid, names, hits); + if (hits.length) record(result, "tree", oid, hits, (o) => treeEntry(names, o), false); return; } const hits = scanBuffer(body, literals); - if (hits.length) { - record(result, type === "commit" ? "commit" : type === "tag" ? "tag" : "blob", oid, body, hits); + if (hits.length === 0) return; + if (type === "commit" || type === "tag") { + const data = body; + record(result, type, oid, hits, (o) => objectField(data, o)); + } else { + record(result, "blob", oid, hits); } }; @@ -410,9 +436,8 @@ export async function auditFiles( const abs = path.join(dir, rel); for (const ent of await readdir(abs, { withFileTypes: true })) { const r = rel ? `${rel}/${ent.name}` : ent.name; - const nameBuf = Buffer.from(r, "utf8"); - const nameHits = scanBuffer(nameBuf, literals); - if (nameHits.length) record(result, "file", maskLiterals(r, literals), nameBuf, nameHits); + const nameHits = scanBuffer(Buffer.from(r, "utf8"), literals); + if (nameHits.length) record(result, "file", maskLiterals(r, literals), nameHits, () => "path"); if (ent.isSymbolicLink()) { throw new SourceRefusal(`the stage holds a symlink (${maskLiterals(r, literals)}); nothing published may point outside it`); } @@ -425,7 +450,9 @@ export async function auditFiles( let data = await readFile(path.join(abs, ent.name)); if (ent.name.endsWith(".gz")) data = gunzipSync(data); const hits = scanBuffer(data, literals); - if (hits.length) record(result, "file", maskLiterals(r, literals), data, hits); + if (hits.length) { + record(result, "file", maskLiterals(r, literals), hits, () => (ent.name.endsWith(".gz") ? "decompressed" : "contents")); + } } }; if ((await lstat(dir)).isDirectory()) await walk(""); @@ -519,6 +546,7 @@ export async function runGitleaks( opts.literals, ), lit: -1, + offset: -1, })), }; } @@ -576,8 +604,9 @@ export function tildify(p: string): string { /** * The report, as lines. Clean: one line of counts. Hits: the counts per - * literal and kind, the first hits with their redacted context, and what to - * do. No line carries a literal. + * literal and kind, then the first hits — each by object, field and byte + * offset, never by its bytes — and what to do. No line carries a literal or + * anything read from beside one. */ export function formatAuditReport( result: AuditResult, @@ -604,13 +633,12 @@ export function formatAuditReport( }); lines.push(`[source] ${label}: ${parts.join(", ")}`); } - const shown = result.hits.filter((h) => h.context !== undefined || h.kind === "gitleaks").slice(0, CONTEXTS); + const shown = result.hits.slice(0, LISTED); for (const h of shown) { - const label = h.lit === -1 ? "" : ` ${literalLabel(literals, h.lit)}`; - lines.push( - `[source] ${h.kind} ${h.kind === "file" || h.kind === "gitleaks" ? h.where : shortWhere(h.where)}${label}` + - (h.context !== undefined ? `: ${h.context}` : ""), - ); + const where = h.kind === "file" || h.kind === "gitleaks" ? h.where : shortWhere(h.where); + const at = [h.field, h.offset >= 0 ? `byte ${h.offset}` : ""].filter(Boolean).join(", "); + const label = h.lit === -1 ? "" : `: ${literalLabel(literals, h.lit)}`; + lines.push(`[source] ${h.kind} ${where}${at ? ` (${at})` : ""}${label}`); } if (result.hits.length > shown.length) { lines.push(`[source] … and ${result.hits.length - shown.length} more`); diff --git a/common/publish/sourceTree.test.ts b/common/publish/sourceTree.test.ts @@ -62,7 +62,7 @@ test("breadcrumbs hop up relatively; the root has no `..` row", () => { assert.match(deep, /<tr><td class="name"><a href="\.\.\/">\.\.<\/a>/); }); -test("writeTreeIndexes pages every directory and counts the tree, refusing a tracked index.html or a symlink", async () => { +test("writeTreeIndexes pages every directory and counts the tree, refusing a tracked index.html, a 404.html or a symlink", async () => { const tree = path.join(TMP, "t1"); mkdirSync(path.join(tree, "app", "[slug]"), { recursive: true }); writeFileSync(path.join(tree, "README.md"), "hello\n"); @@ -79,6 +79,16 @@ test("writeTreeIndexes pages every directory and counts the tree, refusing a tra writeFileSync(path.join(withIndex, "docs", "index.html"), "<p>tracked</p>"); await assert.rejects(writeTreeIndexes(withIndex, META), (e) => e instanceof SourceRefusal && /docs\/index\.html/.test(e.message)); + // A tracked 404.html anywhere: Pages would serve it, as HTML on this + // origin, for every missing path below its directory. + const with404 = path.join(TMP, "t4"); + mkdirSync(path.join(with404, "docs", "deep"), { recursive: true }); + writeFileSync(path.join(with404, "docs", "deep", "404.html"), "<script>x</script>"); + await assert.rejects( + writeTreeIndexes(with404, META), + (e) => e instanceof SourceRefusal && /docs\/deep\/404\.html; Pages would serve it, as HTML/.test(e.message), + ); + const withLink = path.join(TMP, "t3"); mkdirSync(withLink); symlinkSync("/etc/hostname", path.join(withLink, "leak")); diff --git a/common/publish/sourceTree.ts b/common/publish/sourceTree.ts @@ -134,8 +134,9 @@ export function renderTreeIndex(opts: { /** * Write an index.html into every directory under `treeDir`, the root * included. REFUSES when a directory already holds an `index.html` (a tracked - * one would be overwritten, or served in the index's place) or when anything - * is a symlink. Returns the tree's own files and bytes (not the pages) and how + * one would be overwritten, or served in the index's place) or a `404.html` + * (Pages would serve it as HTML for any missing path below it), or when + * anything is a symlink. Returns the tree's own files and bytes (not the pages) and how * many directories got a page. */ export async function writeTreeIndexes( @@ -161,6 +162,13 @@ export async function writeTreeIndexes( `the tree already has ${r}; its directory page would replace it — rename the file or publish without the raw tree`, ); } + // Pages answers a missing path with the NEAREST 404.html, as HTML (the + // directory-page rule): a tracked one would run on this origin. + if (ent.name === "404.html") { + throw new SourceRefusal( + `the tree has ${r}; Pages would serve it, as HTML on this origin, for every missing path below it — rename the file or publish without the raw tree`, + ); + } const { size } = await stat(path.join(abs, ent.name)); entries.push({ name: ent.name, dir: false, bytes: size }); totals.files++;