commit aab32fe5f987522b39bce42934849eea556bc30d
parent df5107ec4a64ba18e9377953e5302ea96597d9e2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 14:22:47 -0400
common: the audit report names a literal by where it was written and prints no object bytes; a tracked 404.html is refused
Review Q3/L4/L5: a label is `denylist line 3 (len 5)`, `scrub line 2 lhs (len
11)` or `built-in home rule (len 11)` — no character of the literal (a first
letter and a length all but spell a short first name). A hit is listed by
kind, object id (the blob's path in history once known), byte offset and, for a
commit or tag, the header field it sits in (author, committer, tagger, …) or
`message`; a tree hit by its entry number. The ±24-byte context is gone: it
printed whatever sat beside a denied name. `operatorLines` drops a byte-order
mark and CRLF endings from the denylist. Review L1: writeTreeIndexes refuses a
tracked `404.html` anywhere — Pages would serve it, as HTML on this origin, for
every missing path below it.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
4 files changed, 168 insertions(+), 102 deletions(-)
diff --git a/common/publish/sourceAudit.test.ts b/common/publish/sourceAudit.test.ts
@@ -13,8 +13,8 @@ import {
formatAuditReport,
literalLabel,
maskLiterals,
+ objectField,
parseDenylist,
- redactHit,
runGitleaks,
scanBuffer,
treeEntryNames,
@@ -38,7 +38,8 @@ after(() => rmSync(TMP, { recursive: true, force: true }));
// The planted literal every test hunts for. Never a real name.
const PLANTED = "plantedhome";
-const lits = (...xs: string[]): Literal[] => xs.map((x) => ({ bytes: Buffer.from(x), ci: false }));
+const lits = (...xs: string[]): Literal[] =>
+ xs.map((x, i) => ({ bytes: Buffer.from(x), ci: false, from: `denylist line ${i + 1}` }));
let n = 0;
function repo(): string {
@@ -64,11 +65,17 @@ function commit(dir: string, files: Record<string, string>, message: string, aut
test("the denylist: one literal a line, i: for any case, comments and blanks skipped, duplicates once", () => {
const list = parseDenylist(`# a comment\n\n${PLANTED}\ni:MiXeD\n ${PLANTED} \r\ni:\n`);
assert.equal(list.length, 2);
- assert.deepEqual(list[0], { bytes: Buffer.from(PLANTED), ci: false });
- assert.deepEqual(list[1], { bytes: Buffer.from("mixed"), ci: true });
- // Named by position, first character and length — never by its bytes.
- assert.equal(literalLabel(list, 0), `#1 (p…, len ${PLANTED.length})`);
- assert.equal(literalLabel(list, 1), "#2 (m…, len 5, any case)");
+ assert.deepEqual(list[0], { bytes: Buffer.from(PLANTED), ci: false, from: "denylist line 3" });
+ assert.deepEqual(list[1], { bytes: Buffer.from("mixed"), ci: true, from: "denylist line 4" });
+ // Named by where it was written and its length — none of its characters.
+ assert.equal(literalLabel(list, 0), `denylist line 3 (len ${PLANTED.length})`);
+ assert.equal(literalLabel(list, 1), "denylist line 4 (len 5, any case)");
+});
+
+test("a byte-order mark and CRLF endings never become part of a literal", () => {
+ const list = parseDenylist(`\uFEFF${PLANTED}\r\ni:Other\r\n`);
+ assert.deepEqual(list.map((l) => l.bytes.toString()), [PLANTED, "other"]);
+ assert.equal(list[1].from, "denylist line 2");
});
test("scanBuffer finds every occurrence; an i: literal matches any ASCII case", () => {
@@ -80,20 +87,25 @@ test("scanBuffer finds every occurrence; an i: literal matches any ASCII case",
);
});
-test("redaction masks every occurrence in the window, a half-shown one included, and dots the unprintable", () => {
- const list = lits(PLANTED);
- // The second occurrence starts 20 bytes after the first ends: the ±24-byte
- // window shows only its first four bytes, which must be masked too.
- const buf = Buffer.from(`\u0001st ${PLANTED}${"-".repeat(20)}${PLANTED} tail`);
- const hits = scanBuffer(buf, list);
- assert.equal(hits.length, 2);
- const shown = redactHit(buf, hits[0], hits);
- assert.equal(shown, `.st [REDACTED]${"-".repeat(20)}[REDACTED]`);
- assert.ok(!shown.includes("plan"), shown);
+test("maskLiterals merges overlapping occurrences into one mark and leaves the rest", () => {
+ const list = lits(PLANTED, "homeplanted");
assert.equal(maskLiterals(`/srv/${PLANTED}/x ${PLANTED}`, list), "/srv/[REDACTED]/x [REDACTED]");
+ assert.equal(maskLiterals(`a plantedhomeplanted b`, list), "a [REDACTED] b");
assert.equal(maskLiterals("nothing here", list), "nothing here");
});
+test("objectField names a commit's header field or its message, never its bytes", () => {
+ const commit = Buffer.from(
+ "tree 1234\nparent 5678\nauthor A <a@x> 1 +0000\ncommitter C <c@x> 1 +0000\ngpgsig -----BEGIN\n sig line\n -----END\n\nthe message\n",
+ );
+ const at = (s: string) => commit.indexOf(s);
+ assert.equal(objectField(commit, at("a@x")), "author");
+ assert.equal(objectField(commit, at("c@x")), "committer");
+ assert.equal(objectField(commit, at("sig line")), "gpgsig");
+ assert.equal(objectField(commit, at("message")), "message");
+ assert.equal(objectField(commit, at("5678")), "parent");
+});
+
test("a tree's entry names are what is scanned, not its binary ids", () => {
const id = Buffer.alloc(20, 0x70); // 'p' x20: would match "ppp" if ids were read
const tree = Buffer.concat([
@@ -112,20 +124,28 @@ test("the object walk finds a literal in a blob, a commit message, an author lin
commit(dir, { "README.md": "clean 3\n" }, "an identity", `Planted <${PLANTED}@example.invalid>`);
commit(dir, { [`dir-${PLANTED}/x.txt`]: "x\n" }, "a name");
const result = await auditObjects(path.join(dir, ".git"), lits(PLANTED));
- const kinds = result.hits.map((h) => h.kind).sort();
+ const kinds = result.hits.map((h) => `${h.kind}:${h.field ?? ""}`).sort();
// The message and the author line are one commit object each.
- assert.deepEqual(kinds, ["blob", "commit", "commit", "tree"]);
+ // The tree's hit is its second entry: README.md, dir-…, notes.txt.
+ assert.deepEqual(kinds, ["blob:", "commit:author", "commit:message", "tree:entry 2"]);
assert.equal(result.commits, 5);
const objects = execFileSync("git", ["count-objects", "-v"], { cwd: dir }).toString();
assert.equal(result.objects, Number(/^count: (\d+)/m.exec(objects)![1]), "every object was read");
- for (const h of result.hits) assert.ok(!h.context?.includes(PLANTED), h.context);
- // The report names the literal by number and never prints it.
+ // The report names the literal by where it was written, and prints no
+ // byte of any object: not the literal, not what sits beside it.
const report = formatAuditReport(result, lits(PLANTED), {
scrubFile: path.join(os.homedir(), ".config", "archilyzer", "source-scrub.txt"),
}).join("\n");
assert.ok(!report.includes(PLANTED), report);
+ for (const beside of ["/srv/", "/data", "example.invalid", "Planted <", "the message says"]) {
+ assert.ok(!report.includes(beside), `the report carries "${beside}" from an object`);
+ }
assert.match(report, /AUDIT REFUSED: 4 hits/);
- assert.match(report, /#1 \(p…, len 11\): 1 in blob, 2 in commits, 1 in tree/);
+ assert.match(report, /denylist line 1 \(len 11\): 1 in blob, 2 in commits, 1 in tree/);
+ assert.match(report, /commit [0-9a-f]{12} \(author, byte \d+\): denylist line 1 \(len 11\)/);
+ assert.match(report, /commit [0-9a-f]{12} \(message, byte \d+\): denylist line 1 \(len 11\)/);
+ assert.match(report, /blob [0-9a-f]{12} \(byte 10\): denylist line 1 \(len 11\)/);
+ assert.match(report, /tree [0-9a-f]{12} \(entry 2\): denylist line 1 \(len 11\)/);
assert.match(report, /add a rule to ~\/\.config\/archilyzer\/source-scrub\.txt or drop the file from history, then re-run\.$/);
});
@@ -163,11 +183,11 @@ test("the staged-file sweep reads contents, names and gzip'd bytes decompressed,
writeFileSync(path.join(dir, "t.tar.gz"), gzipSync(Buffer.from(`inside ${PLANTED}`)));
writeFileSync(path.join(dir, "pack-1.pack"), PLANTED); // the object walk's job
const result = await auditFiles(dir, lits(PLANTED));
- assert.deepEqual(result.hits.map((h) => h.where).sort(), [
- "t.tar.gz",
- "tree/bad.txt",
- "tree/d-[REDACTED]",
- "tree/d-[REDACTED]/f.txt",
+ assert.deepEqual(result.hits.map((h) => `${h.where} ${h.field}`).sort(), [
+ "t.tar.gz decompressed",
+ "tree/bad.txt contents",
+ "tree/d-[REDACTED] path",
+ "tree/d-[REDACTED]/f.txt path",
]);
assert.equal(result.files, 5);
symlinkSync("/etc/hostname", path.join(dir, "tree", "link"));
diff --git a/common/publish/sourceAudit.ts b/common/publish/sourceAudit.ts
@@ -14,11 +14,15 @@
// - runGitleaks: the secret scanner over the mirror's history, when it is
// installed (a WARNING, not a refusal, when it is not).
//
-// THE REPORT NEVER PRINTS A LITERAL. A literal is named `#n (x…, len L)` — its
-// position in the list, its first character, its length — and every byte of
-// every occurrence of every literal inside a context window is replaced by
-// `[REDACTED]`, including an occurrence the window only half covers. Paths and
-// subjects printed from the repo go through the same mask.
+// THE REPORT NEVER PRINTS A LITERAL, AND NO BYTES OF THE OBJECT AROUND ONE. A
+// literal is named by where the operator wrote it — `denylist line 3 (len 5)`,
+// `scrub line 2 lhs (len 11)`, `built-in home rule (len 11)` — never by any of
+// its characters. A hit is named by its object (kind, id, the path when one
+// is known), the byte offset, and for a commit or tag the field it sits in
+// (author, committer, tagger, message): a context window would print what sits
+// NEXT to a denied name (a surname, the rest of an address), which is exactly
+// as private. Paths and child lines quoted in the log are masked
+// (`[REDACTED]`).
import { spawn } from "node:child_process";
import { existsSync, statSync, accessSync, constants } from "node:fs";
@@ -40,9 +44,11 @@ export class SourceRefusal extends Error {
/**
* One denied literal. `ci` literals match ASCII case-insensitively: their
- * `bytes` are stored folded, and are searched for in a folded copy.
+ * `bytes` are stored folded, and are searched for in a folded copy. `from`
+ * says where the operator wrote it (`denylist line 3`), which is how a report
+ * names it.
*/
-export type Literal = { bytes: Buffer; ci: boolean };
+export type Literal = { bytes: Buffer; ci: boolean; from?: string };
const LOWER_A = 0x61;
const UPPER_A = 0x41;
@@ -59,22 +65,33 @@ export function foldAscii(buf: Uint8Array): Buffer {
}
/**
+ * An operator file's text as lines: a UTF-8 byte-order mark at the start is
+ * dropped and CRLF endings become LF. An editor that writes either must not
+ * turn the first rule (and the denial it implies) into one that never matches.
+ */
+export function operatorLines(text: string): string[] {
+ return text.replace(/^\uFEFF/, "").split("\n").map((l) => l.replace(/\r$/, ""));
+}
+
+/**
* The denylist file: one literal per line. `i:` in front makes it ASCII
* case-insensitive. Blank lines and lines starting with `#` are skipped;
- * surrounding whitespace is trimmed. Duplicates collapse.
+ * surrounding whitespace is trimmed. Duplicates collapse (the first line
+ * names it).
*/
export function parseDenylist(text: string): Literal[] {
const out: Literal[] = [];
- for (const raw of text.split("\n")) {
+ operatorLines(text).forEach((raw, i) => {
const line = raw.trim();
- if (line === "" || line.startsWith("#")) continue;
+ if (line === "" || line.startsWith("#")) return;
+ const from = `denylist line ${i + 1}`;
if (line.startsWith("i:")) {
const lit = line.slice(2).trim();
- if (lit) out.push({ bytes: foldAscii(Buffer.from(lit, "utf8")), ci: true });
+ if (lit) out.push({ bytes: foldAscii(Buffer.from(lit, "utf8")), ci: true, from });
} else {
- out.push({ bytes: Buffer.from(line, "utf8"), ci: false });
+ out.push({ bytes: Buffer.from(line, "utf8"), ci: false, from });
}
- }
+ });
return dedupeLiterals(out);
}
@@ -90,12 +107,14 @@ export function dedupeLiterals(list: Literal[]): Literal[] {
return out;
}
-/** How a literal is named in a report: never its bytes. */
+/**
+ * How a literal is named in a report: where it was written and its length —
+ * never any of its characters (a first letter and a length all but spell a
+ * short first name).
+ */
export function literalLabel(literals: readonly Literal[], index: number): string {
const l = literals[index];
- const first = l.bytes.subarray(0, 4).toString("utf8").slice(0, 1);
- const shown = /^[\x21-\x7e]$/.test(first) ? first : "?";
- return `#${index + 1} (${shown}…, len ${l.bytes.length}${l.ci ? ", any case" : ""})`;
+ return `${l.from ?? `literal ${index + 1}`} (len ${l.bytes.length}${l.ci ? ", any case" : ""})`;
}
// ── scanning ────────────────────────────────────────────────────────────────
@@ -121,37 +140,6 @@ export function scanBuffer(buf: Buffer, literals: readonly Literal[]): Hit[] {
return hits.sort((a, b) => a.offset - b.offset || a.lit - b.lit);
}
-const RADIUS = 24;
-
-/**
- * ±24 bytes around `hit`, printable: every byte any hit covers becomes one
- * `[REDACTED]` per run (so a second literal the window only half shows is
- * masked too), anything outside 0x20–0x7e becomes `.`.
- */
-export function redactHit(buf: Buffer, hit: Hit, allHits: readonly Hit[]): string {
- const from = Math.max(0, hit.offset - RADIUS);
- const to = Math.min(buf.length, hit.offset + hit.length + RADIUS);
- const covered = new Uint8Array(to - from);
- for (const h of allHits) {
- const a = Math.max(from, h.offset);
- const b = Math.min(to, h.offset + h.length);
- for (let i = a; i < b; i++) covered[i - from] = 1;
- }
- let out = "";
- let inRun = false;
- for (let i = from; i < to; i++) {
- if (covered[i - from]) {
- if (!inRun) out += "[REDACTED]";
- inRun = true;
- continue;
- }
- inRun = false;
- const b = buf[i];
- out += b >= 0x20 && b <= 0x7e ? String.fromCharCode(b) : ".";
- }
- return out;
-}
-
/** `text` with every occurrence of every literal replaced by `[REDACTED]`. */
export function maskLiterals(text: string, literals: readonly Literal[]): string {
const buf = Buffer.from(text, "utf8");
@@ -181,13 +169,17 @@ const KIND_ORDER: readonly HitKind[] = ["blob", "commit", "tree", "tag", "file",
export type AuditHit = {
kind: HitKind;
- // An object id (blob/commit/tree/tag), a path relative to the staged dir
- // (file) or the finding's rule (gitleaks).
+ // An object id (blob/commit/tree/tag) — with the blob's path in history
+ // once known — a path relative to the staged dir (file), or the finding
+ // (gitleaks). Paths are masked.
where: string;
// A literal's index, or -1 for a gitleaks finding.
lit: number;
- // The redacted context (only the first CONTEXTS hits carry one).
- context?: string;
+ // The byte offset of the hit in the object or file (-1 for gitleaks).
+ offset: number;
+ // Where in the object, without its bytes: a commit's or tag's header field
+ // (`author`, `committer`, `tagger`, …) or `message`; a tree's `entry N`.
+ field?: string;
};
export type AuditResult = {
@@ -199,9 +191,9 @@ export type AuditResult = {
hits: AuditHit[];
};
-// Only this many hits keep a context line: a literal every commit carries
+// Only this many hits get a line of their own: a literal every commit carries
// (the gate's own planted `Co-Authored-By`) would otherwise print thousands.
-const CONTEXTS = 20;
+const LISTED = 20;
export function emptyAudit(literals: number): AuditResult {
return { literals, objects: 0, commits: 0, files: 0, gitleaks: "not run", hits: [] };
@@ -211,20 +203,50 @@ function record(
result: AuditResult,
kind: HitKind,
where: string,
- buf: Buffer,
hits: Hit[],
+ fieldOf?: (offset: number) => string,
+ // A tree's hits are offsets into its joined entry NAMES, not the object:
+ // the entry number says where, and the offset is left out.
+ withOffset = true,
): void {
for (const h of hits) {
- const withContext = result.hits.length < CONTEXTS;
result.hits.push({
kind,
where,
lit: h.lit,
- ...(withContext ? { context: redactHit(buf, h, hits) } : {}),
+ offset: withOffset ? h.offset : -1,
+ ...(fieldOf ? { field: fieldOf(h.offset) } : {}),
});
}
}
+/**
+ * Which part of a commit or tag object `offset` falls in: `message` past the
+ * blank line that ends the headers, else the header line's keyword (`author`,
+ * `committer`, `tagger`, `tree`, `parent`, …) — a word git wrote, never the
+ * operator's bytes.
+ */
+export function objectField(data: Buffer, offset: number): string {
+ const end = data.indexOf("\n\n");
+ if (end !== -1 && offset > end) return "message";
+ let lineStart = data.lastIndexOf(0x0a, offset - 1) + 1;
+ // A continuation line (a signature, a mergetag) begins with a space: walk
+ // back to the header it continues.
+ while (lineStart > 0 && data[lineStart] === 0x20) {
+ lineStart = data.lastIndexOf(0x0a, lineStart - 2) + 1;
+ }
+ const sp = data.indexOf(0x20, lineStart);
+ const word = data.subarray(lineStart, sp === -1 ? lineStart : sp).toString("latin1");
+ return /^[a-z][a-z-]{0,15}$/.test(word) ? word : "header";
+}
+
+/** A tree hit's entry number (1-based) in the NUL-separated names buffer. */
+function treeEntry(names: Buffer, offset: number): string {
+ let n = 1;
+ for (let i = names.indexOf(0, 0); i !== -1 && i < offset; i = names.indexOf(0, i + 1)) n++;
+ return `entry ${n}`;
+}
+
// ── the object walk ─────────────────────────────────────────────────────────
/** The git environment a child must not inherit: it would point git elsewhere. */
@@ -292,12 +314,16 @@ export async function auditObjects(
if (type === "tree") {
const names = treeEntryNames(body, oid.length / 2);
const hits = scanBuffer(names, literals);
- if (hits.length) record(result, "tree", oid, names, hits);
+ if (hits.length) record(result, "tree", oid, hits, (o) => treeEntry(names, o), false);
return;
}
const hits = scanBuffer(body, literals);
- if (hits.length) {
- record(result, type === "commit" ? "commit" : type === "tag" ? "tag" : "blob", oid, body, hits);
+ if (hits.length === 0) return;
+ if (type === "commit" || type === "tag") {
+ const data = body;
+ record(result, type, oid, hits, (o) => objectField(data, o));
+ } else {
+ record(result, "blob", oid, hits);
}
};
@@ -410,9 +436,8 @@ export async function auditFiles(
const abs = path.join(dir, rel);
for (const ent of await readdir(abs, { withFileTypes: true })) {
const r = rel ? `${rel}/${ent.name}` : ent.name;
- const nameBuf = Buffer.from(r, "utf8");
- const nameHits = scanBuffer(nameBuf, literals);
- if (nameHits.length) record(result, "file", maskLiterals(r, literals), nameBuf, nameHits);
+ const nameHits = scanBuffer(Buffer.from(r, "utf8"), literals);
+ if (nameHits.length) record(result, "file", maskLiterals(r, literals), nameHits, () => "path");
if (ent.isSymbolicLink()) {
throw new SourceRefusal(`the stage holds a symlink (${maskLiterals(r, literals)}); nothing published may point outside it`);
}
@@ -425,7 +450,9 @@ export async function auditFiles(
let data = await readFile(path.join(abs, ent.name));
if (ent.name.endsWith(".gz")) data = gunzipSync(data);
const hits = scanBuffer(data, literals);
- if (hits.length) record(result, "file", maskLiterals(r, literals), data, hits);
+ if (hits.length) {
+ record(result, "file", maskLiterals(r, literals), hits, () => (ent.name.endsWith(".gz") ? "decompressed" : "contents"));
+ }
}
};
if ((await lstat(dir)).isDirectory()) await walk("");
@@ -519,6 +546,7 @@ export async function runGitleaks(
opts.literals,
),
lit: -1,
+ offset: -1,
})),
};
}
@@ -576,8 +604,9 @@ export function tildify(p: string): string {
/**
* The report, as lines. Clean: one line of counts. Hits: the counts per
- * literal and kind, the first hits with their redacted context, and what to
- * do. No line carries a literal.
+ * literal and kind, then the first hits — each by object, field and byte
+ * offset, never by its bytes — and what to do. No line carries a literal or
+ * anything read from beside one.
*/
export function formatAuditReport(
result: AuditResult,
@@ -604,13 +633,12 @@ export function formatAuditReport(
});
lines.push(`[source] ${label}: ${parts.join(", ")}`);
}
- const shown = result.hits.filter((h) => h.context !== undefined || h.kind === "gitleaks").slice(0, CONTEXTS);
+ const shown = result.hits.slice(0, LISTED);
for (const h of shown) {
- const label = h.lit === -1 ? "" : ` ${literalLabel(literals, h.lit)}`;
- lines.push(
- `[source] ${h.kind} ${h.kind === "file" || h.kind === "gitleaks" ? h.where : shortWhere(h.where)}${label}` +
- (h.context !== undefined ? `: ${h.context}` : ""),
- );
+ const where = h.kind === "file" || h.kind === "gitleaks" ? h.where : shortWhere(h.where);
+ const at = [h.field, h.offset >= 0 ? `byte ${h.offset}` : ""].filter(Boolean).join(", ");
+ const label = h.lit === -1 ? "" : `: ${literalLabel(literals, h.lit)}`;
+ lines.push(`[source] ${h.kind} ${where}${at ? ` (${at})` : ""}${label}`);
}
if (result.hits.length > shown.length) {
lines.push(`[source] … and ${result.hits.length - shown.length} more`);
diff --git a/common/publish/sourceTree.test.ts b/common/publish/sourceTree.test.ts
@@ -62,7 +62,7 @@ test("breadcrumbs hop up relatively; the root has no `..` row", () => {
assert.match(deep, /<tr><td class="name"><a href="\.\.\/">\.\.<\/a>/);
});
-test("writeTreeIndexes pages every directory and counts the tree, refusing a tracked index.html or a symlink", async () => {
+test("writeTreeIndexes pages every directory and counts the tree, refusing a tracked index.html, a 404.html or a symlink", async () => {
const tree = path.join(TMP, "t1");
mkdirSync(path.join(tree, "app", "[slug]"), { recursive: true });
writeFileSync(path.join(tree, "README.md"), "hello\n");
@@ -79,6 +79,16 @@ test("writeTreeIndexes pages every directory and counts the tree, refusing a tra
writeFileSync(path.join(withIndex, "docs", "index.html"), "<p>tracked</p>");
await assert.rejects(writeTreeIndexes(withIndex, META), (e) => e instanceof SourceRefusal && /docs\/index\.html/.test(e.message));
+ // A tracked 404.html anywhere: Pages would serve it, as HTML on this
+ // origin, for every missing path below its directory.
+ const with404 = path.join(TMP, "t4");
+ mkdirSync(path.join(with404, "docs", "deep"), { recursive: true });
+ writeFileSync(path.join(with404, "docs", "deep", "404.html"), "<script>x</script>");
+ await assert.rejects(
+ writeTreeIndexes(with404, META),
+ (e) => e instanceof SourceRefusal && /docs\/deep\/404\.html; Pages would serve it, as HTML/.test(e.message),
+ );
+
const withLink = path.join(TMP, "t3");
mkdirSync(withLink);
symlinkSync("/etc/hostname", path.join(withLink, "leak"));
diff --git a/common/publish/sourceTree.ts b/common/publish/sourceTree.ts
@@ -134,8 +134,9 @@ export function renderTreeIndex(opts: {
/**
* Write an index.html into every directory under `treeDir`, the root
* included. REFUSES when a directory already holds an `index.html` (a tracked
- * one would be overwritten, or served in the index's place) or when anything
- * is a symlink. Returns the tree's own files and bytes (not the pages) and how
+ * one would be overwritten, or served in the index's place) or a `404.html`
+ * (Pages would serve it as HTML for any missing path below it), or when
+ * anything is a symlink. Returns the tree's own files and bytes (not the pages) and how
* many directories got a page.
*/
export async function writeTreeIndexes(
@@ -161,6 +162,13 @@ export async function writeTreeIndexes(
`the tree already has ${r}; its directory page would replace it — rename the file or publish without the raw tree`,
);
}
+ // Pages answers a missing path with the NEAREST 404.html, as HTML (the
+ // directory-page rule): a tracked one would run on this origin.
+ if (ent.name === "404.html") {
+ throw new SourceRefusal(
+ `the tree has ${r}; Pages would serve it, as HTML on this origin, for every missing path below it — rename the file or publish without the raw tree`,
+ );
+ }
const { size } = await stat(path.join(abs, ent.name));
entries.push({ name: ent.name, dir: false, bytes: size });
totals.files++;