commit 1bbe569d4816f82630fd9876033a071291143cb5
parent 5a2b1667e8d3ba644ef5b166f0d4208dd56a6f1c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 16:21:42 -0400
report-to-video: shoot-page — a highlighted sentence from a saved page, as a PNG
Loads a saved HTML file offline (only file:// under the page's own
directory), finds each quote by normalised text across element
boundaries, wraps it in marks and screenshots the containing block at
deviceScaleFactor 2. Batch mode writes <id>.png and results.json; a miss
is listed and exits 1. The matcher is pure and tested; the browser test
runs with SHOOT_PAGE_BROWSER=1.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 772 insertions(+), 0 deletions(-)
diff --git a/umtool/report-to-video/shoot-page.mjs b/umtool/report-to-video/shoot-page.mjs
@@ -0,0 +1,535 @@
+#!/usr/bin/env node
+// shoot-page.mjs — a highlighted sentence from a SAVED web page, as a PNG.
+//
+// An article a report cites is evidence the way a clip is, and an `image` entry
+// is how a still gets into the cut. This makes that still: load the page as it
+// was saved to disk, find the quoted sentence in its text, paint it like a
+// highlighter, and screenshot the paragraph that holds it.
+//
+// The page is loaded OFFLINE. Every request is refused except a file:// one
+// under the page's own directory (the `<name>_files/` folder a browser's "save
+// page, complete" writes), so a shot never reaches the network and never reads
+// a file the page was not saved with. Page scripts are off by default: a saved
+// page is already rendered, and its scripts, with nothing to talk to, are more
+// likely to hide the article than to finish drawing it (`--js` turns them on).
+//
+// The quote is found by TEXT, not by selector, and loosely in exactly the ways
+// copying text off a page is loose: curly and straight quotes and apostrophes
+// are the same character, NBSP and every other space are a space, runs of
+// whitespace are one, and soft hyphens and zero-width characters are not there
+// at all. A match runs across element boundaries — a link, an `<em>`, a `<br>`
+// — and the highlight is one `<mark>` per text node it covers. A quote that is
+// not on the page is a MISS: listed in the results, summarised on stderr, and
+// the exit status is 1. It is never skipped.
+//
+// On the CLI:
+// node umtool/report-to-video/shoot-page.mjs --page <file.html> --quote "<text>" --out <shot.png>
+// node umtool/report-to-video/shoot-page.mjs --batch <items.json> --out <dir>
+//
+// Batch items are `[{ id, page, quote, context? }]` (or `{ items: [...] }`),
+// `page` relative to the items file. Each writes `<dir>/<id>.png`, and the run
+// writes `<dir>/results.json`. `context` is a longer stretch of text around the
+// quote, for a quote that occurs more than once.
+//
+// Options:
+// --context <text> (single) as an item's `context`
+// --color <css> The highlight colour (default: #ffe14d)
+// --padding <px> CSS px around the block (default: 24)
+// --width <px> Viewport width in CSS px (default: 1280)
+// --scale <n> deviceScaleFactor (default: 2)
+// --max-height <px> A taller block is trimmed to this, centred on the
+// quote (default: 1200)
+// --js Run the page's own scripts
+// --timeout <ms> Page load timeout (default: 20000)
+
+import { existsSync, statSync } from "node:fs";
+import { mkdir, readFile, writeFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath, pathToFileURL } from "node:url";
+
+export const DEFAULTS = Object.freeze({
+ color: "#ffe14d",
+ padding: 24,
+ width: 1280,
+ height: 900,
+ scale: 2,
+ maxHeight: 1200,
+ js: false,
+ timeout: 20_000,
+});
+
+// ---------------------------------------------------------------------------
+// The matcher. Pure, and SELF-CONTAINED: these two functions are also sent into
+// the page as source text (see browserShoot), so neither may reference anything
+// outside itself but the other by name and the language's own globals.
+// ---------------------------------------------------------------------------
+
+// `raw` normalised for matching, with `map[k]` = the index in `raw` of the
+// normalised text's k-th character. Several normalised characters may map to
+// one raw one (an ellipsis is three dots); a raw character may map to none.
+export function normaliseWithMap(raw) {
+ const out = [];
+ const map = [];
+ for (let i = 0; i < raw.length; i++) {
+ const c = raw[i];
+ // Soft hyphen, zero-width space / non-joiner / joiner, word joiner, BOM:
+ // present in the DOM, invisible on the page, never in a copied quote.
+ if (/[\u00AD\u200B-\u200D\u2060\uFEFF]/.test(c)) continue;
+ // \s covers NBSP, the U+2000 block, narrow NBSP and the ideographic space.
+ if (/\s/.test(c)) {
+ if (out.length === 0 || out[out.length - 1] === " ") continue;
+ out.push(" ");
+ map.push(i);
+ continue;
+ }
+ let n = c;
+ if (/[\u2018\u2019\u201A\u201B\u2032\u02BC]/.test(c)) n = "'";
+ else if (/[\u201C\u201D\u201E\u201F\u2033]/.test(c)) n = '"';
+ else if (c === "\u2026") n = "...";
+ for (const ch of n) {
+ out.push(ch);
+ map.push(i);
+ }
+ }
+ return { text: out.join(""), map };
+}
+
+// Find `quote` in the text of `segments` (strings, in document order, joined
+// with nothing between them — a caller puts a " " segment where a block or a
+// `<br>` breaks the text). With `context`, the quote must lie inside the first
+// occurrence of the context.
+//
+// Returns `{ ok: true, start, end, occurrences }`, `start`/`end` being
+// `{ segment, offset }` in the RAW segment strings (`end` exclusive, and always
+// in the segment holding the match's last character), or `{ ok: false, reason }`.
+export function matchInSegments(segments, quote, context) {
+ const q = normaliseWithMap(String(quote ?? "")).text.trim();
+ if (!q) return { ok: false, reason: "empty quote" };
+ const raw = segments.join("");
+ const hay = normaliseWithMap(raw);
+
+ let occurrences = 0;
+ for (let i = hay.text.indexOf(q); i !== -1; i = hay.text.indexOf(q, i + 1)) occurrences++;
+
+ let at;
+ if (context != null && String(context).trim()) {
+ const c = normaliseWithMap(String(context)).text.trim();
+ const ci = hay.text.indexOf(c);
+ if (ci === -1) return { ok: false, reason: "context not found", occurrences };
+ at = hay.text.indexOf(q, ci);
+ if (at === -1 || at + q.length > ci + c.length) {
+ return { ok: false, reason: "quote not found inside its context", occurrences };
+ }
+ } else {
+ at = hay.text.indexOf(q);
+ if (at === -1) return { ok: false, reason: "quote not found", occurrences };
+ }
+
+ const rawStart = hay.map[at];
+ const rawLast = hay.map[at + q.length - 1];
+ const locate = (idx) => {
+ let base = 0;
+ for (let s = 0; s < segments.length; s++) {
+ const len = segments[s].length;
+ if (idx < base + len) return { segment: s, offset: idx - base };
+ base += len;
+ }
+ return null;
+ };
+ const start = locate(rawStart);
+ const last = locate(rawLast);
+ return { ok: true, start, end: { segment: last.segment, offset: last.offset + 1 }, occurrences };
+}
+
+// ---------------------------------------------------------------------------
+// The page side. Runs INSIDE the browser (serialised by `pageExpression`): walk
+// the visible text, match, wrap the match in marks, measure what to shoot.
+// ---------------------------------------------------------------------------
+
+function browserShoot(args) {
+ const { quote, context, color, padding, maxHeight } = args;
+ const doc = document;
+ const SKIP = new Set(["SCRIPT", "STYLE", "NOSCRIPT", "TEMPLATE", "TITLE", "HEAD", "SVG", "MATH", "IFRAME", "OBJECT", "SELECT", "TEXTAREA"]);
+ const blockCache = new Map();
+ const isBlock = (el) => {
+ const d = getComputedStyle(el).display;
+ return !(d.startsWith("inline") || d === "contents" || d === "ruby" || d === "none");
+ };
+ const blockOf = (node) => {
+ let el = node.nodeType === 1 ? node : node.parentElement;
+ const seen = [];
+ while (el && el !== doc.body && el !== doc.documentElement) {
+ if (blockCache.has(el)) {
+ const b = blockCache.get(el);
+ for (const s of seen) blockCache.set(s, b);
+ return b;
+ }
+ seen.push(el);
+ if (isBlock(el)) break;
+ el = el.parentElement;
+ }
+ const b = el ?? doc.body;
+ for (const s of seen) blockCache.set(s, b);
+ return b;
+ };
+
+ const segs = [];
+ let lastBlock = null;
+ const walker = doc.createTreeWalker(doc.body ?? doc.documentElement, NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT, {
+ acceptNode(n) {
+ if (n.nodeType === 3) return NodeFilter.FILTER_ACCEPT;
+ if (SKIP.has(n.tagName.toUpperCase())) return NodeFilter.FILTER_REJECT;
+ const st = getComputedStyle(n);
+ if (st.display === "none" || st.visibility === "hidden") return NodeFilter.FILTER_REJECT;
+ return n.tagName === "BR" ? NodeFilter.FILTER_ACCEPT : NodeFilter.FILTER_SKIP;
+ },
+ });
+ for (let n = walker.nextNode(); n; n = walker.nextNode()) {
+ if (n.nodeType === 1) {
+ segs.push({ node: null, text: " " });
+ continue;
+ }
+ const b = blockOf(n);
+ if (lastBlock && b !== lastBlock) segs.push({ node: null, text: " " });
+ lastBlock = b;
+ segs.push({ node: n, text: n.data });
+ }
+
+ const m = matchInSegments(segs.map((s) => s.text), quote, context);
+ if (!m.ok) return m;
+
+ const range = doc.createRange();
+ range.setStart(segs[m.start.segment].node, m.start.offset);
+ range.setEnd(segs[m.end.segment].node, m.end.offset);
+ const matched = range.toString();
+ let common = range.commonAncestorContainer;
+ if (common.nodeType !== 1) common = common.parentElement;
+ const block = common === doc.body || isBlock(common) ? common : blockOf(common);
+
+ // One mark per text node the match covers. splitText at the end first, so
+ // the node in hand stays the left part; then at the start, which returns the
+ // covered middle.
+ const marks = [];
+ for (let i = m.start.segment; i <= m.end.segment; i++) {
+ let node = segs[i].node;
+ if (!node) continue;
+ const a = i === m.start.segment ? m.start.offset : 0;
+ const b = i === m.end.segment ? m.end.offset : node.data.length;
+ if (a >= b) continue;
+ if (b < node.data.length) node.splitText(b);
+ if (a > 0) node = node.splitText(a);
+ const mark = doc.createElement("mark");
+ mark.setAttribute("data-shoot-page", "");
+ // box-shadow, not padding: the highlight must not reflow the paragraph.
+ mark.style.cssText =
+ `background:${color} !important;color:inherit !important;` +
+ `box-shadow:0 0 0 0.12em ${color};border-radius:0.12em;` +
+ "-webkit-box-decoration-break:clone;box-decoration-break:clone;";
+ node.parentNode.insertBefore(mark, node);
+ mark.appendChild(node);
+ marks.push(mark);
+ }
+
+ const sx = window.scrollX;
+ const sy = window.scrollY;
+ const de = doc.documentElement;
+ const docW = Math.max(de.scrollWidth, doc.body?.scrollWidth ?? 0);
+ const docH = Math.max(de.scrollHeight, doc.body?.scrollHeight ?? 0);
+ const br = block.getBoundingClientRect();
+ let top = br.top + sy;
+ let bottom = br.bottom + sy;
+ let trimmed = false;
+ if (bottom - top > maxHeight) {
+ let mt = Infinity;
+ let mb = -Infinity;
+ for (const mk of marks) {
+ for (const r of mk.getClientRects()) {
+ mt = Math.min(mt, r.top + sy);
+ mb = Math.max(mb, r.bottom + sy);
+ }
+ }
+ if (mb - mt >= maxHeight) {
+ top = mt;
+ bottom = mb;
+ } else {
+ const mid = (mt + mb) / 2;
+ const t = Math.min(Math.max(top, mid - maxHeight / 2), bottom - maxHeight);
+ top = t;
+ bottom = t + maxHeight;
+ }
+ trimmed = true;
+ }
+ const x0 = Math.max(0, Math.floor(br.left + sx - padding));
+ const y0 = Math.max(0, Math.floor(top - padding));
+ const x1 = Math.min(docW, Math.ceil(br.right + sx + padding));
+ const y1 = Math.min(docH, Math.ceil(bottom + padding));
+
+ const cssPath = (el) => {
+ const parts = [];
+ while (el && el.nodeType === 1 && el !== de) {
+ if (el.id && doc.querySelectorAll(`#${CSS.escape(el.id)}`).length === 1) {
+ parts.unshift(`#${CSS.escape(el.id)}`);
+ return parts.join(" > ");
+ }
+ const tag = el.tagName.toLowerCase();
+ const same = el.parentElement
+ ? [...el.parentElement.children].filter((c) => c.tagName === el.tagName)
+ : [];
+ parts.unshift(same.length > 1 ? `${tag}:nth-of-type(${same.indexOf(el) + 1})` : tag);
+ el = el.parentElement;
+ }
+ return ["html", ...parts].join(" > ");
+ };
+
+ return {
+ ok: true,
+ matched,
+ block: cssPath(block),
+ crop: { x: x0, y: y0, width: x1 - x0, height: y1 - y0 },
+ trimmed,
+ occurrences: m.occurrences,
+ marks: marks.length,
+ };
+}
+
+// The expression handed to page.evaluate: the matcher and the walker as source,
+// then a call. A string rather than a function so the helpers travel with it.
+export function pageExpression(args) {
+ return `(() => {\n${normaliseWithMap}\n${matchInSegments}\n${browserShoot}\nreturn browserShoot(${JSON.stringify(args)});\n})()`;
+}
+
+// ---------------------------------------------------------------------------
+// The browser.
+// ---------------------------------------------------------------------------
+
+// Is `url` a file the page at `pagePath` may load: file:// and under its dir.
+export function allowedRequest(url, pagePath) {
+ if (!url.startsWith("file:")) return false;
+ let p;
+ try {
+ p = fileURLToPath(url);
+ } catch {
+ return false;
+ }
+ const rel = path.relative(path.dirname(path.resolve(pagePath)), path.resolve(p));
+ return rel === "" || (!rel.startsWith("..") && !path.isAbsolute(rel));
+}
+
+async function loadChromium() {
+ // umtool installs @playwright/test; this package does not declare it, so it
+ // resolves from umtool's node_modules. It is CommonJS: accept either shape.
+ let mod;
+ try {
+ mod = await import("@playwright/test");
+ } catch (err) {
+ throw new Error(`Playwright is not available here (it ships with umtool): ${err.message}`);
+ }
+ const chromium = mod.chromium ?? mod.default?.chromium;
+ if (!chromium) throw new Error("Playwright loaded but has no chromium export");
+ return chromium;
+}
+
+// Shoot every item; one browser for the lot, a fresh page per item (the marks
+// mutate the DOM). Returns the results array; writes nothing but the PNGs.
+//
+// `items`: [{ id, page (absolute or relative to `baseDir`), quote, context?, out? }]
+export async function shootPages(items, opts = {}) {
+ const o = { ...DEFAULTS, ...opts };
+ const chromium = await loadChromium();
+ const browser = await chromium.launch({ headless: true });
+ const results = [];
+ try {
+ const ctx = await browser.newContext({
+ viewport: { width: o.width, height: o.height },
+ deviceScaleFactor: o.scale,
+ javaScriptEnabled: o.js,
+ bypassCSP: true,
+ serviceWorkers: "block",
+ offline: true,
+ });
+ for (const item of items) results.push(await shootOne(ctx, item, o));
+ await ctx.close();
+ } finally {
+ await browser.close();
+ }
+ return results;
+}
+
+async function shootOne(ctx, item, o) {
+ const pagePath = path.resolve(o.baseDir ?? process.cwd(), String(item.page ?? ""));
+ const base = { id: item.id, page: item.page, quote: item.quote };
+ if (!item.page || !existsSync(pagePath) || !statSync(pagePath).isFile()) {
+ return { ...base, ok: false, reason: "page not found" };
+ }
+ const blocked = [];
+ const page = await ctx.newPage();
+ try {
+ await page.route(() => true, (route) => {
+ const url = route.request().url();
+ if (allowedRequest(url, pagePath)) return route.continue();
+ blocked.push(url.length > 200 ? `${url.slice(0, 200)}\u2026` : url);
+ return route.abort("blockedbyclient");
+ });
+ await page.goto(pathToFileURL(pagePath).href, { waitUntil: "load", timeout: o.timeout });
+ // Fonts are not part of "load"; a shot taken before them reflows after.
+ await page.evaluate("document.fonts ? document.fonts.ready.then(() => true) : true");
+ const r = await page.evaluate(
+ pageExpression({
+ quote: item.quote,
+ context: item.context ?? null,
+ color: o.color,
+ padding: o.padding,
+ maxHeight: o.maxHeight,
+ }),
+ );
+ if (!r.ok) return { ...base, ok: false, reason: r.reason, occurrences: r.occurrences, blocked };
+ await mkdir(path.dirname(item.out), { recursive: true });
+ await page.screenshot({ path: item.out, type: "png", fullPage: true, clip: r.crop });
+ return {
+ ...base,
+ ok: true,
+ png: item.out,
+ matched: r.matched,
+ block: r.block,
+ crop: r.crop,
+ scale: o.scale,
+ pixels: { width: Math.round(r.crop.width * o.scale), height: Math.round(r.crop.height * o.scale) },
+ trimmed: r.trimmed,
+ occurrences: r.occurrences,
+ blocked,
+ };
+ } catch (err) {
+ return { ...base, ok: false, reason: `error: ${err.message?.split("\n")[0] ?? err}`, blocked };
+ } finally {
+ await page.close();
+ }
+}
+
+// ---------------------------------------------------------------------------
+// The CLI.
+// ---------------------------------------------------------------------------
+
+// Items from a batch file: an array, or `{ items: [...] }`. Every item needs an
+// id that is a safe file name, a page and a quote; ids are unique. Throws with
+// every problem at once rather than the first.
+export function validateItems(data) {
+ const items = Array.isArray(data) ? data : data?.items;
+ if (!Array.isArray(items)) throw new Error("batch file must be an array of items or { items: [...] }");
+ const problems = [];
+ const seen = new Set();
+ items.forEach((it, i) => {
+ const where = `item ${i}${it?.id ? ` (${it.id})` : ""}`;
+ if (!it || typeof it !== "object") return problems.push(`${where}: not an object`);
+ if (typeof it.id !== "string" || !/^[A-Za-z0-9._-]+$/.test(it.id) || it.id.startsWith(".")) {
+ problems.push(`${where}: id must be a file-name-safe string`);
+ } else if (seen.has(it.id)) problems.push(`${where}: duplicate id`);
+ else seen.add(it.id);
+ if (typeof it.page !== "string" || !it.page) problems.push(`${where}: page is required`);
+ if (typeof it.quote !== "string" || !it.quote.trim()) problems.push(`${where}: quote is required`);
+ if (it.context != null && typeof it.context !== "string") problems.push(`${where}: context must be a string`);
+ });
+ if (problems.length) throw new Error(problems.join("\n"));
+ return items;
+}
+
+export function parseArgs(argv) {
+ const opts = {};
+ const out = { mode: null, opts };
+ const num = (name, v) => {
+ const n = Number(v);
+ if (!Number.isFinite(n) || n < 0) throw new Error(`${name} needs a non-negative number`);
+ return n;
+ };
+ for (let i = 0; i < argv.length; i++) {
+ const a = argv[i];
+ const next = () => {
+ if (i + 1 >= argv.length) throw new Error(`${a} needs a value`);
+ return argv[++i];
+ };
+ switch (a) {
+ case "--page": out.page = next(); break;
+ case "--quote": out.quote = next(); break;
+ case "--context": out.context = next(); break;
+ case "--batch": out.batch = next(); break;
+ case "--out": out.out = next(); break;
+ case "--color": opts.color = next(); break;
+ case "--padding": opts.padding = num(a, next()); break;
+ case "--width": opts.width = num(a, next()); break;
+ case "--scale": opts.scale = num(a, next()); break;
+ case "--max-height": opts.maxHeight = num(a, next()); break;
+ case "--timeout": opts.timeout = num(a, next()); break;
+ case "--js": opts.js = true; break;
+ default: throw new Error(`unknown argument: ${a}`);
+ }
+ }
+ if (out.batch && (out.page || out.quote)) throw new Error("--batch and --page/--quote are two modes; pick one");
+ if (!out.out) throw new Error("--out is required");
+ if (out.batch) out.mode = "batch";
+ else if (out.page && out.quote) out.mode = "single";
+ else throw new Error("give --batch <items.json>, or --page and --quote");
+ return out;
+}
+
+const USAGE =
+ "usage: shoot-page.mjs --page <file.html> --quote <text> [--context <text>] --out <shot.png>\n" +
+ " shoot-page.mjs --batch <items.json> --out <dir>\n" +
+ " [--color <css>] [--padding <px>] [--width <px>] [--scale <n>] [--max-height <px>] [--js] [--timeout <ms>]";
+
+async function main() {
+ let args;
+ try {
+ args = parseArgs(process.argv.slice(2));
+ } catch (err) {
+ console.error(`${err.message}\n${USAGE}`);
+ process.exit(2);
+ }
+
+ let items;
+ let resultsFile = null;
+ if (args.mode === "batch") {
+ const batchPath = path.resolve(args.batch);
+ items = validateItems(JSON.parse(await readFile(batchPath, "utf8")));
+ const outDir = path.resolve(args.out);
+ items = items.map((it) => ({ ...it, out: path.join(outDir, `${it.id}.png`) }));
+ args.opts.baseDir = path.dirname(batchPath);
+ resultsFile = path.join(outDir, "results.json");
+ } else {
+ items = [{ id: "shot", page: args.page, quote: args.quote, context: args.context, out: path.resolve(args.out) }];
+ }
+
+ const results = await shootPages(items, args.opts);
+ const missed = results.filter((r) => !r.ok);
+ const report = {
+ generatedAt: new Date().toISOString(),
+ options: { ...DEFAULTS, ...args.opts },
+ shot: results.length - missed.length,
+ missed: missed.length,
+ items: results,
+ };
+ if (resultsFile) {
+ await mkdir(path.dirname(resultsFile), { recursive: true });
+ await writeFile(resultsFile, JSON.stringify(report, null, 2) + "\n", "utf8");
+ } else {
+ console.log(JSON.stringify(results[0], null, 2));
+ }
+
+ for (const r of results) {
+ if (r.ok && r.occurrences > 1) {
+ console.error(`note: ${r.id}: the quote occurs ${r.occurrences} times; shot the first (give a context to pick another)`);
+ }
+ }
+ if (missed.length) {
+ console.error(`MISSED ${missed.length} of ${results.length}:`);
+ for (const r of missed) console.error(` ${r.id}: ${r.reason} — ${r.page}`);
+ if (resultsFile) console.error(`results: ${resultsFile}`);
+ process.exit(1);
+ }
+ if (resultsFile) console.error(`shot ${results.length} -> ${resultsFile}`);
+}
+
+if (import.meta.url === `file://${process.argv[1]}`) {
+ main().catch((err) => {
+ console.error(err.message ?? err);
+ process.exit(1);
+ });
+}
diff --git a/umtool/report-to-video/shoot-page.test.mjs b/umtool/report-to-video/shoot-page.test.mjs
@@ -0,0 +1,237 @@
+// Tests for shoot-page.mjs — finding a quoted sentence in a saved page's text.
+//
+// The matcher is pure and is tested against text segments shaped like the
+// page's text nodes in document order. One test drives a real Chromium over a
+// fixture page; it launches a browser, so it runs only with SHOOT_PAGE_BROWSER=1.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import { mkdtemp, readFile, rm, writeFile, mkdir } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import test from "node:test";
+
+import {
+ allowedRequest, matchInSegments, normaliseWithMap, pageExpression, parseArgs, shootPages, validateItems,
+} from "./shoot-page.mjs";
+
+// The raw text a match covers, read back off the segments.
+function covered(segments, m) {
+ const out = [];
+ for (let s = m.start.segment; s <= m.end.segment; s++) {
+ const a = s === m.start.segment ? m.start.offset : 0;
+ const b = s === m.end.segment ? m.end.offset : segments[s].length;
+ out.push(segments[s].slice(a, b));
+ }
+ return out.join("");
+}
+
+test("normalising: quotes, spaces, invisibles, ellipsis — with a map back to the raw text", () => {
+ const raw = "\u201CIt\u2019s\u00A0fine\u201D, she\u00ADsaid\u200B\u2026";
+ const { text, map } = normaliseWithMap(raw);
+ assert.equal(text, "\"It's fine\", shesaid...");
+ assert.equal(map.length, text.length);
+ // Every normalised character points at the raw one it came from.
+ assert.equal(raw[map[text.indexOf("s", 4)]], "s");
+ assert.equal(map[text.length - 1], raw.length - 1);
+ assert.equal(map[text.length - 3], raw.length - 1);
+ // Leading whitespace goes; a run of it is one space.
+ assert.equal(normaliseWithMap(" \n\t a \n b").text, "a b");
+});
+
+test("a plain quote inside one text node", () => {
+ const segs = ["The committee said the plan was not funded and would be dropped."];
+ const m = matchInSegments(segs, "the plan was not funded");
+ assert.equal(m.ok, true);
+ assert.deepEqual(m.start, { segment: 0, offset: 19 });
+ assert.equal(covered(segs, m), "the plan was not funded");
+ assert.equal(m.occurrences, 1);
+});
+
+test("a quote split across <em> and <a> matches across the boundaries", () => {
+ // <p>She wrote that it was <em>never</em> about <a href>the money</a>, at all.</p>
+ const segs = ["She wrote that it was ", "never", " about ", "the money", ", at all."];
+ const m = matchInSegments(segs, "it was never about the money");
+ assert.equal(m.ok, true);
+ assert.deepEqual(m.start, { segment: 0, offset: 15 });
+ // The end lies in the link's text node, not at the start of the next one.
+ assert.deepEqual(m.end, { segment: 3, offset: 9 });
+ assert.equal(covered(segs, m), "it was never about the money");
+});
+
+test("curly on the page, straight in the quote — and the other way round", () => {
+ const curly = ["He said \u201Cwe\u2019re not going\u201D and left."];
+ const m1 = matchInSegments(curly, "\"we're not going\"");
+ assert.equal(m1.ok, true);
+ assert.equal(covered(curly, m1), "\u201Cwe\u2019re not going\u201D");
+
+ const straight = ["He said \"we're not going\" and left."];
+ const m2 = matchInSegments(straight, "\u201Cwe\u2019re not going\u201D");
+ assert.equal(m2.ok, true);
+ assert.equal(covered(straight, m2), "\"we're not going\"");
+});
+
+test("NBSP, soft hyphens, zero-width characters and source line breaks do not stop a match", () => {
+ const segs = ["an un\u00ADprecedented\u00A0and\n ", "unre\u200Bcoverable", " loss"];
+ const m = matchInSegments(segs, "an unprecedented and unrecoverable loss");
+ assert.equal(m.ok, true);
+ assert.equal(covered(segs, m), segs.join(""));
+});
+
+test("a block break between two text nodes reads as a space, never as nothing", () => {
+ // The caller puts a " " segment where one block ends and another starts.
+ const segs = ["first paragraph.", " ", "Second paragraph."];
+ assert.equal(matchInSegments(segs, "paragraph. Second").ok, true);
+ assert.equal(matchInSegments(segs, "paragraph.Second").ok, false);
+});
+
+test("a quote that is not there is a miss with a reason", () => {
+ const m = matchInSegments(["Nothing to see here."], "something else entirely");
+ assert.deepEqual(m, { ok: false, reason: "quote not found", occurrences: 0 });
+ assert.equal(matchInSegments(["text"], " ").reason, "empty quote");
+});
+
+test("context picks the occurrence, and is itself required to be there", () => {
+ const segs = ["It was late. ", "Later, it was late again, said the second witness."];
+ const plain = matchInSegments(segs, "it was late");
+ // Case matters: only the second sentence's lowercase "it was late" matches.
+ assert.equal(plain.ok, true);
+ assert.equal(plain.start.segment, 1);
+
+ const two = ["It was late. It was late again."];
+ const first = matchInSegments(two, "It was late");
+ assert.equal(first.occurrences, 2);
+ assert.equal(first.start.offset, 0);
+ const second = matchInSegments(two, "It was late", "It was late again");
+ assert.equal(second.ok, true);
+ assert.equal(second.start.offset, 13);
+
+ assert.equal(matchInSegments(two, "It was late", "not on the page").reason, "context not found");
+ assert.equal(
+ matchInSegments(two, "late again", "It was late. It").reason,
+ "quote not found inside its context",
+ );
+});
+
+test("the page expression carries the matcher with it", () => {
+ const expr = pageExpression({ quote: "a \"b\"", context: null, color: "#ff0", padding: 4, maxHeight: 100 });
+ assert.match(expr, /function normaliseWithMap/);
+ assert.match(expr, /function matchInSegments/);
+ assert.match(expr, /return browserShoot\(\{"quote":"a \\"b\\""/);
+ // It parses as one expression.
+ assert.doesNotThrow(() => new Function(`return ${expr.replace(/return browserShoot\([^\n]*\);/, "return 0;")}`));
+});
+
+test("only file:// under the page's own directory is allowed", () => {
+ const page = "/tmp/saved/article.html";
+ assert.equal(allowedRequest("file:///tmp/saved/article.html", page), true);
+ assert.equal(allowedRequest("file:///tmp/saved/article_files/style.css", page), true);
+ assert.equal(allowedRequest("file:///tmp/saved/../other/x.css", page), false);
+ assert.equal(allowedRequest("file:///tmp/savedx/x.css", page), false);
+ assert.equal(allowedRequest("file:///etc/passwd", page), false);
+ assert.equal(allowedRequest("https://example.com/x.png", page), false);
+ assert.equal(allowedRequest("data:image/png;base64,AAAA", page), false);
+});
+
+test("batch items: every problem at once", () => {
+ assert.deepEqual(validateItems({ items: [{ id: "a1", page: "p.html", quote: "q" }] }).length, 1);
+ assert.throws(
+ () => validateItems([
+ { id: "a1", page: "p.html", quote: "q" },
+ { id: "a1", page: "p.html", quote: "q" },
+ { id: "../x", page: "", quote: " " },
+ ]),
+ (err) => /duplicate id/.test(err.message) && /file-name-safe/.test(err.message)
+ && /page is required/.test(err.message) && /quote is required/.test(err.message),
+ );
+ assert.throws(() => validateItems({}), /array of items/);
+});
+
+test("arguments: two modes, never both", () => {
+ assert.equal(parseArgs(["--batch", "i.json", "--out", "d"]).mode, "batch");
+ const s = parseArgs(["--page", "p.html", "--quote", "q", "--out", "s.png", "--scale", "3", "--js"]);
+ assert.equal(s.mode, "single");
+ assert.deepEqual(s.opts, { scale: 3, js: true });
+ assert.throws(() => parseArgs(["--batch", "i.json", "--page", "p", "--out", "d"]), /two modes/);
+ assert.throws(() => parseArgs(["--page", "p.html", "--out", "s.png"]), /--batch/);
+ assert.throws(() => parseArgs(["--page", "p.html", "--quote", "q"]), /--out is required/);
+ assert.throws(() => parseArgs(["--padding", "-1"]), /non-negative/);
+ assert.throws(() => parseArgs(["--bogus"]), /unknown argument/);
+});
+
+// ---------------------------------------------------------------------------
+// A real browser over a fixture page. Heavy, so opt-in.
+// ---------------------------------------------------------------------------
+
+const FIXTURE = `<!doctype html>
+<html><head><meta charset="utf-8">
+<link rel="stylesheet" href="article_files/style.css">
+<link rel="stylesheet" href="https://fonts.example.invalid/remote.css">
+<style>body { margin: 0; padding: 40px; font: 18px/1.5 serif; width: 700px; }</style>
+</head><body>
+<header style="position: sticky; top: 0">Header</header>
+<p id="plain">The council voted on Tuesday. The plan was not funded and would be dropped.</p>
+<p>She wrote that it was <em>never</em> about <a href="https://example.invalid/">the money</a>, at all.</p>
+<p>He said \u201Cwe\u2019re not going\u201D and left.</p>
+<div style="display:none">a hidden sentence that must not match</div>
+<img src="https://example.invalid/tracker.png" alt="">
+<img src="../outside.png" alt="">
+</body></html>`;
+
+function pngSize(buf) {
+ // IHDR: width and height are the big-endian words at bytes 16 and 20.
+ assert.equal(buf.toString("ascii", 1, 4), "PNG");
+ return { width: buf.readUInt32BE(16), height: buf.readUInt32BE(20) };
+}
+
+test("a real page: shots, marks, a miss, and nothing fetched from outside", {
+ skip: process.env.SHOOT_PAGE_BROWSER === "1" ? false : "set SHOOT_PAGE_BROWSER=1 to launch Chromium",
+}, async () => {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "shoot-page-"));
+ try {
+ const saved = path.join(dir, "saved");
+ await mkdir(path.join(saved, "article_files"), { recursive: true });
+ await writeFile(path.join(saved, "article.html"), FIXTURE);
+ await writeFile(path.join(saved, "article_files", "style.css"), "p { color: #222; }");
+ await writeFile(path.join(dir, "outside.png"), "");
+ const out = path.join(dir, "out");
+ const items = [
+ { id: "plain", page: "saved/article.html", quote: "The plan was not funded" },
+ { id: "split", page: "saved/article.html", quote: "it was never about the money" },
+ { id: "curly", page: "saved/article.html", quote: "\"we're not going\"" },
+ { id: "hidden", page: "saved/article.html", quote: "a hidden sentence" },
+ { id: "nopage", page: "saved/missing.html", quote: "anything" },
+ ].map((it) => ({ ...it, out: path.join(out, `${it.id}.png`) }));
+
+ const results = await shootPages(items, { baseDir: dir, padding: 10 });
+ const by = Object.fromEntries(results.map((r) => [r.id, r]));
+
+ assert.equal(by.plain.ok, true);
+ assert.equal(by.plain.matched, "The plan was not funded");
+ assert.equal(by.plain.block, "#plain");
+ assert.equal(by.split.ok, true);
+ assert.equal(by.split.matched, "it was never about the money");
+ assert.equal(by.split.block, "html > body > p:nth-of-type(2)");
+ assert.equal(by.curly.ok, true);
+ assert.equal(by.curly.matched, "\u201Cwe\u2019re not going\u201D");
+
+ assert.deepEqual([by.hidden.ok, by.hidden.reason], [false, "quote not found"]);
+ assert.deepEqual([by.nopage.ok, by.nopage.reason], [false, "page not found"]);
+
+ for (const id of ["plain", "split", "curly"]) {
+ const r = by[id];
+ const size = pngSize(await readFile(r.png));
+ // deviceScaleFactor 2: the PNG is twice the CSS crop.
+ assert.deepEqual(size, { width: r.crop.width * 2, height: r.crop.height * 2 });
+ assert.deepEqual(r.pixels, size);
+ assert.equal(r.trimmed, false);
+ // Remote and outside-the-folder requests were refused, the saved CSS was not.
+ assert.ok(r.blocked.some((u) => u.startsWith("https://fonts.example.invalid/")), id);
+ assert.ok(r.blocked.some((u) => u.startsWith("https://example.invalid/tracker")), id);
+ assert.ok(r.blocked.some((u) => u.endsWith("/outside.png")), id);
+ assert.ok(!r.blocked.some((u) => u.endsWith("style.css")), id);
+ }
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});