Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 1bbe569d4816f82630fd9876033a071291143cb5
parent 5a2b1667e8d3ba644ef5b166f0d4208dd56a6f1c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 16:21:42 -0400

report-to-video: shoot-page — a highlighted sentence from a saved page, as a PNG

Loads a saved HTML file offline (only file:// under the page's own
directory), finds each quote by normalised text across element
boundaries, wraps it in marks and screenshots the containing block at
deviceScaleFactor 2. Batch mode writes <id>.png and results.json; a miss
is listed and exits 1. The matcher is pure and tested; the browser test
runs with SHOOT_PAGE_BROWSER=1.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Aumtool/report-to-video/shoot-page.mjs | 535+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/report-to-video/shoot-page.test.mjs | 237+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 772 insertions(+), 0 deletions(-)

diff --git a/umtool/report-to-video/shoot-page.mjs b/umtool/report-to-video/shoot-page.mjs @@ -0,0 +1,535 @@ +#!/usr/bin/env node +// shoot-page.mjs — a highlighted sentence from a SAVED web page, as a PNG. +// +// An article a report cites is evidence the way a clip is, and an `image` entry +// is how a still gets into the cut. This makes that still: load the page as it +// was saved to disk, find the quoted sentence in its text, paint it like a +// highlighter, and screenshot the paragraph that holds it. +// +// The page is loaded OFFLINE. Every request is refused except a file:// one +// under the page's own directory (the `<name>_files/` folder a browser's "save +// page, complete" writes), so a shot never reaches the network and never reads +// a file the page was not saved with. Page scripts are off by default: a saved +// page is already rendered, and its scripts, with nothing to talk to, are more +// likely to hide the article than to finish drawing it (`--js` turns them on). +// +// The quote is found by TEXT, not by selector, and loosely in exactly the ways +// copying text off a page is loose: curly and straight quotes and apostrophes +// are the same character, NBSP and every other space are a space, runs of +// whitespace are one, and soft hyphens and zero-width characters are not there +// at all. A match runs across element boundaries — a link, an `<em>`, a `<br>` +// — and the highlight is one `<mark>` per text node it covers. A quote that is +// not on the page is a MISS: listed in the results, summarised on stderr, and +// the exit status is 1. It is never skipped. +// +// On the CLI: +// node umtool/report-to-video/shoot-page.mjs --page <file.html> --quote "<text>" --out <shot.png> +// node umtool/report-to-video/shoot-page.mjs --batch <items.json> --out <dir> +// +// Batch items are `[{ id, page, quote, context? }]` (or `{ items: [...] }`), +// `page` relative to the items file. Each writes `<dir>/<id>.png`, and the run +// writes `<dir>/results.json`. `context` is a longer stretch of text around the +// quote, for a quote that occurs more than once. +// +// Options: +// --context <text> (single) as an item's `context` +// --color <css> The highlight colour (default: #ffe14d) +// --padding <px> CSS px around the block (default: 24) +// --width <px> Viewport width in CSS px (default: 1280) +// --scale <n> deviceScaleFactor (default: 2) +// --max-height <px> A taller block is trimmed to this, centred on the +// quote (default: 1200) +// --js Run the page's own scripts +// --timeout <ms> Page load timeout (default: 20000) + +import { existsSync, statSync } from "node:fs"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; +import path from "node:path"; +import { fileURLToPath, pathToFileURL } from "node:url"; + +export const DEFAULTS = Object.freeze({ + color: "#ffe14d", + padding: 24, + width: 1280, + height: 900, + scale: 2, + maxHeight: 1200, + js: false, + timeout: 20_000, +}); + +// --------------------------------------------------------------------------- +// The matcher. Pure, and SELF-CONTAINED: these two functions are also sent into +// the page as source text (see browserShoot), so neither may reference anything +// outside itself but the other by name and the language's own globals. +// --------------------------------------------------------------------------- + +// `raw` normalised for matching, with `map[k]` = the index in `raw` of the +// normalised text's k-th character. Several normalised characters may map to +// one raw one (an ellipsis is three dots); a raw character may map to none. +export function normaliseWithMap(raw) { + const out = []; + const map = []; + for (let i = 0; i < raw.length; i++) { + const c = raw[i]; + // Soft hyphen, zero-width space / non-joiner / joiner, word joiner, BOM: + // present in the DOM, invisible on the page, never in a copied quote. + if (/[\u00AD\u200B-\u200D\u2060\uFEFF]/.test(c)) continue; + // \s covers NBSP, the U+2000 block, narrow NBSP and the ideographic space. + if (/\s/.test(c)) { + if (out.length === 0 || out[out.length - 1] === " ") continue; + out.push(" "); + map.push(i); + continue; + } + let n = c; + if (/[\u2018\u2019\u201A\u201B\u2032\u02BC]/.test(c)) n = "'"; + else if (/[\u201C\u201D\u201E\u201F\u2033]/.test(c)) n = '"'; + else if (c === "\u2026") n = "..."; + for (const ch of n) { + out.push(ch); + map.push(i); + } + } + return { text: out.join(""), map }; +} + +// Find `quote` in the text of `segments` (strings, in document order, joined +// with nothing between them — a caller puts a " " segment where a block or a +// `<br>` breaks the text). With `context`, the quote must lie inside the first +// occurrence of the context. +// +// Returns `{ ok: true, start, end, occurrences }`, `start`/`end` being +// `{ segment, offset }` in the RAW segment strings (`end` exclusive, and always +// in the segment holding the match's last character), or `{ ok: false, reason }`. +export function matchInSegments(segments, quote, context) { + const q = normaliseWithMap(String(quote ?? "")).text.trim(); + if (!q) return { ok: false, reason: "empty quote" }; + const raw = segments.join(""); + const hay = normaliseWithMap(raw); + + let occurrences = 0; + for (let i = hay.text.indexOf(q); i !== -1; i = hay.text.indexOf(q, i + 1)) occurrences++; + + let at; + if (context != null && String(context).trim()) { + const c = normaliseWithMap(String(context)).text.trim(); + const ci = hay.text.indexOf(c); + if (ci === -1) return { ok: false, reason: "context not found", occurrences }; + at = hay.text.indexOf(q, ci); + if (at === -1 || at + q.length > ci + c.length) { + return { ok: false, reason: "quote not found inside its context", occurrences }; + } + } else { + at = hay.text.indexOf(q); + if (at === -1) return { ok: false, reason: "quote not found", occurrences }; + } + + const rawStart = hay.map[at]; + const rawLast = hay.map[at + q.length - 1]; + const locate = (idx) => { + let base = 0; + for (let s = 0; s < segments.length; s++) { + const len = segments[s].length; + if (idx < base + len) return { segment: s, offset: idx - base }; + base += len; + } + return null; + }; + const start = locate(rawStart); + const last = locate(rawLast); + return { ok: true, start, end: { segment: last.segment, offset: last.offset + 1 }, occurrences }; +} + +// --------------------------------------------------------------------------- +// The page side. Runs INSIDE the browser (serialised by `pageExpression`): walk +// the visible text, match, wrap the match in marks, measure what to shoot. +// --------------------------------------------------------------------------- + +function browserShoot(args) { + const { quote, context, color, padding, maxHeight } = args; + const doc = document; + const SKIP = new Set(["SCRIPT", "STYLE", "NOSCRIPT", "TEMPLATE", "TITLE", "HEAD", "SVG", "MATH", "IFRAME", "OBJECT", "SELECT", "TEXTAREA"]); + const blockCache = new Map(); + const isBlock = (el) => { + const d = getComputedStyle(el).display; + return !(d.startsWith("inline") || d === "contents" || d === "ruby" || d === "none"); + }; + const blockOf = (node) => { + let el = node.nodeType === 1 ? node : node.parentElement; + const seen = []; + while (el && el !== doc.body && el !== doc.documentElement) { + if (blockCache.has(el)) { + const b = blockCache.get(el); + for (const s of seen) blockCache.set(s, b); + return b; + } + seen.push(el); + if (isBlock(el)) break; + el = el.parentElement; + } + const b = el ?? doc.body; + for (const s of seen) blockCache.set(s, b); + return b; + }; + + const segs = []; + let lastBlock = null; + const walker = doc.createTreeWalker(doc.body ?? doc.documentElement, NodeFilter.SHOW_ELEMENT | NodeFilter.SHOW_TEXT, { + acceptNode(n) { + if (n.nodeType === 3) return NodeFilter.FILTER_ACCEPT; + if (SKIP.has(n.tagName.toUpperCase())) return NodeFilter.FILTER_REJECT; + const st = getComputedStyle(n); + if (st.display === "none" || st.visibility === "hidden") return NodeFilter.FILTER_REJECT; + return n.tagName === "BR" ? NodeFilter.FILTER_ACCEPT : NodeFilter.FILTER_SKIP; + }, + }); + for (let n = walker.nextNode(); n; n = walker.nextNode()) { + if (n.nodeType === 1) { + segs.push({ node: null, text: " " }); + continue; + } + const b = blockOf(n); + if (lastBlock && b !== lastBlock) segs.push({ node: null, text: " " }); + lastBlock = b; + segs.push({ node: n, text: n.data }); + } + + const m = matchInSegments(segs.map((s) => s.text), quote, context); + if (!m.ok) return m; + + const range = doc.createRange(); + range.setStart(segs[m.start.segment].node, m.start.offset); + range.setEnd(segs[m.end.segment].node, m.end.offset); + const matched = range.toString(); + let common = range.commonAncestorContainer; + if (common.nodeType !== 1) common = common.parentElement; + const block = common === doc.body || isBlock(common) ? common : blockOf(common); + + // One mark per text node the match covers. splitText at the end first, so + // the node in hand stays the left part; then at the start, which returns the + // covered middle. + const marks = []; + for (let i = m.start.segment; i <= m.end.segment; i++) { + let node = segs[i].node; + if (!node) continue; + const a = i === m.start.segment ? m.start.offset : 0; + const b = i === m.end.segment ? m.end.offset : node.data.length; + if (a >= b) continue; + if (b < node.data.length) node.splitText(b); + if (a > 0) node = node.splitText(a); + const mark = doc.createElement("mark"); + mark.setAttribute("data-shoot-page", ""); + // box-shadow, not padding: the highlight must not reflow the paragraph. + mark.style.cssText = + `background:${color} !important;color:inherit !important;` + + `box-shadow:0 0 0 0.12em ${color};border-radius:0.12em;` + + "-webkit-box-decoration-break:clone;box-decoration-break:clone;"; + node.parentNode.insertBefore(mark, node); + mark.appendChild(node); + marks.push(mark); + } + + const sx = window.scrollX; + const sy = window.scrollY; + const de = doc.documentElement; + const docW = Math.max(de.scrollWidth, doc.body?.scrollWidth ?? 0); + const docH = Math.max(de.scrollHeight, doc.body?.scrollHeight ?? 0); + const br = block.getBoundingClientRect(); + let top = br.top + sy; + let bottom = br.bottom + sy; + let trimmed = false; + if (bottom - top > maxHeight) { + let mt = Infinity; + let mb = -Infinity; + for (const mk of marks) { + for (const r of mk.getClientRects()) { + mt = Math.min(mt, r.top + sy); + mb = Math.max(mb, r.bottom + sy); + } + } + if (mb - mt >= maxHeight) { + top = mt; + bottom = mb; + } else { + const mid = (mt + mb) / 2; + const t = Math.min(Math.max(top, mid - maxHeight / 2), bottom - maxHeight); + top = t; + bottom = t + maxHeight; + } + trimmed = true; + } + const x0 = Math.max(0, Math.floor(br.left + sx - padding)); + const y0 = Math.max(0, Math.floor(top - padding)); + const x1 = Math.min(docW, Math.ceil(br.right + sx + padding)); + const y1 = Math.min(docH, Math.ceil(bottom + padding)); + + const cssPath = (el) => { + const parts = []; + while (el && el.nodeType === 1 && el !== de) { + if (el.id && doc.querySelectorAll(`#${CSS.escape(el.id)}`).length === 1) { + parts.unshift(`#${CSS.escape(el.id)}`); + return parts.join(" > "); + } + const tag = el.tagName.toLowerCase(); + const same = el.parentElement + ? [...el.parentElement.children].filter((c) => c.tagName === el.tagName) + : []; + parts.unshift(same.length > 1 ? `${tag}:nth-of-type(${same.indexOf(el) + 1})` : tag); + el = el.parentElement; + } + return ["html", ...parts].join(" > "); + }; + + return { + ok: true, + matched, + block: cssPath(block), + crop: { x: x0, y: y0, width: x1 - x0, height: y1 - y0 }, + trimmed, + occurrences: m.occurrences, + marks: marks.length, + }; +} + +// The expression handed to page.evaluate: the matcher and the walker as source, +// then a call. A string rather than a function so the helpers travel with it. +export function pageExpression(args) { + return `(() => {\n${normaliseWithMap}\n${matchInSegments}\n${browserShoot}\nreturn browserShoot(${JSON.stringify(args)});\n})()`; +} + +// --------------------------------------------------------------------------- +// The browser. +// --------------------------------------------------------------------------- + +// Is `url` a file the page at `pagePath` may load: file:// and under its dir. +export function allowedRequest(url, pagePath) { + if (!url.startsWith("file:")) return false; + let p; + try { + p = fileURLToPath(url); + } catch { + return false; + } + const rel = path.relative(path.dirname(path.resolve(pagePath)), path.resolve(p)); + return rel === "" || (!rel.startsWith("..") && !path.isAbsolute(rel)); +} + +async function loadChromium() { + // umtool installs @playwright/test; this package does not declare it, so it + // resolves from umtool's node_modules. It is CommonJS: accept either shape. + let mod; + try { + mod = await import("@playwright/test"); + } catch (err) { + throw new Error(`Playwright is not available here (it ships with umtool): ${err.message}`); + } + const chromium = mod.chromium ?? mod.default?.chromium; + if (!chromium) throw new Error("Playwright loaded but has no chromium export"); + return chromium; +} + +// Shoot every item; one browser for the lot, a fresh page per item (the marks +// mutate the DOM). Returns the results array; writes nothing but the PNGs. +// +// `items`: [{ id, page (absolute or relative to `baseDir`), quote, context?, out? }] +export async function shootPages(items, opts = {}) { + const o = { ...DEFAULTS, ...opts }; + const chromium = await loadChromium(); + const browser = await chromium.launch({ headless: true }); + const results = []; + try { + const ctx = await browser.newContext({ + viewport: { width: o.width, height: o.height }, + deviceScaleFactor: o.scale, + javaScriptEnabled: o.js, + bypassCSP: true, + serviceWorkers: "block", + offline: true, + }); + for (const item of items) results.push(await shootOne(ctx, item, o)); + await ctx.close(); + } finally { + await browser.close(); + } + return results; +} + +async function shootOne(ctx, item, o) { + const pagePath = path.resolve(o.baseDir ?? process.cwd(), String(item.page ?? "")); + const base = { id: item.id, page: item.page, quote: item.quote }; + if (!item.page || !existsSync(pagePath) || !statSync(pagePath).isFile()) { + return { ...base, ok: false, reason: "page not found" }; + } + const blocked = []; + const page = await ctx.newPage(); + try { + await page.route(() => true, (route) => { + const url = route.request().url(); + if (allowedRequest(url, pagePath)) return route.continue(); + blocked.push(url.length > 200 ? `${url.slice(0, 200)}\u2026` : url); + return route.abort("blockedbyclient"); + }); + await page.goto(pathToFileURL(pagePath).href, { waitUntil: "load", timeout: o.timeout }); + // Fonts are not part of "load"; a shot taken before them reflows after. + await page.evaluate("document.fonts ? document.fonts.ready.then(() => true) : true"); + const r = await page.evaluate( + pageExpression({ + quote: item.quote, + context: item.context ?? null, + color: o.color, + padding: o.padding, + maxHeight: o.maxHeight, + }), + ); + if (!r.ok) return { ...base, ok: false, reason: r.reason, occurrences: r.occurrences, blocked }; + await mkdir(path.dirname(item.out), { recursive: true }); + await page.screenshot({ path: item.out, type: "png", fullPage: true, clip: r.crop }); + return { + ...base, + ok: true, + png: item.out, + matched: r.matched, + block: r.block, + crop: r.crop, + scale: o.scale, + pixels: { width: Math.round(r.crop.width * o.scale), height: Math.round(r.crop.height * o.scale) }, + trimmed: r.trimmed, + occurrences: r.occurrences, + blocked, + }; + } catch (err) { + return { ...base, ok: false, reason: `error: ${err.message?.split("\n")[0] ?? err}`, blocked }; + } finally { + await page.close(); + } +} + +// --------------------------------------------------------------------------- +// The CLI. +// --------------------------------------------------------------------------- + +// Items from a batch file: an array, or `{ items: [...] }`. Every item needs an +// id that is a safe file name, a page and a quote; ids are unique. Throws with +// every problem at once rather than the first. +export function validateItems(data) { + const items = Array.isArray(data) ? data : data?.items; + if (!Array.isArray(items)) throw new Error("batch file must be an array of items or { items: [...] }"); + const problems = []; + const seen = new Set(); + items.forEach((it, i) => { + const where = `item ${i}${it?.id ? ` (${it.id})` : ""}`; + if (!it || typeof it !== "object") return problems.push(`${where}: not an object`); + if (typeof it.id !== "string" || !/^[A-Za-z0-9._-]+$/.test(it.id) || it.id.startsWith(".")) { + problems.push(`${where}: id must be a file-name-safe string`); + } else if (seen.has(it.id)) problems.push(`${where}: duplicate id`); + else seen.add(it.id); + if (typeof it.page !== "string" || !it.page) problems.push(`${where}: page is required`); + if (typeof it.quote !== "string" || !it.quote.trim()) problems.push(`${where}: quote is required`); + if (it.context != null && typeof it.context !== "string") problems.push(`${where}: context must be a string`); + }); + if (problems.length) throw new Error(problems.join("\n")); + return items; +} + +export function parseArgs(argv) { + const opts = {}; + const out = { mode: null, opts }; + const num = (name, v) => { + const n = Number(v); + if (!Number.isFinite(n) || n < 0) throw new Error(`${name} needs a non-negative number`); + return n; + }; + for (let i = 0; i < argv.length; i++) { + const a = argv[i]; + const next = () => { + if (i + 1 >= argv.length) throw new Error(`${a} needs a value`); + return argv[++i]; + }; + switch (a) { + case "--page": out.page = next(); break; + case "--quote": out.quote = next(); break; + case "--context": out.context = next(); break; + case "--batch": out.batch = next(); break; + case "--out": out.out = next(); break; + case "--color": opts.color = next(); break; + case "--padding": opts.padding = num(a, next()); break; + case "--width": opts.width = num(a, next()); break; + case "--scale": opts.scale = num(a, next()); break; + case "--max-height": opts.maxHeight = num(a, next()); break; + case "--timeout": opts.timeout = num(a, next()); break; + case "--js": opts.js = true; break; + default: throw new Error(`unknown argument: ${a}`); + } + } + if (out.batch && (out.page || out.quote)) throw new Error("--batch and --page/--quote are two modes; pick one"); + if (!out.out) throw new Error("--out is required"); + if (out.batch) out.mode = "batch"; + else if (out.page && out.quote) out.mode = "single"; + else throw new Error("give --batch <items.json>, or --page and --quote"); + return out; +} + +const USAGE = + "usage: shoot-page.mjs --page <file.html> --quote <text> [--context <text>] --out <shot.png>\n" + + " shoot-page.mjs --batch <items.json> --out <dir>\n" + + " [--color <css>] [--padding <px>] [--width <px>] [--scale <n>] [--max-height <px>] [--js] [--timeout <ms>]"; + +async function main() { + let args; + try { + args = parseArgs(process.argv.slice(2)); + } catch (err) { + console.error(`${err.message}\n${USAGE}`); + process.exit(2); + } + + let items; + let resultsFile = null; + if (args.mode === "batch") { + const batchPath = path.resolve(args.batch); + items = validateItems(JSON.parse(await readFile(batchPath, "utf8"))); + const outDir = path.resolve(args.out); + items = items.map((it) => ({ ...it, out: path.join(outDir, `${it.id}.png`) })); + args.opts.baseDir = path.dirname(batchPath); + resultsFile = path.join(outDir, "results.json"); + } else { + items = [{ id: "shot", page: args.page, quote: args.quote, context: args.context, out: path.resolve(args.out) }]; + } + + const results = await shootPages(items, args.opts); + const missed = results.filter((r) => !r.ok); + const report = { + generatedAt: new Date().toISOString(), + options: { ...DEFAULTS, ...args.opts }, + shot: results.length - missed.length, + missed: missed.length, + items: results, + }; + if (resultsFile) { + await mkdir(path.dirname(resultsFile), { recursive: true }); + await writeFile(resultsFile, JSON.stringify(report, null, 2) + "\n", "utf8"); + } else { + console.log(JSON.stringify(results[0], null, 2)); + } + + for (const r of results) { + if (r.ok && r.occurrences > 1) { + console.error(`note: ${r.id}: the quote occurs ${r.occurrences} times; shot the first (give a context to pick another)`); + } + } + if (missed.length) { + console.error(`MISSED ${missed.length} of ${results.length}:`); + for (const r of missed) console.error(` ${r.id}: ${r.reason} — ${r.page}`); + if (resultsFile) console.error(`results: ${resultsFile}`); + process.exit(1); + } + if (resultsFile) console.error(`shot ${results.length} -> ${resultsFile}`); +} + +if (import.meta.url === `file://${process.argv[1]}`) { + main().catch((err) => { + console.error(err.message ?? err); + process.exit(1); + }); +} diff --git a/umtool/report-to-video/shoot-page.test.mjs b/umtool/report-to-video/shoot-page.test.mjs @@ -0,0 +1,237 @@ +// Tests for shoot-page.mjs — finding a quoted sentence in a saved page's text. +// +// The matcher is pure and is tested against text segments shaped like the +// page's text nodes in document order. One test drives a real Chromium over a +// fixture page; it launches a browser, so it runs only with SHOOT_PAGE_BROWSER=1. +// +// Run with: pnpm test:scripts +import assert from "node:assert/strict"; +import { mkdtemp, readFile, rm, writeFile, mkdir } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import test from "node:test"; + +import { + allowedRequest, matchInSegments, normaliseWithMap, pageExpression, parseArgs, shootPages, validateItems, +} from "./shoot-page.mjs"; + +// The raw text a match covers, read back off the segments. +function covered(segments, m) { + const out = []; + for (let s = m.start.segment; s <= m.end.segment; s++) { + const a = s === m.start.segment ? m.start.offset : 0; + const b = s === m.end.segment ? m.end.offset : segments[s].length; + out.push(segments[s].slice(a, b)); + } + return out.join(""); +} + +test("normalising: quotes, spaces, invisibles, ellipsis — with a map back to the raw text", () => { + const raw = "\u201CIt\u2019s\u00A0fine\u201D, she\u00ADsaid\u200B\u2026"; + const { text, map } = normaliseWithMap(raw); + assert.equal(text, "\"It's fine\", shesaid..."); + assert.equal(map.length, text.length); + // Every normalised character points at the raw one it came from. + assert.equal(raw[map[text.indexOf("s", 4)]], "s"); + assert.equal(map[text.length - 1], raw.length - 1); + assert.equal(map[text.length - 3], raw.length - 1); + // Leading whitespace goes; a run of it is one space. + assert.equal(normaliseWithMap(" \n\t a \n b").text, "a b"); +}); + +test("a plain quote inside one text node", () => { + const segs = ["The committee said the plan was not funded and would be dropped."]; + const m = matchInSegments(segs, "the plan was not funded"); + assert.equal(m.ok, true); + assert.deepEqual(m.start, { segment: 0, offset: 19 }); + assert.equal(covered(segs, m), "the plan was not funded"); + assert.equal(m.occurrences, 1); +}); + +test("a quote split across <em> and <a> matches across the boundaries", () => { + // <p>She wrote that it was <em>never</em> about <a href>the money</a>, at all.</p> + const segs = ["She wrote that it was ", "never", " about ", "the money", ", at all."]; + const m = matchInSegments(segs, "it was never about the money"); + assert.equal(m.ok, true); + assert.deepEqual(m.start, { segment: 0, offset: 15 }); + // The end lies in the link's text node, not at the start of the next one. + assert.deepEqual(m.end, { segment: 3, offset: 9 }); + assert.equal(covered(segs, m), "it was never about the money"); +}); + +test("curly on the page, straight in the quote — and the other way round", () => { + const curly = ["He said \u201Cwe\u2019re not going\u201D and left."]; + const m1 = matchInSegments(curly, "\"we're not going\""); + assert.equal(m1.ok, true); + assert.equal(covered(curly, m1), "\u201Cwe\u2019re not going\u201D"); + + const straight = ["He said \"we're not going\" and left."]; + const m2 = matchInSegments(straight, "\u201Cwe\u2019re not going\u201D"); + assert.equal(m2.ok, true); + assert.equal(covered(straight, m2), "\"we're not going\""); +}); + +test("NBSP, soft hyphens, zero-width characters and source line breaks do not stop a match", () => { + const segs = ["an un\u00ADprecedented\u00A0and\n ", "unre\u200Bcoverable", " loss"]; + const m = matchInSegments(segs, "an unprecedented and unrecoverable loss"); + assert.equal(m.ok, true); + assert.equal(covered(segs, m), segs.join("")); +}); + +test("a block break between two text nodes reads as a space, never as nothing", () => { + // The caller puts a " " segment where one block ends and another starts. + const segs = ["first paragraph.", " ", "Second paragraph."]; + assert.equal(matchInSegments(segs, "paragraph. Second").ok, true); + assert.equal(matchInSegments(segs, "paragraph.Second").ok, false); +}); + +test("a quote that is not there is a miss with a reason", () => { + const m = matchInSegments(["Nothing to see here."], "something else entirely"); + assert.deepEqual(m, { ok: false, reason: "quote not found", occurrences: 0 }); + assert.equal(matchInSegments(["text"], " ").reason, "empty quote"); +}); + +test("context picks the occurrence, and is itself required to be there", () => { + const segs = ["It was late. ", "Later, it was late again, said the second witness."]; + const plain = matchInSegments(segs, "it was late"); + // Case matters: only the second sentence's lowercase "it was late" matches. + assert.equal(plain.ok, true); + assert.equal(plain.start.segment, 1); + + const two = ["It was late. It was late again."]; + const first = matchInSegments(two, "It was late"); + assert.equal(first.occurrences, 2); + assert.equal(first.start.offset, 0); + const second = matchInSegments(two, "It was late", "It was late again"); + assert.equal(second.ok, true); + assert.equal(second.start.offset, 13); + + assert.equal(matchInSegments(two, "It was late", "not on the page").reason, "context not found"); + assert.equal( + matchInSegments(two, "late again", "It was late. It").reason, + "quote not found inside its context", + ); +}); + +test("the page expression carries the matcher with it", () => { + const expr = pageExpression({ quote: "a \"b\"", context: null, color: "#ff0", padding: 4, maxHeight: 100 }); + assert.match(expr, /function normaliseWithMap/); + assert.match(expr, /function matchInSegments/); + assert.match(expr, /return browserShoot\(\{"quote":"a \\"b\\""/); + // It parses as one expression. + assert.doesNotThrow(() => new Function(`return ${expr.replace(/return browserShoot\([^\n]*\);/, "return 0;")}`)); +}); + +test("only file:// under the page's own directory is allowed", () => { + const page = "/tmp/saved/article.html"; + assert.equal(allowedRequest("file:///tmp/saved/article.html", page), true); + assert.equal(allowedRequest("file:///tmp/saved/article_files/style.css", page), true); + assert.equal(allowedRequest("file:///tmp/saved/../other/x.css", page), false); + assert.equal(allowedRequest("file:///tmp/savedx/x.css", page), false); + assert.equal(allowedRequest("file:///etc/passwd", page), false); + assert.equal(allowedRequest("https://example.com/x.png", page), false); + assert.equal(allowedRequest("data:image/png;base64,AAAA", page), false); +}); + +test("batch items: every problem at once", () => { + assert.deepEqual(validateItems({ items: [{ id: "a1", page: "p.html", quote: "q" }] }).length, 1); + assert.throws( + () => validateItems([ + { id: "a1", page: "p.html", quote: "q" }, + { id: "a1", page: "p.html", quote: "q" }, + { id: "../x", page: "", quote: " " }, + ]), + (err) => /duplicate id/.test(err.message) && /file-name-safe/.test(err.message) + && /page is required/.test(err.message) && /quote is required/.test(err.message), + ); + assert.throws(() => validateItems({}), /array of items/); +}); + +test("arguments: two modes, never both", () => { + assert.equal(parseArgs(["--batch", "i.json", "--out", "d"]).mode, "batch"); + const s = parseArgs(["--page", "p.html", "--quote", "q", "--out", "s.png", "--scale", "3", "--js"]); + assert.equal(s.mode, "single"); + assert.deepEqual(s.opts, { scale: 3, js: true }); + assert.throws(() => parseArgs(["--batch", "i.json", "--page", "p", "--out", "d"]), /two modes/); + assert.throws(() => parseArgs(["--page", "p.html", "--out", "s.png"]), /--batch/); + assert.throws(() => parseArgs(["--page", "p.html", "--quote", "q"]), /--out is required/); + assert.throws(() => parseArgs(["--padding", "-1"]), /non-negative/); + assert.throws(() => parseArgs(["--bogus"]), /unknown argument/); +}); + +// --------------------------------------------------------------------------- +// A real browser over a fixture page. Heavy, so opt-in. +// --------------------------------------------------------------------------- + +const FIXTURE = `<!doctype html> +<html><head><meta charset="utf-8"> +<link rel="stylesheet" href="article_files/style.css"> +<link rel="stylesheet" href="https://fonts.example.invalid/remote.css"> +<style>body { margin: 0; padding: 40px; font: 18px/1.5 serif; width: 700px; }</style> +</head><body> +<header style="position: sticky; top: 0">Header</header> +<p id="plain">The council voted on Tuesday. The plan was not funded and would be dropped.</p> +<p>She wrote that it was <em>never</em> about <a href="https://example.invalid/">the money</a>, at all.</p> +<p>He said \u201Cwe\u2019re not going\u201D and left.</p> +<div style="display:none">a hidden sentence that must not match</div> +<img src="https://example.invalid/tracker.png" alt=""> +<img src="../outside.png" alt=""> +</body></html>`; + +function pngSize(buf) { + // IHDR: width and height are the big-endian words at bytes 16 and 20. + assert.equal(buf.toString("ascii", 1, 4), "PNG"); + return { width: buf.readUInt32BE(16), height: buf.readUInt32BE(20) }; +} + +test("a real page: shots, marks, a miss, and nothing fetched from outside", { + skip: process.env.SHOOT_PAGE_BROWSER === "1" ? false : "set SHOOT_PAGE_BROWSER=1 to launch Chromium", +}, async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), "shoot-page-")); + try { + const saved = path.join(dir, "saved"); + await mkdir(path.join(saved, "article_files"), { recursive: true }); + await writeFile(path.join(saved, "article.html"), FIXTURE); + await writeFile(path.join(saved, "article_files", "style.css"), "p { color: #222; }"); + await writeFile(path.join(dir, "outside.png"), ""); + const out = path.join(dir, "out"); + const items = [ + { id: "plain", page: "saved/article.html", quote: "The plan was not funded" }, + { id: "split", page: "saved/article.html", quote: "it was never about the money" }, + { id: "curly", page: "saved/article.html", quote: "\"we're not going\"" }, + { id: "hidden", page: "saved/article.html", quote: "a hidden sentence" }, + { id: "nopage", page: "saved/missing.html", quote: "anything" }, + ].map((it) => ({ ...it, out: path.join(out, `${it.id}.png`) })); + + const results = await shootPages(items, { baseDir: dir, padding: 10 }); + const by = Object.fromEntries(results.map((r) => [r.id, r])); + + assert.equal(by.plain.ok, true); + assert.equal(by.plain.matched, "The plan was not funded"); + assert.equal(by.plain.block, "#plain"); + assert.equal(by.split.ok, true); + assert.equal(by.split.matched, "it was never about the money"); + assert.equal(by.split.block, "html > body > p:nth-of-type(2)"); + assert.equal(by.curly.ok, true); + assert.equal(by.curly.matched, "\u201Cwe\u2019re not going\u201D"); + + assert.deepEqual([by.hidden.ok, by.hidden.reason], [false, "quote not found"]); + assert.deepEqual([by.nopage.ok, by.nopage.reason], [false, "page not found"]); + + for (const id of ["plain", "split", "curly"]) { + const r = by[id]; + const size = pngSize(await readFile(r.png)); + // deviceScaleFactor 2: the PNG is twice the CSS crop. + assert.deepEqual(size, { width: r.crop.width * 2, height: r.crop.height * 2 }); + assert.deepEqual(r.pixels, size); + assert.equal(r.trimmed, false); + // Remote and outside-the-folder requests were refused, the saved CSS was not. + assert.ok(r.blocked.some((u) => u.startsWith("https://fonts.example.invalid/")), id); + assert.ok(r.blocked.some((u) => u.startsWith("https://example.invalid/tracker")), id); + assert.ok(r.blocked.some((u) => u.endsWith("/outside.png")), id); + assert.ok(!r.blocked.some((u) => u.endsWith("style.css")), id); + } + } finally { + await rm(dir, { recursive: true, force: true }); + } +});