// Finding a text anchor again in text that may have changed. // // A note on an article is anchored by its QUOTE plus 32 characters of context // either side (the W3C TextQuoteSelector), never by an offset: report.json is // regenerated from a draft, and an offset into the old text points at nothing // after the agent edits the paragraph above it. // // PURE and client-safe: no node imports. The article page runs it against the // rendered DOM text of a section; `umtool notes` runs it against the section's // plain text; the unit test runs it against both. export const CONTEXT = 32; /** Common-suffix length of `a` and `b` (how much of the prefix still precedes). */ function suffixMatch(a, b) { let n = 0; while (n < a.length && n < b.length && a[a.length - 1 - n] === b[b.length - 1 - n]) n += 1; return n; } /** Common-prefix length of `a` and `b` (how much of the suffix still follows). */ function prefixMatch(a, b) { let n = 0; while (n < a.length && n < b.length && a[n] === b[n]) n += 1; return n; } function allIndexes(hay, needle) { const out = []; if (!needle) return out; for (let i = hay.indexOf(needle); i !== -1; i = hay.indexOf(needle, i + 1)) out.push(i); return out; } /** Of several hits, the one whose surroundings best match prefix/suffix. Ties: the first. */ function best(hay, hits, len, prefix, suffix) { let top = hits[0]; let topScore = -1; for (const i of hits) { const score = suffixMatch(hay.slice(Math.max(0, i - prefix.length), i), prefix) + prefixMatch(hay.slice(i + len, i + len + suffix.length), suffix); if (score > topScore) { top = i; topScore = score; } } return top; } /** * Whitespace collapsed (and typographic quotes/dashes folded), with a map from * each normalised index back to the original one. `lower` also lowercases. */ export function normalise(text, { lower = false } = {}) { const chars = []; const map = []; let space = false; for (let i = 0; i < text.length; i += 1) { let c = text[i]; if (/\s/.test(c)) { if (space || chars.length === 0) continue; space = true; chars.push(" "); map.push(i); continue; } space = false; if (c === "‘" || c === "’") c = "'"; else if (c === "“" || c === "”") c = '"'; else if (c === "–" || c === "—") c = "-"; else if (c === "…") c = "."; if (lower) c = c.toLowerCase(); chars.push(c); map.push(i); } if (chars[chars.length - 1] === " ") { chars.pop(); map.pop(); } map.push(text.length); return { text: chars.join(""), map }; } const norm = (s, lower) => normalise(s ?? "", { lower }).text; /** * Locate a text anchor. `{ found: true, start, end, how }` with offsets into * `text`, or `{ found: false }` -- an ORPHANED note, still shown, pinned to its * section. `how`: "exact", "normalised" (whitespace/quotes/case differ), or * "context" (the quote itself was edited, but what came before and after it is * still there, close together). * * @param {string} text * @param {{ quote: string, prefix?: string, suffix?: string }} anchor */ export function locateQuote(text, anchor) { const quote = anchor?.quote ?? ""; const prefix = anchor?.prefix ?? ""; const suffix = anchor?.suffix ?? ""; if (!text || !quote) return { found: false }; const exact = allIndexes(text, quote); if (exact.length) { const i = best(text, exact, quote.length, prefix, suffix); return { found: true, start: i, end: i + quote.length, how: "exact" }; } for (const lower of [false, true]) { const n = normalise(text, { lower }); const q = norm(quote, lower); if (!q) continue; const hits = allIndexes(n.text, q); if (hits.length) { const i = best(n.text, hits, q.length, norm(prefix, lower), norm(suffix, lower)); return { found: true, start: n.map[i], end: n.map[i + q.length - 1] + 1, how: "normalised" }; } } // The quote was rewritten. If the context on BOTH sides survives, close // together, the span between them is where it was. const n = normalise(text, { lower: true }); const p = norm(prefix, true); const s = norm(suffix, true); if (p.length >= 8 && s.length >= 8) { const q = norm(quote, true); for (const pi of allIndexes(n.text, p)) { const from = pi + p.length; const si = n.text.indexOf(s, from); if (si === -1) continue; const span = si - from; if (span <= 0 || span > Math.max(q.length * 2, q.length + 80)) continue; // Trim the separator spaces the normalised text keeps around the span. let a = from; let b = si; while (a < b && n.text[a] === " ") a += 1; while (b > a && n.text[b - 1] === " ") b -= 1; if (a >= b) continue; return { found: true, start: n.map[a], end: n.map[b - 1] + 1, how: "context" }; } } return { found: false }; } /** * The anchor for a selection `[start, end)` of `text`: the quote (trimmed of * surrounding whitespace) and up to CONTEXT characters either side. * * @param {string} text * @param {number} start * @param {number} end * @param {number} [context] */ export function quoteAnchor(text, start, end, context = CONTEXT) { let a = Math.max(0, Math.min(start, end)); let b = Math.min(text.length, Math.max(start, end)); while (a < b && /\s/.test(text[a])) a += 1; while (b > a && /\s/.test(text[b - 1])) b -= 1; return { quote: text.slice(a, b), prefix: text.slice(Math.max(0, a - context), a), suffix: text.slice(b, b + context), }; } /** The sentence of `text` around `[start, end)`, for a reader with no page open. */ export function sentenceAround(text, start, end, max = 400) { const before = text.slice(0, start); const after = text.slice(end); const boundary = /[.?!]\s+|\n/g; let from = 0; for (let m = boundary.exec(before); m; m = boundary.exec(before)) from = m.index + m[0].length; const m = after.search(/[.?!](\s|$)|\n/); const to = m === -1 ? text.length : end + m + (after[m] === "\n" ? 0 : 1); const out = text.slice(from, to).replace(/\s+/g, " ").trim(); return out.length > max ? `${out.slice(0, max - 1)}…` : out; }