// QUOTE VERIFICATION — how compose checks that a citation's `quote` is what // the record says at the cited place, and the block it writes. // // THE METHOD (QUOTE_CHECK_METHOD, written into every block so a reader knows // what the number means): // // 1. The record's text at the cited place: for a span, the text of every cue // overlapping [start − 5 s, end + 5 s] (QUOTE_WINDOW_SLACK_SECONDS) — a cue // boundary is where a caption line wrapped, so a quote may run a little // past the span; for a post, the post's text. // 2. Both are normalised the same way: lowercased (Unicode-aware), every // character that is not a letter, a digit or whitespace removed (so // "don't" is "dont" and "U.S." is "us"), split on whitespace. // 3. The score is TOKEN RECALL: the share of the quote's tokens found in the // record's tokens, each record token used at most once (a multiset). A // verbatim quote scores 1 however much more the window says; a quote // with no tokens scores 0. // // The score is rounded to two decimals. Below QUOTE_DRIFT_THRESHOLD the quote // has DRIFTED from the record (a paraphrase, the wrong span, cues that moved // since the quote was taken) and compose fails on it — CITATIONS.md promises // a quote is verbatim, and this is the check that holds it to that. // // The block is COMPUTED: compose overwrites whatever a document carried // (lib/citations/schema.ts verificationSchema), and nothing here can vouch for // a voice, so `voiceChecked` is never written. // // Pure, no imports but types: the export site and the browser can use it. import type { CitationVerification } from "./schema"; export const QUOTE_WINDOW_SLACK_SECONDS = 5; export const QUOTE_DRIFT_THRESHOLD = 0.6; export const QUOTE_CHECK_METHOD = "token recall v1: quote vs the cues within ±5 s of the span (a post: its text); lowercase, punctuation stripped"; type TimedText = { start: number; end: number; text: string }; // A text's comparison tokens (step 2 above). export function quoteTokens(text: string): string[] { return text .toLowerCase() .replace(/[^\p{L}\p{N}\s]/gu, "") .split(/\s+/) .filter(Boolean); } // The share of `quote`'s tokens found in `text` (step 3), unrounded. export function tokenRecall(quote: string, text: string): number { const want = quoteTokens(quote); if (want.length === 0) return 0; const have = new Map(); for (const t of quoteTokens(text)) have.set(t, (have.get(t) ?? 0) + 1); let found = 0; for (const t of want) { const n = have.get(t) ?? 0; if (n > 0) { found++; have.set(t, n - 1); } } return found / want.length; } // The text of every cue overlapping [start − slack, end + slack], in order. export function cueWindowText( cues: readonly TimedText[], start: number, end: number, slack = QUOTE_WINDOW_SLACK_SECONDS, ): string { const from = start - slack; const to = end + slack; return cues .filter((c) => c.end > from && c.start < to) .map((c) => c.text) .join(" "); } export function roundScore(score: number): number { return Math.round(score * 100) / 100; } // The verification block for a quote checked against `text` at `checkedAt`. export function quoteVerification(quote: string, text: string, checkedAt: string): CitationVerification { return { quoteScore: roundScore(tokenRecall(quote, text)), quoteCheckedAt: checkedAt, method: QUOTE_CHECK_METHOD, }; } export function quoteDrifted(v: Pick): boolean { return (v.quoteScore ?? 0) < QUOTE_DRIFT_THRESHOLD; }