commit 5f571b84df3de9e6651a31b261017de78a9ebc23
parent 0a11ff20a3c61dbd15d3155fa648ec54e67fd70d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 19 Sep 2026 03:11:39 -0400
resolve-windows: `--cut-to-quote` — where inside the extent the quote is
A reviewed window is the clip's USEFUL EXTENT: the operator widens it until
the material is worth having, which is a judgement about the recording and
not about any one paragraph. The cut the video plays is a different question
-- the sentences the quote is actually made of -- and deriving the second
from the first is what lets somebody review once and re-cut later without
watching anything again.
The match is deliberately loose, because a quote is prose a human wrote about
speech a machine transcribed: case and punctuation go, `[bracketed]` editorial
words and `[Speaker]` markers go, and an ellipsis SPLITS the quote into
fragments located independently -- "A … B" is two places in the recording and
the cut spans from the first to the last. A fragment counts as located at half
its words; below that the run is a coincidence of common words and using it as
an edge would cut somewhere nobody chose. The whole quote must reach
--min-match (0.6) or the clip is reported UNMATCHED with its best partial and
nothing is written.
Ties go to the SHORTER run, which the tests caught: a run that adds the
preceding cue without matching another word scores identically and would hand
back the whole lead-in as the cut on every clip whose quote starts at a cue
boundary.
It is its own pass rather than part of the widening run. Widening moves the
EXTENT, and a flag that silently did both would make the operator's judgement
for them.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
2 files changed, 274 insertions(+), 0 deletions(-)
diff --git a/umtool/report-to-video/cut-to-quote.test.mjs b/umtool/report-to-video/cut-to-quote.test.mjs
@@ -0,0 +1,70 @@
+// Tests for `--cut-to-quote`'s matcher: where inside a reviewed window the
+// quote actually is.
+//
+// Pure, so it needs no cue files and no audio. The quote is prose a human
+// wrote about speech a machine transcribed, and every case here is one of the
+// ways those two disagree in the real manifests.
+//
+// Run with: pnpm test:scripts
+import assert from "node:assert/strict";
+import test from "node:test";
+
+import { cutToQuote, quoteFragments, normaliseWords } from "./resolve-windows.mjs";
+
+/** A cue list at one second per cue, which makes every expectation exact. */
+const cues = [
+ "Before any of this, a bit of throat clearing.",
+ "So here is the thing I actually said.",
+ "And it carries on into a second sentence.",
+ "Then something else entirely, unrelated.",
+ "A last line, for the tail.",
+].map((text, i) => ({ start: i * 4, end: i * 4 + 4, text }));
+
+const extent = { start: 0, end: 20 };
+
+test("the cut is the sentences the quote is in, not the whole window", () => {
+ const r = cutToQuote(cues, "here is the thing I actually said", extent);
+ assert.equal(r.ok, true);
+ assert.equal(r.cutStart, 4);
+ assert.equal(r.cutEnd, 8);
+ assert.ok(r.score >= 0.9, `score ${r.score}`);
+});
+
+test("an ellipsis is two places, and the cut spans both", () => {
+ const r = cutToQuote(cues, "here is the thing … into a second sentence", extent);
+ assert.equal(r.ok, true);
+ assert.equal(r.cutStart, 4);
+ assert.equal(r.cutEnd, 12);
+});
+
+test("bracketed editorial words are not looked for", () => {
+ // "[He]" is the report's word, not his, and counting it against the match
+ // would push a good quote under the threshold.
+ const r = cutToQuote(cues, "[He said] here is the thing I actually said", extent);
+ assert.equal(r.ok, true);
+ assert.equal(r.cutStart, 4);
+});
+
+test("a quote that is not in the window is UNMATCHED, not a guess", () => {
+ const r = cutToQuote(cues, "nothing of the sort was ever uttered aloud here", extent);
+ assert.equal(r.ok, false);
+ assert.ok(r.score < 0.6, `score ${r.score}`);
+});
+
+test("the cut never leaves the extent", () => {
+ const tight = { start: 5, end: 7 };
+ const r = cutToQuote(cues, "here is the thing I actually said", tight);
+ assert.equal(r.ok, true);
+ assert.ok(r.cutStart >= tight.start);
+ assert.ok(r.cutEnd <= tight.end);
+});
+
+test("no quote is not a failure to match, it is nothing to match", () => {
+ assert.equal(cutToQuote(cues, "", extent).why, "no quote");
+ assert.equal(cutToQuote(cues, null, extent).why, "no quote");
+});
+
+test("normalisation drops case, punctuation and brackets", () => {
+ assert.equal(normaliseWords("So, [Speaker] HERE -- is it?"), "so here is it");
+ assert.deepEqual(quoteFragments("one two ... three"), [["one", "two"], ["three"]]);
+});
diff --git a/umtool/report-to-video/resolve-windows.mjs b/umtool/report-to-video/resolve-windows.mjs
@@ -18,6 +18,14 @@
// node umtool/report-to-video/resolve-windows.mjs <manifest.json> [--write]
//
// Options:
+// --cut-to-quote A DIFFERENT PASS: leave the windows alone and derive each
+// clip's `cutStart`/`cutEnd` -- the tight cut the video
+// renders -- from where its `quote` actually is inside its
+// window. See "extent vs cut" in docs/report-video.md.
+// --min-match <f> Reject a cut whose quote is less than this well matched
+// (default 0.6). Below it the clip is reported UNMATCHED and
+// nothing is written for it.
+// --force-cut Overwrite cut fields that are already there
// --write Rewrite the manifest in place (default: dry run, print a table)
// --max-lead <s> Max seconds to expand backwards (default 9)
// --max-tail <s> Max seconds to expand forwards (default 12)
@@ -101,6 +109,182 @@ export function widen(cues, start, end, { maxLead = 8, maxTail = 12 } = {}) {
};
}
+// ---------------------------------------------------------------------------
+// THE CUT INSIDE THE EXTENT.
+//
+// A reviewed window is the clip's USEFUL EXTENT -- the operator widens it until
+// the material is worth having, which is a judgement about the recording. The
+// cut the video plays is a different question: the sentences the quote is
+// actually made of. Deriving the second from the first is what lets somebody
+// review once and re-cut later without watching anything again.
+//
+// The quote is prose a human wrote about speech a machine transcribed, so the
+// match is deliberately loose: case and punctuation go, `[bracketed]` editorial
+// words and `[Speaker]` markers go, and an ellipsis SPLITS the quote into
+// fragments that are located independently -- "A … B" is two places in the
+// recording and the cut is the span from the first to the last.
+// ---------------------------------------------------------------------------
+
+/** Lowercase, no punctuation, no bracketed editorial, single spaces. */
+export const normaliseWords = (s) =>
+ String(s ?? "")
+ .toLowerCase()
+ .replace(/\[[^\]]*\]/g, " ")
+ .replace(/[^a-z0-9' ]+/g, " ")
+ .replace(/\s+/g, " ")
+ .trim();
+
+/** A quote as the fragments an ellipsis divides it into, each a token list. */
+export function quoteFragments(quote) {
+ return String(quote ?? "")
+ .replace(/\[[^\]]*\]/g, " ")
+ .split(/\s*(?:\.\.\.|…)\s*/)
+ .map((f) => normaliseWords(f).split(" ").filter(Boolean))
+ .filter((toks) => toks.length);
+}
+
+/** How many of `want`'s tokens appear in `have`, counting repeats once each. */
+function overlap(want, have) {
+ const pool = new Map();
+ for (const t of have) pool.set(t, (pool.get(t) ?? 0) + 1);
+ let n = 0;
+ for (const t of want) {
+ const c = pool.get(t) ?? 0;
+ if (c > 0) {
+ pool.set(t, c - 1);
+ n += 1;
+ }
+ }
+ return n;
+}
+
+/** The contiguous run of cues that holds the most of one fragment. */
+function bestRun(cueToks, frag) {
+ let best = { i: 0, j: -1, hit: 0, len: Infinity };
+ const cap = frag.length * 2 + 6;
+ for (let i = 0; i < cueToks.length; i += 1) {
+ let acc = [];
+ for (let j = i; j < cueToks.length; j += 1) {
+ acc = acc.concat(cueToks[j]);
+ if (acc.length > cap) break;
+ const hit = overlap(frag, acc);
+ // TIES GO TO THE SHORTER RUN, and that is not a detail: a run that adds
+ // the preceding cue without matching another word scores the same and
+ // would hand back an edge nobody chose -- the whole lead-in, on every
+ // clip whose quote starts at a cue boundary.
+ if (hit > best.hit || (hit === best.hit && hit > 0 && acc.length < best.len)) {
+ best = { i, j, hit, len: acc.length };
+ }
+ }
+ }
+ return best;
+}
+
+/**
+ * Where inside [start,end] the quote actually is.
+ *
+ * @returns {{ok:true,cutStart:number,cutEnd:number,score:number,matched:string}
+ * | {ok:false,score:number,why:string,matched:string}}
+ */
+export function cutToQuote(cues, quote, extent, { minMatch = 0.6, maxLead = 8, maxTail = 12 } = {}) {
+ const frags = quoteFragments(quote);
+ if (!frags.length) return { ok: false, score: 0, why: "no quote", matched: "" };
+
+ const inside = cues.filter((c) => c.end > extent.start + EPS && c.start < extent.end - EPS);
+ if (!inside.length) {
+ return { ok: false, score: 0, why: "no cues inside the window", matched: "" };
+ }
+ const cueToks = inside.map((c) => normaliseWords(c.text).split(" ").filter(Boolean));
+
+ let hits = 0;
+ let total = 0;
+ let lo = null;
+ let hi = null;
+ const matched = [];
+ for (const frag of frags) {
+ total += frag.length;
+ const b = bestRun(cueToks, frag);
+ if (b.j < b.i) continue;
+ hits += b.hit;
+ // A fragment counts as LOCATED at half its words. Below that the run is a
+ // coincidence of common words and using it as an edge would cut somewhere
+ // nobody chose.
+ if (b.hit / frag.length < 0.5) continue;
+ lo = lo === null ? b.i : Math.min(lo, b.i);
+ hi = hi === null ? b.j : Math.max(hi, b.j);
+ matched.push(inside.slice(b.i, b.j + 1).map((c) => String(c.text).trim()).join(" "));
+ }
+ const score = total ? hits / total : 0;
+ const text = matched.join(" … ");
+ if (lo === null) return { ok: false, score, why: "quote not found in the window", matched: text };
+ if (score < minMatch) return { ok: false, score, why: "below --min-match", matched: text };
+
+ // Outward to sentence edges, the same widen() the extent pass uses -- a cut
+ // that starts mid-clause is the defect this whole file exists to remove --
+ // and then back inside the extent, which is the reviewed judgement and wins.
+ const w = widen(cues, inside[lo].start, inside[hi].end, { maxLead, maxTail });
+ const cutStart = Math.max(extent.start, Math.min(w.start, inside[lo].start));
+ const cutEnd = Math.min(extent.end, Math.max(w.end, inside[hi].end));
+ if (cutEnd - cutStart < 0.5) {
+ return { ok: false, score, why: "the matched span is under half a second", matched: text };
+ }
+ return {
+ ok: true,
+ cutStart: Number(cutStart.toFixed(2)),
+ cutEnd: Number(cutEnd.toFixed(2)),
+ score: Number(score.toFixed(2)),
+ matched: text,
+ };
+}
+
+/**
+ * The `--cut-to-quote` pass: derive every clip's cut, print it, maybe write it.
+ *
+ * Deliberately NOT part of the widening run. Widening moves the extent, and the
+ * extent is the operator's own judgement about how much of the recording is
+ * worth having; a flag that silently did both would make one of those two
+ * decisions on their behalf.
+ */
+async function cutPass(manifest, { loadCues, slug, opts, write, force }) {
+ let changed = 0;
+ let unmatched = 0;
+ for (const e of manifest.timeline) {
+ if (e.type !== "clip") continue;
+ const id = String(e.id).padEnd(4);
+ if (e.lockCut) {
+ console.log(`${id} ${String(e.video).padEnd(12)} lockCut — left at ${e.cutStart}–${e.cutEnd}`);
+ continue;
+ }
+ if (!force && (e.cutStart != null || e.cutEnd != null)) {
+ console.log(`${id} ${String(e.video).padEnd(12)} already cut ${e.cutStart}–${e.cutEnd} (--force-cut to redo)`);
+ continue;
+ }
+ const chan = e.channel ?? slug;
+ const cues = await loadCues(e.video, chan, { siteChannel: e.siteChannel, siteVideo: e.siteVideo });
+ const r = cutToQuote(cues, e.quote, { start: e.start, end: e.end }, opts);
+ const extent = `${e.start.toFixed(1)}–${e.end.toFixed(1)}`;
+ if (!r.ok) {
+ unmatched += 1;
+ console.log(
+ `${id} ${String(e.video).padEnd(12)} ${extent} UNMATCHED (${r.score.toFixed(2)}) — ${r.why}`,
+ );
+ if (r.matched) console.log(` best partial: ${r.matched.slice(0, 120)}`);
+ continue;
+ }
+ console.log(
+ `${id} ${String(e.video).padEnd(12)} ${extent} -> cut ${r.cutStart.toFixed(1)}–${r.cutEnd.toFixed(1)} ` +
+ `(${(r.cutEnd - r.cutStart).toFixed(1)}s, match ${r.score.toFixed(2)})`,
+ );
+ console.log(` ${r.matched.slice(0, 120)}`);
+ if (write) {
+ e.cutStart = r.cutStart;
+ e.cutEnd = r.cutEnd;
+ }
+ changed += 1;
+ }
+ return { changed, unmatched };
+}
+
async function main() {
const argv = process.argv.slice(2);
const manifestPath = argv.find((a) => !a.startsWith("--"));
@@ -136,6 +320,26 @@ async function main() {
const loadCues = (videoId, channelSlug, hints) =>
cues.load(channelSlug, videoId, hints).then((r) => r.cues);
+ // ---- the cut pass, which is a different question ----
+ if (argv.includes("--cut-to-quote")) {
+ const r = await cutPass(manifest, {
+ loadCues,
+ slug,
+ opts: { ...opts, minMatch: num("--min-match", 0.6) },
+ write: argv.includes("--write"),
+ force: argv.includes("--force-cut"),
+ });
+ if (argv.includes("--write")) {
+ await writeFile(manifestPath, JSON.stringify(manifest, null, 2) + "\n", "utf8");
+ console.log(`\nwrote ${manifestPath} (${r.changed} cut(s) set, ${r.unmatched} unmatched)`);
+ } else {
+ console.log(
+ `\ndry run — ${r.changed} cut(s) would be set, ${r.unmatched} unmatched; pass --write to apply`,
+ );
+ }
+ return;
+ }
+
let changed = 0;
for (const e of manifest.timeline) {
if (e.type !== "clip") continue;