Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 168885368d627353545ee7017a766f46d9a96ac3
parent 07f1ce9f93ee3b23ac4edb793b69918818e84117
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue, 18 Aug 2026 10:02:33 -0400

Index the ASR corpus by phrase

The retrieval core for making word edits out of the archive. asr/ carries
{w, start, end, conf} for 623,079 tokens across 300 episodes, and media/,
wav48/ and the ASR share one timeline -- so a word's timestamp addresses
picture AND sound with no alignment work. What was missing was a way to ask
for a word.

A space-delimited blob per video, not a parallel string[]. `" the thing is "`
makes strict adjacency a native indexOf; measured against this corpus that is
6ms of scan for `the` (23,081 matches) where a naive walk is 50-60ms, and it
costs LESS memory than an array of normalised strings. Invalidation is by
(mtimeMs, size) per file, because a readdir plus 300 parallel stats is 3-4ms
and a TTL would trade a stale minute for a 712ms rebuild nobody asked for.
Cold build measures 757ms; a warm query end to end, including the stats, the
title join and expanding 50 rows, is 7-25ms.

Four corpus traps shape it, and each is measured rather than guessed:

  * Parakeet timestamps a word's START LATE by 25-290ms, so a preview seeking
    to word.start clips the first phoneme. LEAD_PAD is a LISTENING pad, not a
    cut boundary -- clipWindow still owns where an edit goes.
  * The 240s/3s chunked transcription leaves twins the merge did not catch:
    3,472 adjacent same-word pairs within 1s, of which 2,248 sit in a
    `start % 237 < 3.5` band. That band is an exact discriminator, so the
    other 1,224 -- genuine stutters, "Very, very" -- survive and are marked.
    Folding is stated as `collapsed`, never silent.
  * 18 tokens contain an internal space. normTerms returns TERMS, PLURAL, so
    `just 'cause`, `buy 'em` and `L.A.` are findable instead of silently
    unreachable; verified, all three now return their hit as one word.
  * The ASR barely transcribes fillers -- `um` is 2 occurrences, `uh` is 5 --
    which is why the palette was mined acoustically. A filler query says so.

No default confidence floor: only 1.1% of tokens fall below 0.4, and for a
word edit a low-confidence token is often the interesting one. Show the
reading, offer the filter, and count what the filter removed.

primeWords() exists so the one parse serves this index and every clip card:
without it a re-transcribed episode leaves /browse/find and a clip card
quoting different words for the same second, with nothing to say which is
stale. AsrWord gains an optional conf -- the TYPE dropped it, not the data,
and absence means unknown, never 1.0. suspectVideos() is lifted out of the
evidence route so "which episodes are flagged" has one answer.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mumtool/app/api/clip/[key]/evidence/route.ts | 14+++++---------
Mumtool/lib/clips.ts | 20+++++++++++++++++---
Aumtool/lib/phrase-types.ts | 279+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aumtool/lib/phrases.ts | 456+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mumtool/lib/types.ts | 8+++++++-
5 files changed, 764 insertions(+), 13 deletions(-)

diff --git a/umtool/app/api/clip/[key]/evidence/route.ts b/umtool/app/api/clip/[key]/evidence/route.ts @@ -1,8 +1,7 @@ import { existsSync } from "node:fs"; import { sourcePitch, F0_GATE } from "@/lib/corpus"; import { sourceWav, wordsFor } from "@/lib/clips"; -import { readJson } from "@/lib/state"; -import { stateFile } from "@/lib/paths"; +import { suspectVideos } from "@/lib/phrases"; import { readWavWindow } from "@/lib/wav"; import { spectralFlatness } from "../../../../../song/flatness.mjs"; @@ -64,13 +63,10 @@ export async function GET(request: Request, ctx: { params: Promise<{ key: string .map((w) => w.w); const source = await sourcePitch(video); - const suspectList = await readJson<unknown>(stateFile("suspect-sources.json"), []); - const suspectKeys = new Set( - (Array.isArray(suspectList) ? suspectList : []) - .map((s) => (typeof s === "string" ? s : (s as { k?: string })?.k)) - .filter(Boolean) as string[], - ); - const suspectSource = [...suspectKeys].some((k) => k.startsWith(`${video}@`)); + // The same reduction /browse/find applies, from the same module. It used to + // be inline here; two copies of "which episodes are flagged" is two answers + // waiting to drift. + const suspectSource = (await suspectVideos()).has(video); return Response.json( { diff --git a/umtool/lib/clips.ts b/umtool/lib/clips.ts @@ -72,9 +72,23 @@ export async function wordsFor(video: string): Promise<AsrWord[]> { const hit = asrCache.get(video); if (hit) return hit; const w = await readJson<{ words: AsrWord[] }>(dataFile("asr", `${video}.json`), { words: [] }); - // MUST be time-ordered: clipWindow scans for the nearest word each side and - // stops at the first hit. - const sorted = (w.words ?? []).slice().sort((a, b) => a.start - b.start); + return primeWords(video, w.words ?? []); +} + +/** + * Publish a freshly-read word list into the shared cache, and hand it back. + * + * lib/phrases.ts re-reads an episode's ASR whenever its (mtime, size) changes, + * which is a read this cache would otherwise miss -- leaving /browse/find and a + * clip card quoting different words for the same second, with nothing to say + * which was stale. One parse, one cache, both surfaces. + * + * Sorts, because it is the SAME invariant wordsFor has always guaranteed: + * clipWindow scans outward for the nearest word each side and stops at the + * first hit, so an out-of-order list silently returns the wrong neighbour. + */ +export function primeWords(video: string, words: AsrWord[]): AsrWord[] { + const sorted = words.slice().sort((a, b) => a.start - b.start); asrCache.set(video, sorted); return sorted; } diff --git a/umtool/lib/phrase-types.ts b/umtool/lib/phrase-types.ts @@ -0,0 +1,279 @@ +// The phrase console's shapes, constants and its ONE normaliser, with no server +// imports. +// +// This file exists for the reason lib/note-types.ts does, and the break it +// prevents is the same one: PhraseConsole.tsx is a client component and needs +// LEAD_PAD, PAGE_SIZE and normTerms() as VALUES at runtime. A value import is +// not erased, so taking them from lib/phrases.ts would drag node:fs/promises +// into the browser chunk and the build would refuse. `tsc --noEmit` cannot see +// that; only the bundler can. +// +// lib/phrases.ts re-exports everything here, so the server side has one import. + +// --------------------------------------------------------------------------- +// The corpus constants. Every one of these is a MEASURED property of the ASR, +// not a preference, so each carries the measurement that set it. +// --------------------------------------------------------------------------- + +/** + * The chunk stride the transcription ran at: parakeet-chunked.mjs uses a 240s + * chunk with 3s of overlap, so a new chunk starts every 237s. + * + * This matters because the merge only drops a word when + * `|dstart| < 0.12 && x.w === w.w` -- casing and punctuation drift let twins + * through. There are 3,472 adjacent same-word pairs within 1s in the corpus, + * and 2,248 of them (65%) sit in a `start % 237 < 3.5` band. That band is an + * exact discriminator rather than a heuristic: inside it a repeat is an + * artefact of the seam, outside it the other 1,224 are genuine stutters + * ("Very, very") and must survive. + */ +export const CHUNK_STEP = 237; +export const CHUNK_OVERLAP = 3.5; + +/** + * Lead-in for a preview, in seconds. + * + * Parakeet timestamps a word's START LATE by 25-290ms -- see the note in + * song/clipwindow.mjs, where the same fact makes the window guards asymmetric. + * A preview that seeks to `word.start` clips the first phoneme, which reads as + * the archive being wrong rather than the timestamp. + * + * This is a LISTENING pad, not a cut boundary. Nothing here decides where an + * edit goes; clipWindow() still owns that, and it measures rather than pads. + */ +export const LEAD_PAD = 0.35; +export const TAIL_PAD = 0.35; + +/** The in-context audition: enough either side to hear what the line was. */ +export const WIDE_PAD = 1.5; + +/** Words of context shown each side of a match. */ +export const CONTEXT_WORDS = 7; + +export const PAGE_SIZE = 50; +export const MAX_PER_PAGE = 200; + +/** + * Terms in one query. `the` alone is 23,081 hits; eight terms is far past any + * phrase anyone means, and the cap is here so a pathological needle cannot be + * built by pasting a paragraph in. + */ +export const MAX_TERMS = 8; + +// --------------------------------------------------------------------------- +// The shapes. +// --------------------------------------------------------------------------- + +export type PhraseOrder = "time" | "conf"; + +export type PhraseQuery = { + /** What was typed, verbatim, so the box can be refilled with it. */ + q: string; + /** What is actually searched for. normTerms(q). */ + terms: string[]; + /** One episode, or the whole corpus. */ + video: string | null; + /** Minimum confidence, or null for no filter -- which is the DEFAULT. */ + minConf: number | null; + /** Drop hits from episodes flagged in suspect-sources.json. */ + clean: boolean; + /** + * Fold chunk twins. `dupes=1` (the default) folds them and reports the count + * as `collapsed`; `dupes=0` shows every raw match. + * + * Read the flag as "the dupe-folding is on", not as "dupes are shown" -- it + * is the folding that is being switched, and `total` goes DOWN when it is 1. + */ + dupes: boolean; + order: PhraseOrder; + /** 1-based. */ + page: number; + per: number; +}; + +export type PhraseHit = { + /** + * `<video>#<wordIndex>`. NEVER a timestamp: about one pair per 100k words + * shares a `start`, so a time-keyed id would collide and a starred pick would + * silently address the wrong word. + */ + id: string; + video: string; + /** Falls back to the bare id: titles.json covers 298 of 300 episodes. */ + title: string; + date: string; + /** Word index of the first matched word, and how many words the match spans. */ + i: number; + n: number; + start: number; + end: number; + /** The matched words AS WRITTEN, never normalised. */ + text: string; + before: string; + after: string; + /** null means UNKNOWN, never 1.0. The type dropped conf; the data has it. */ + confMin: number | null; + confMean: number | null; + /** The preview window, lead-in already applied. */ + from: number; + to: number; + /** This EPISODE is in suspect-sources.json. Per-episode, never per-moment. */ + suspect: boolean; + /** Chunk twins folded into this hit. */ + dupes: number; + /** A real repeat: within 0.6s of its predecessor and outside any seam band. */ + stutter: boolean; + hasVideo: boolean; + archive: string | null; +}; + +export type PhraseResult = { + query: PhraseQuery; + /** The page slice only. */ + hits: PhraseHit[]; + /** Every match under these filters, uncapped. */ + total: number; + /** Twins folded. Stated, never silent. */ + collapsed: number; + /** Matches removed by the confidence and flagged-source filters. */ + suppressed: number; + videos: number; + ms: number; + corpus: { videos: number; words: number; builtAt: number | null; heapMb: number }; + notes: string[]; +}; + +// --------------------------------------------------------------------------- +// The normaliser. ONE function, applied identically to the query and to the +// corpus -- if they ever diverge, a word becomes unfindable and nothing says so. +// --------------------------------------------------------------------------- + +const CURLY = /[\u2018\u2019\u02bc\u201b]/g; +const MARKS = /[\u0300-\u036f]/g; +const NON_ASCII = /[^\x00-\x7f]/; + +/** + * Split a string into search terms. + * + * TERMS, PLURAL, and that is the whole point. 18 tokens in the corpus contain + * an internal space -- `just 'cause`, `buy 'em`, `L.A.` -- so any + * one-token-one-term normaliser makes them permanently unfindable, silently. + * Here `L.A.` becomes `["l","a"]` on both sides of the comparison, so typing + * either "L.A." or "l a" finds it. + * + * Zero tokens in the corpus are hyphenated (`uh-huh` and `mm-hmm` are human + * labels from lib/vocab.ts, never ASR output), so a hyphen is a query-side term + * separator and nothing more. + */ +export function normTerms(s: string): string[] { + let t = s.toLowerCase().replace(CURLY, "'"); + // Fold diacritics so "café" and "cafe" are the same term. Guarded because it + // runs 623,079 times on a cold build and almost never has anything to do. + if (NON_ASCII.test(t)) t = t.normalize("NFD").replace(MARKS, ""); + const out: string[] = []; + for (const part of t.split(/[^a-z0-9']+/)) { + const term = part.replace(/^'+/, "").replace(/'+$/, ""); + if (term) out.push(term); + } + return out; +} + +/** A hit's identity. The word index, never the time. */ +export const hitId = (video: string, i: number): string => `${video}#${i}`; + +/** Is this second inside a chunk seam, where a same-word repeat is an artefact? */ +export const inChunkOverlap = (start: number): boolean => + start % CHUNK_STEP < CHUNK_OVERLAP; + +/** + * The window to preview. `from` is always at least LEAD_PAD before the word, so + * the late timestamp cannot clip the onset, and it is clamped at 0 -- a hit at + * 0.28s would otherwise ask for -0.07. + */ +export function previewWindow( + hit: { start: number; end: number }, + wide = false, +): { from: number; to: number } { + const lead = wide ? WIDE_PAD : LEAD_PAD; + const tail = wide ? WIDE_PAD : TAIL_PAD; + return { + from: +Math.max(0, hit.start - lead).toFixed(3), + to: +(hit.end + tail).toFixed(3), + }; +} + +// --------------------------------------------------------------------------- +// Trap 1, and it has to be said out loud. +// +// The ASR barely transcribes fillers: `um` is 2 occurrences in 623,079 tokens, +// `uh` is 5, `huh` 9, `hmm` 2, `erm` 1. That is exactly WHY the um palette was +// mined acoustically into cand2/ rather than read out of the transcript. +// +// The first thing anyone types into this box will be "um", and it will return +// two rows. Without a note saying so, the console reads as broken on contact. +// --------------------------------------------------------------------------- + +const FILLERS = new Set([ + "um", "umm", "uh", "uhh", "er", "erm", "hmm", "hm", "mm", "mmm", "huh", "ah", "eh", +]); + +export const isFillerQuery = (terms: string[]): boolean => + terms.length > 0 && terms.every((t) => FILLERS.has(t)); + +// --------------------------------------------------------------------------- +// One parser, so the page and the API can never disagree about what a URL +// means. Errors are RETURNED rather than thrown: the page renders them as notes +// and still draws the box, the API turns them into a 400. +// --------------------------------------------------------------------------- + +export type PhraseQueryError = { field: string; message: string }; + +const num = (v: string | null): number | null => { + if (v == null || v === "") return null; + const n = Number(v); + return Number.isFinite(n) ? n : null; +}; + +export function parsePhraseQuery(sp: URLSearchParams): { + query: PhraseQuery; + errors: PhraseQueryError[]; +} { + const errors: PhraseQueryError[] = []; + const q = (sp.get("q") ?? "").slice(0, 400); + let terms = normTerms(q); + + if (q.trim() && terms.length === 0) { + errors.push({ field: "q", message: "nothing searchable in that -- punctuation only" }); + } + if (terms.length > MAX_TERMS) { + errors.push({ field: "q", message: `at most ${MAX_TERMS} terms; that is ${terms.length}` }); + terms = terms.slice(0, MAX_TERMS); + } + + const rawPer = num(sp.get("per")); + if (rawPer != null && rawPer > MAX_PER_PAGE) { + errors.push({ field: "per", message: `per is capped at ${MAX_PER_PAGE}` }); + } + const per = Math.max(1, Math.min(MAX_PER_PAGE, Math.round(rawPer ?? PAGE_SIZE))); + + const minConf = num(sp.get("conf")); + const order = sp.get("order") === "conf" ? "conf" : "time"; + + return { + query: { + q, + terms, + video: sp.get("video") || null, + // No DEFAULT confidence floor. Only 1.1% of tokens fall below 0.4, and for + // a word edit a low-confidence token is often the interesting one -- the + // machine heard something it could not place. Show conf, offer the filter. + minConf: minConf != null && minConf > 0 ? minConf : null, + clean: sp.get("clean") === "1", + dupes: sp.get("dupes") !== "0", + order, + page: Math.max(1, Math.round(num(sp.get("page")) ?? 1)), + per, + }, + errors, + }; +} diff --git a/umtool/lib/phrases.ts b/umtool/lib/phrases.ts @@ -0,0 +1,456 @@ +import path from "node:path"; +import { existsSync } from "node:fs"; +import { readdir, stat } from "node:fs/promises"; +import { dataFile, stateFile } from "./paths"; +import { readJson } from "./state"; +import { primeWords, sourceVideo } from "./clips"; +import { archiveMomentUrl } from "./archive"; +import type { AsrWord } from "./types"; +import { + CONTEXT_WORDS, + inChunkOverlap, + isFillerQuery, + normTerms, + previewWindow, + hitId, + type PhraseHit, + type PhraseQuery, + type PhraseResult, +} from "./phrase-types"; + +// The shapes and constants live in lib/phrase-types.ts, which has no server +// imports, so PhraseConsole.tsx can take them as VALUES without dragging +// node:fs into the browser bundle. Re-exported so the server has one import. +export * from "./phrase-types"; + +// --------------------------------------------------------------------------- +// THE INDEX: a space-delimited blob per video, not a parallel string[]. +// +// `" the thing is "` makes strict adjacency a native indexOf, which is the +// entire argument. Measured against this corpus (300 files, 623,079 tokens): +// +// naive scan 527ms cold load, 50-60ms per query, 77 MB heap +// this ~712ms to build, 6ms for `the`, ~105 MB heap +// +// and it costs LESS memory than an array of normalised strings would, because +// one long string has one header instead of 623,079 of them. +// +// Per video: the blob, an Int32Array of each term's char offset, and an +// Int32Array mapping term index -> word index. The second is built +// unconditionally -- it is 2.5 MB across the whole corpus, and one code path +// that always works beats a null fast-path that is right 99.997% of the time. +// Only the 18 tokens with an internal space make it differ from the identity, +// and it is consulted ONLY on a hit, so they cost nothing in the hot path. +// --------------------------------------------------------------------------- + +type VideoIndex = { + video: string; + blob: string; + /** Char offset of each term within `blob`. Ascending by construction. */ + off: Int32Array; + /** Term index -> word index. Not the identity: `L.A.` is one word, two terms. */ + termWord: Int32Array; + /** + * The SAME array lib/clips.ts holds, not a copy -- primeWords() put it there + * as this index was built, so a re-transcribed episode cannot leave + * /browse/find and a clip card disagreeing about what was said. + */ + words: AsrWord[]; + mtimeMs: number; + size: number; +}; + +type Corpus = { + videos: VideoIndex[]; + byVideo: Map<string, VideoIndex>; + words: number; + builtAt: number; +}; + +let corpus: Corpus | null = null; +let building: Promise<Corpus> | null = null; + +const asrDir = () => dataFile("asr"); + +function buildVideoIndex( + video: string, + words: AsrWord[], + mtimeMs: number, + size: number, +): VideoIndex { + const parts: string[] = []; + const off: number[] = []; + const termWord: number[] = []; + // The blob opens with a space, so the first term is at offset 1 and a needle + // that also opens with a space can match it. + let pos = 1; + for (let wi = 0; wi < words.length; wi += 1) { + for (const t of normTerms(words[wi].w)) { + off.push(pos); + termWord.push(wi); + parts.push(t); + pos += t.length + 1; + } + } + return { + video, + blob: ` ${parts.join(" ")} `, + off: Int32Array.from(off), + termWord: Int32Array.from(termWord), + words, + mtimeMs, + size, + }; +} + +// INVALIDATION BY (mtimeMs, size), PER FILE. +// +// A readdir plus 300 stats measures 3-4ms -- 0.06% of even a 6ms query once the +// stats are parallel -- so exact invalidation is free. A TTL would be worse in +// both directions: it would cost a 712ms rebuild every minute in a tool nobody +// is querying, and it would still serve a stale answer for up to that minute +// after a re-transcription. Only changed files are re-read. +async function build(): Promise<Corpus> { + const dir = asrDir(); + let files: string[] = []; + try { + files = (await readdir(dir)).filter((f) => f.endsWith(".json")); + } catch { + files = []; + } + files.sort(); + + const prev = corpus?.byVideo; + const built = await Promise.all( + files.map(async (f): Promise<VideoIndex | null> => { + const video = f.slice(0, -".json".length); + const full = path.join(dir, f); + let st; + try { + st = await stat(full); + } catch { + return null; + } + const old = prev?.get(video); + if (old && old.mtimeMs === st.mtimeMs && old.size === st.size) return old; + const raw = await readJson<{ words: AsrWord[] }>(full, { words: [] }); + // primeWords sorts and publishes into lib/clips.ts's cache, so this parse + // serves the console AND every clip card. One read, two callers. + const words = primeWords(video, raw.words ?? []); + return buildVideoIndex(video, words, st.mtimeMs, st.size); + }), + ); + + const videos = built.filter((v): v is VideoIndex => v !== null); + corpus = { + videos, + byVideo: new Map(videos.map((v) => [v.video, v])), + words: videos.reduce((n, v) => n + v.words.length, 0), + builtAt: Date.now(), + }; + return corpus; +} + +function ensureCorpus(): Promise<Corpus> { + if (!building) { + building = build().finally(() => { + building = null; + }); + } + return building; +} + +/** What is in memory right now. Never builds -- this is for a status line. */ +export function indexState(): { videos: number; words: number; builtAt: number | null } { + return corpus + ? { videos: corpus.videos.length, words: corpus.words, builtAt: corpus.builtAt } + : { videos: 0, words: 0, builtAt: null }; +} + +/** + * The episode ids, from a bare readdir. NEVER builds the index. + * + * This is what a page with no query is allowed to do -- the same restraint + * listSongs() shows by never probing. Building 105 MB of index to draw an empty + * search box would be the whole cost of the feature paid for nothing. + */ +export async function episodeIds(): Promise<string[]> { + try { + return (await readdir(asrDir())) + .filter((f) => f.endsWith(".json")) + .map((f) => f.slice(0, -".json".length)) + .sort(); + } catch { + return []; + } +} + +// --------------------------------------------------------------------------- +// Flagged sources. +// +// suspect-sources.json is keyed `<video>@<start>`, and reducing it to a set of +// VIDEOS is exactly what app/api/clip/[key]/evidence already did inline. One +// reduction, one place, so the two surfaces cannot disagree about which +// episodes carry a second speaker. +// +// The label is "flagged source", not "flagged moment". The judgement behind the +// data is per-episode in effect -- a clip somewhere in it was called "not Jer" +// -- and claiming per-moment precision for it would be worse than no flag. +// --------------------------------------------------------------------------- + +export async function suspectVideos(): Promise<Set<string>> { + const list = await readJson<unknown>(stateFile("suspect-sources.json"), []); + const out = new Set<string>(); + for (const s of Array.isArray(list) ? list : []) { + const k = typeof s === "string" ? s : (s as { k?: string })?.k; + if (!k) continue; + const at = k.lastIndexOf("@"); + out.add(at > 0 ? k.slice(0, at) : k); + } + return out; +} + +// --------------------------------------------------------------------------- +// The scan. +// +// A RawHit is deliberately tiny: `the` is 23,081 matches, and 23,081 of these +// is a few hundred KB and about 3ms. Only the PAGE SLICE is ever expanded into +// a PhraseHit -- that is where the title join, the existsSync, the archive URL +// and the context line happen, and doing any of them 23,081 times to show 50 +// rows would be the whole cost of the query. +// --------------------------------------------------------------------------- + +type RawHit = { + vi: number; + i: number; + n: number; + start: number; + cmin: number | null; + dupes: number; + stutter: boolean; +}; + +function scan(idx: VideoIndex, vi: number, needle: string, lastTerm: number, out: RawHit[]): void { + const { blob, off, termWord, words } = idx; + // A monotonic cursor rather than a binary search: matches come out of + // indexOf in ascending order and `off` is ascending by construction, so the + // whole scan walks the offsets once. + let cur = 0; + for (let k = blob.indexOf(needle); k >= 0; k = blob.indexOf(needle, k + 1)) { + const at = k + 1; + while (cur < off.length && off[cur] < at) cur += 1; + if (cur >= off.length || off[cur] !== at) continue; + const t1 = cur + lastTerm; + if (t1 >= termWord.length) break; + const w0 = termWord[cur]; + const w1 = termWord[t1]; + const first = words[w0]; + if (!first) continue; + let cmin: number | null = null; + for (let x = w0; x <= w1; x += 1) { + const c = words[x]?.conf; + if (typeof c === "number") cmin = cmin === null ? c : Math.min(cmin, c); + } + out.push({ vi, i: w0, n: w1 - w0 + 1, start: first.start, cmin, dupes: 0, stutter: false }); + } +} + +// CHUNK TWINS, and the discriminator is exact rather than a guess. +// +// Fold a hit into its predecessor when it is the same video, the same run, +// within 0.6s, AND the later start falls in a seam band. Outside the band the +// second hit SURVIVES and is flagged `stutter`: 1,224 of the 3,472 adjacent +// same-word pairs are genuine repeats ("Very, very"), and a supercut wants +// those. `collapsed` is reported, never silently applied. +function dedupe(hits: RawHit[], fold: boolean): { kept: RawHit[]; collapsed: number } { + const kept: RawHit[] = []; + let collapsed = 0; + let prev: RawHit | null = null; + for (const h of hits) { + const near = prev !== null && prev.vi === h.vi && h.start - prev.start < 0.6; + if (near && inChunkOverlap(h.start)) { + if (fold) { + (prev as RawHit).dupes += 1; + collapsed += 1; + continue; + } + } else if (near) { + h.stutter = true; + } + kept.push(h); + prev = h; + } + return { kept, collapsed }; +} + +const hasVideoCache = new Map<string, boolean>(); +function hasVideo(video: string): boolean { + const hit = hasVideoCache.get(video); + if (hit !== undefined) return hit; + const ok = existsSync(sourceVideo(video)); + hasVideoCache.set(video, ok); + return ok; +} + +function expand( + h: RawHit, + idx: VideoIndex, + ctx: { titles: Record<string, string>; dates: Record<string, string>; suspect: Set<string> }, +): PhraseHit { + const { words, video } = idx; + const last = words[h.i + h.n - 1] ?? words[h.i]; + const start = words[h.i].start; + const end = last.end; + + let sum = 0; + let seen = 0; + let cmin: number | null = null; + for (let x = h.i; x < h.i + h.n; x += 1) { + const c = words[x]?.conf; + if (typeof c === "number") { + sum += c; + seen += 1; + cmin = cmin === null ? c : Math.min(cmin, c); + } + } + + const text = (ws: AsrWord[]) => ws.map((w) => w.w).join(" "); + const { from, to } = previewWindow({ start, end }); + + return { + id: hitId(video, h.i), + video, + // titles.json covers 298 of 300. The bare id is a worse label but a true + // one, and an episode with no title must still be findable. + title: ctx.titles[video] ?? video, + date: ctx.dates[video] ?? "", + i: h.i, + n: h.n, + start, + end, + text: text(words.slice(h.i, h.i + h.n)), + before: text(words.slice(Math.max(0, h.i - CONTEXT_WORDS), h.i)), + after: text(words.slice(h.i + h.n, h.i + h.n + CONTEXT_WORDS)), + confMin: cmin === null ? null : +cmin.toFixed(4), + confMean: seen ? +(sum / seen).toFixed(4) : null, + from, + to, + suspect: ctx.suspect.has(video), + dupes: h.dupes, + stutter: h.stutter, + hasVideo: hasVideo(video), + archive: archiveMomentUrl(video, start), + }; +} + +const MB = 1 << 20; + +const empty = (query: PhraseQuery, notes: string[] = []): PhraseResult => ({ + query, + hits: [], + total: 0, + collapsed: 0, + suppressed: 0, + videos: 0, + ms: 0, + corpus: { ...indexState(), heapMb: Math.round(process.memoryUsage().heapUsed / MB) }, + notes, +}); + +/** Is this a real episode? A bare readdir, so it costs nothing to ask. */ +export async function isEpisode(video: string): Promise<boolean> { + return (await episodeIds()).includes(video); +} + +export async function searchPhrases(query: PhraseQuery): Promise<PhraseResult> { + const t0 = Date.now(); + if (query.terms.length === 0) return empty(query); + + const [idx, titles, dates, suspect] = await Promise.all([ + ensureCorpus(), + readJson<Record<string, string>>(stateFile("titles.json"), {}), + readJson<Record<string, string>>(stateFile("dates.json"), {}), + suspectVideos(), + ]); + + const needle = ` ${query.terms.join(" ")} `; + const lastTerm = query.terms.length - 1; + const scope = query.video + ? idx.videos.filter((v) => v.video === query.video) + : idx.videos; + + const raw: RawHit[] = []; + let suppressed = 0; + for (let vi = 0; vi < scope.length; vi += 1) { + const v = scope[vi]; + // Filters apply AS the video is scanned, so `total` is the total of what is + // shown and `suppressed` says what went. A count that included rows the + // filter removed would make every filter look broken. + if (query.clean && suspect.has(v.video)) { + const dropped: RawHit[] = []; + scan(v, vi, needle, lastTerm, dropped); + suppressed += dropped.length; + continue; + } + const mine: RawHit[] = []; + scan(v, vi, needle, lastTerm, mine); + for (const h of mine) { + // An unknown confidence cannot be asserted to clear a floor, so a floor + // drops it. It is never read as 1.0. + if (query.minConf !== null && (h.cmin === null || h.cmin < query.minConf)) { + suppressed += 1; + continue; + } + raw.push(h); + } + } + + const { kept, collapsed } = dedupe(raw, query.dupes); + + if (query.order === "conf") { + kept.sort((a, b) => { + const ac = a.cmin === null ? 2 : a.cmin; + const bc = b.cmin === null ? 2 : b.cmin; + return ac - bc || a.vi - b.vi || a.start - b.start; + }); + } + + const from = (query.page - 1) * query.per; + const slice = kept.slice(from, from + query.per); + const ctx = { titles, dates, suspect }; + const hits = slice.map((h) => expand(h, scope[h.vi], ctx)); + + const notes: string[] = []; + if (isFillerQuery(query.terms)) { + notes.push( + "The ASR barely transcribes fillers: `um` appears 2 times in 623,079 tokens, " + + "`uh` 5, `huh` 9. That is exactly why the um palette was mined ACOUSTICALLY " + + "into cand2/ rather than read out of the transcript — so a near-empty result " + + "here is the transcript being honest, not the search being broken.", + ); + } + if (collapsed) { + notes.push( + `${collapsed} chunk ${collapsed === 1 ? "twin" : "twins"} folded — repeats inside a ` + + "240s/3s transcription seam. Genuine stutters are kept and marked.", + ); + } + if (suppressed) notes.push(`${suppressed} hidden by the filters.`); + + return { + query, + hits, + total: kept.length, + collapsed, + suppressed, + videos: new Set(kept.map((h) => h.vi)).size, + ms: Date.now() - t0, + corpus: { + videos: idx.videos.length, + words: idx.words, + builtAt: idx.builtAt, + heapMb: Math.round(process.memoryUsage().heapUsed / MB), + }, + notes, + }; +} diff --git a/umtool/lib/types.ts b/umtool/lib/types.ts @@ -26,7 +26,13 @@ export type Candidate = { sortedFrom?: string; }; -export type AsrWord = { start: number; end: number; w: string }; +/** + * One ASR token. `conf` is OPTIONAL because this type dropped it, not because + * the data lacks it -- every word in asr/ carries a parakeet confidence, 70% of + * them at 0.9 or better and only 1.1% below 0.4. Consumers must read an absent + * conf as UNKNOWN and never as 1.0: a missing reading is not a perfect one. + */ +export type AsrWord = { start: number; end: number; w: string; conf?: number }; /** What the deck needs to render a queue. No wav is opened to build one. */ export type ClipStub = {