import path from "node:path"; import { existsSync } from "node:fs"; import { readdir, stat } from "node:fs/promises"; import { dataFile, stateFile } from "./paths"; import { readJson } from "./state"; import { primeWords, sourceVideo } from "./clips"; import { archiveMomentUrl } from "./archive"; import type { AsrWord } from "./types"; import { CONTEXT_WORDS, inChunkOverlap, isFillerQuery, normTerms, previewWindow, hitId, type PhraseHit, type PhraseQuery, type PhraseResult, } from "./phrase-types"; // The shapes and constants live in lib/phrase-types.ts, which has no server // imports, so PhraseConsole.tsx can take them as VALUES without dragging // node:fs into the browser bundle. Re-exported so the server has one import. export * from "./phrase-types"; // --------------------------------------------------------------------------- // THE INDEX: a space-delimited blob per video, not a parallel string[]. // // `" the thing is "` makes strict adjacency a native indexOf, which is the // entire argument. Measured against this corpus (300 files, 623,079 tokens): // // naive scan 527ms cold load, 50-60ms per query, 77 MB heap // this ~712ms to build, 6ms for `the`, ~105 MB heap // // and it costs LESS memory than an array of normalised strings would, because // one long string has one header instead of 623,079 of them. // // Per video: the blob, an Int32Array of each term's char offset, and an // Int32Array mapping term index -> word index. The second is built // unconditionally -- it is 2.5 MB across the whole corpus, and one code path // that always works beats a null fast-path that is right 99.997% of the time. // Only the 18 tokens with an internal space make it differ from the identity, // and it is consulted ONLY on a hit, so they cost nothing in the hot path. // --------------------------------------------------------------------------- type VideoIndex = { video: string; blob: string; /** Char offset of each term within `blob`. Ascending by construction. */ off: Int32Array; /** Term index -> word index. Not the identity: `L.A.` is one word, two terms. */ termWord: Int32Array; /** * The SAME array lib/clips.ts holds, not a copy -- primeWords() put it there * as this index was built, so a re-transcribed episode cannot leave * /browse/find and a clip card disagreeing about what was said. */ words: AsrWord[]; mtimeMs: number; size: number; }; type Corpus = { videos: VideoIndex[]; byVideo: Map; words: number; builtAt: number; }; let corpus: Corpus | null = null; let building: Promise | null = null; const asrDir = () => dataFile("asr"); function buildVideoIndex( video: string, words: AsrWord[], mtimeMs: number, size: number, ): VideoIndex { const parts: string[] = []; const off: number[] = []; const termWord: number[] = []; // The blob opens with a space, so the first term is at offset 1 and a needle // that also opens with a space can match it. let pos = 1; for (let wi = 0; wi < words.length; wi += 1) { for (const t of normTerms(words[wi].w)) { off.push(pos); termWord.push(wi); parts.push(t); pos += t.length + 1; } } return { video, blob: ` ${parts.join(" ")} `, off: Int32Array.from(off), termWord: Int32Array.from(termWord), words, mtimeMs, size, }; } // INVALIDATION BY (mtimeMs, size), PER FILE. // // A readdir plus 300 stats measures 3-4ms -- 0.06% of even a 6ms query once the // stats are parallel -- so exact invalidation is free. A TTL would be worse in // both directions: it would cost a 712ms rebuild every minute in a tool nobody // is querying, and it would still serve a stale answer for up to that minute // after a re-transcription. Only changed files are re-read. async function build(): Promise { const dir = asrDir(); let files: string[] = []; try { files = (await readdir(dir)).filter((f) => f.endsWith(".json")); } catch { files = []; } files.sort(); const prev = corpus?.byVideo; const built = await Promise.all( files.map(async (f): Promise => { const video = f.slice(0, -".json".length); const full = path.join(dir, f); let st; try { st = await stat(full); } catch { return null; } const old = prev?.get(video); if (old && old.mtimeMs === st.mtimeMs && old.size === st.size) return old; const raw = await readJson<{ words: AsrWord[] }>(full, { words: [] }); // primeWords sorts and publishes into lib/clips.ts's cache, so this parse // serves the console AND every clip card. One read, two callers. const words = primeWords(video, raw.words ?? []); return buildVideoIndex(video, words, st.mtimeMs, st.size); }), ); const videos = built.filter((v): v is VideoIndex => v !== null); corpus = { videos, byVideo: new Map(videos.map((v) => [v.video, v])), words: videos.reduce((n, v) => n + v.words.length, 0), builtAt: Date.now(), }; return corpus; } function ensureCorpus(): Promise { if (!building) { building = build().finally(() => { building = null; }); } return building; } /** What is in memory right now. Never builds -- this is for a status line. */ export function indexState(): { videos: number; words: number; builtAt: number | null } { return corpus ? { videos: corpus.videos.length, words: corpus.words, builtAt: corpus.builtAt } : { videos: 0, words: 0, builtAt: null }; } /** * The episode ids, from a bare readdir. NEVER builds the index. * * This is what a page with no query is allowed to do -- the same restraint * listSongs() shows by never probing. Building 105 MB of index to draw an empty * search box would be the whole cost of the feature paid for nothing. */ export async function episodeIds(): Promise { try { return (await readdir(asrDir())) .filter((f) => f.endsWith(".json")) .map((f) => f.slice(0, -".json".length)) .sort(); } catch { return []; } } // --------------------------------------------------------------------------- // Flagged sources. // // suspect-sources.json is keyed `