// FlexSearch index web worker. Builds and queries a per-channel full-text index // entirely off the main thread, so indexing hundreds of MB of transcript text // never blocks the UI. Bundled by Turbopack under output:"export" (proven to // emit into static out/) via `new Worker(new URL('./searchIndex.worker.ts', // import.meta.url), {type:'module'})` in searchIndexWorkerClient. // // The worker fetches the same /transcripts//{manifest,page-NNNN}.json // shards the scanner uses — so a service-worker-cached (offline) channel indexes // and searches with no network. import { Index, Charset } from "flexsearch"; import type { IndexHit, WorkerRequest, WorkerResponse, } from "./searchIndexWorkerProtocol"; // Minimal worker-global surface (avoids depending on the webworker tsconfig lib). const ctx = globalThis as unknown as { postMessage(msg: WorkerResponse): void; addEventListener( type: "message", cb: (e: MessageEvent) => void, ): void; }; type ChannelIndex = { idx: Index; // docId → the hit it resolves to (FlexSearch returns ids, not the source). meta: IndexHit[]; }; const channels = new Map(); function post(msg: WorkerResponse): void { ctx.postMessage(msg); } function pageFileName(index: number): string { return `page-${String(index).padStart(4, "0")}.json`; } function mkIndex(): Index { // forward tokenizer → prefix matches (search-as-you-type); LatinBalance folds // case/diacritics for recall without the heavier phonetic passes. return new Index({ tokenize: "forward", encoder: Charset.LatinBalance, resolution: 9, } as never); } type RawCue = { start?: number; text?: unknown }; type RawEntry = { slug?: unknown; cues?: unknown; altTracks?: unknown }; async function build(reqId: number, slug: string): Promise { // Manifest → page count. const manRes = await fetch(`/transcripts/${slug}/manifest.json`, { cache: "force-cache", }); if (!manRes.ok) throw new Error(`manifest ${manRes.status}`); const manifest = (await manRes.json()) as { pageCount?: number }; const pageCount = manifest.pageCount ?? 0; const idx = mkIndex(); const meta: IndexHit[] = []; let id = 0; for (let p = 0; p < pageCount; p++) { const res = await fetch(`/transcripts/${slug}/${pageFileName(p)}`, { cache: "force-cache", }); if (!res.ok) throw new Error(`page ${p} ${res.status}`); const entries = (await res.json()) as RawEntry[]; for (const entry of entries) { const eslug = typeof entry.slug === "string" ? entry.slug : slug; const cues = Array.isArray(entry.cues) ? entry.cues : []; const said = new Set(); for (const c of cues as RawCue[]) { const text = typeof c.text === "string" ? c.text : ""; if (!text) continue; said.add(text); idx.add(id, text); meta[id] = { slug: eslug, start: c.start, text }; id++; } // The record's alternate English tracks (lib/captionTracks.ts): a line // the transcript itself does not say is indexed too, under its track. const alts = Array.isArray(entry.altTracks) ? entry.altTracks : []; for (const alt of alts as { track?: unknown; cues?: unknown }[]) { const track = typeof alt.track === "string" ? alt.track : ""; if (!track || !Array.isArray(alt.cues)) continue; for (const c of alt.cues as RawCue[]) { const text = typeof c.text === "string" ? c.text : ""; if (!text || said.has(text)) continue; idx.add(id, text); meta[id] = { slug: eslug, start: c.start, text, track }; id++; } } } post({ type: "progress", reqId, slug, done: p + 1, total: pageCount }); } channels.set(slug, { idx, meta }); post({ type: "built", reqId, slug, docCount: meta.length }); } async function search( reqId: number, slug: string, term: string, limit: number, ): Promise { const ch = channels.get(slug); if (!ch) { post({ type: "result", reqId, slug, hits: [] }); return; } // The in-worker Index is synchronous; search returns ids directly. const ids = ch.idx.search(term, { limit }) as unknown as number[]; const hits: IndexHit[] = []; for (const rid of ids) { const m = ch.meta[rid]; if (m) hits.push(m); } post({ type: "result", reqId, slug, hits }); } ctx.addEventListener("message", (e: MessageEvent) => { const msg = e.data; const fail = (err: unknown) => post({ type: "error", reqId: msg.reqId, slug: (msg as { slug?: string }).slug ?? "", message: err instanceof Error ? err.message : String(err), }); switch (msg.type) { case "build": build(msg.reqId, msg.slug).catch(fail); break; case "search": search(msg.reqId, msg.slug, msg.term, msg.limit).catch(fail); break; case "status": { const ch = channels.get(msg.slug); post({ type: "status", reqId: msg.reqId, slug: msg.slug, ready: !!ch, docCount: ch ? ch.meta.length : 0, }); break; } case "drop": channels.delete(msg.slug); post({ type: "dropped", reqId: msg.reqId, slug: msg.slug }); break; } });