// Dry-running a curated-tag rule over the transcript index, READ-ONLY and // BOUNDED. The editor's /tags preview is the only caller; it lives here (and // not in editor/) because this is where `lmdb` is a dependency and where every // other reader of index.mdb already lives. // // A rule is only worth anything if the operator can see what it would catch // before it is baked into a build, and the only place the corpus's titles, // captions and chat already exist in decoded form is transcripts/index.mdb. // So this opens that index the way recencyIndex.ts and channelSignature.ts do — // `readOnly: true`, guarded on existsSync, every failure degrading to "no // results" rather than throwing — and never writes, never builds, never boots // a second editor against the corpus. // // THREE THINGS KEEP IT BOUNDED, because the largest channel here has ~11,000 // videos and the corpus ~79,000: // // 1. CHANNEL SCOPE FIRST. A rule that names channels scans only those, by // range over the `byChannel` sub-DB, whose key is [slug, uploadDate, id]. // 2. A SCAN CAP PER CALL, and it is much lower for the cue kinds: a metadata // rule decodes one small summary per video, while a caption rule decodes // an entire transcript's cues. They are not the same cost and must not // share a budget. // 3. A MATCH CAP. The preview is for judging a rule, not for enumerating its // hits — a rule matching ten thousand videos is answered with the first // page and "more". // // A call that stops on a cap returns a CURSOR (the last key it looked at), so // "Scan more" resumes exactly where it left off instead of re-reading the head. import { existsSync } from "node:fs"; import { open } from "lmdb"; import type { Paths } from "../lib/paths"; import type { TranscriptSummary } from "../lib/transcripts"; import type { StoredSubs } from "../lib/subs"; import type { Cue } from "../lib/vtt"; import { compileTagRules, evaluateCompiledRules, ruleApplies, type CuratedTagDef, type CompiledTagRule, type CompiledTagRules, } from "../lib/curatedTags"; // buildIndex.ts's own key shapes. [uploadDate, channelSlug, id] for the value // DBs; [channelSlug, uploadDate, id] for the channel index. type IndexKey = [string, string, string]; type ChannelKey = [string, string, string]; // The live chat track's name in `subs`, set by buildIndex from the sidecar's // own filename. chat-author rules read this track and no other. const LIVE_CHAT_TRACK = "live_chat"; export type TagIndexMatch = { channelSlug: string; id: string; title: string; uploadDate: string; }; export type TagPreviewResult = { matches: TagIndexMatch[]; // Videos actually looked at by this call (not the corpus size). scanned: number; // Null when the whole scope was scanned; otherwise resume with it. nextCursor: string | null; // Why it stopped, for a sentence the UI can show rather than a silent cap. stoppedBy: "end" | "scan-cap" | "match-cap"; // The index is absent or unreadable — a corpus that has never been built. // NOT an error: a rule can be written before the first build, it just cannot // be previewed. indexAvailable: boolean; // Records this call could not decode. Counted and surfaced rather than // thrown: the header promises that a preview degrades to fewer results, and // a scan that dies halfway through a channel because one value is unreadable // would take the whole video-detail page down with it. unreadable: number; }; // Per-call budgets. Metadata decodes one summary per video; the cue kinds // decode a whole transcript or chat log, which is why they get ~1/20th of it. const SCAN_CAP_METADATA = 5_000; const SCAN_CAP_CUES = 250; const MATCH_CAP = 100; type Opened = { sums: { get(key: IndexKey): TranscriptSummary | undefined }; cues: { get(key: IndexKey): Cue[] | undefined }; subs: { get(key: IndexKey): StoredSubs | undefined }; byChannel: { getRange(opts: { start: ChannelKey; end: ChannelKey }): Iterable<{ key: unknown; }>; }; close(): Promise; }; // Same shape as openUploadDateIndex: never throws, returns null when there is // nothing to read. function openIndex(paths: Paths): Opened | null { if (!existsSync(paths.lmdbPath)) return null; let root: ReturnType; try { // maxDbs matches buildIndex's, which is the process that created the file. // // `compression: true` IS NOT OPTIONAL, and leaving it off does not merely // read slowly — it reads WRONG. buildIndex writes with compression // (buildIndex.ts's own open), and lmdb-js only interprets the compressed // status byte when the READING store carries a compression object; without // one, every value over its ~1 KB threshold throws "Data read, but end of // buffer not reached". That is every normal description and every // transcript — i.e. exactly the values a tag rule exists to match. Every // other reader of this file says it too (digestPlan.ts, buildStats.ts). root = open({ path: paths.lmdbPath, readOnly: true, maxDbs: 18, compression: true, }); } catch { return null; } try { const sums = root.openDB({ name: "sums", encoding: "msgpack", }); const cues = root.openDB({ name: "cues", encoding: "msgpack", }); const subs = root.openDB({ name: "subs", encoding: "msgpack", }); const byChannel = root.openDB({ name: "byChannel", encoding: "msgpack", }); return { sums, cues, subs, byChannel, close: () => root.close().catch(() => {}), } as unknown as Opened; } catch { void root.close(); return null; } } function encodeCursor(k: ChannelKey): string { return k.join(""); } function decodeCursor(cursor: string | null): ChannelKey | null { if (!cursor) return null; const parts = cursor.split(""); return parts.length === 3 ? [parts[0], parts[1], parts[2]] : null; } // True when every enabled rule is metadata-only — the cheap path, which never // touches `cues` or `subs`. function isMetadataOnly(compiled: CompiledTagRules): boolean { return compiled.chatAuthor.length === 0 && compiled.caption.length === 0; } // The live-chat cues of a video, or undefined. buildIndex stores every sub // track under one key; chat-author rules want exactly one of them. function chatCuesOf(stored: StoredSubs | undefined): Cue[] | undefined { return stored?.find((t) => t.track === LIVE_CHAT_TRACK)?.cues; } // Does any rule of this kind survive the channel/date scope for THIS record? // // evaluateCompiledRules applies the same test internally, but it does so AFTER // the caller has handed it the cues — and decoding a whole transcript for a // record that no caption rule can fire on is the single most expensive thing a // preview can do. So the scope test runs first, here, and the decode only // happens when a rule could actually use it. It asks curatedTags.ts's exported // `ruleApplies` — the same predicate the evaluation uses — rather than // restating it: the second copy that used to live here stopped predicting the // evaluation it guards the moment a scope field was added (channelsExclude, // 2026-09-22), which this time cost only wasted decodes. function anyRuleInScope( rules: CompiledTagRule[], channelSlug: string, uploadDate: string | undefined, ): boolean { return rules.some((rule) => ruleApplies(rule, { channelSlug, uploadDate })); } export type TagPreviewInput = { // The tag being previewed, with the rules to dry-run. Only its `rules` and // `id` are read — a def that does not exist yet previews fine. def: CuratedTagDef; // Channel slugs to scan. Empty = every slug given in `allSlugs`. channels: string[]; allSlugs: string[]; cursor?: string | null; }; // Dry-run a tag's rules over the index. Pure read: nothing is persisted, which // is the contract — rule hits are DERIVED at build time and never stored. export function previewTagRule( paths: Paths, input: TagPreviewInput, ): TagPreviewResult { const compiled = compileTagRules([input.def]); const empty: TagPreviewResult = { matches: [], scanned: 0, nextCursor: null, stoppedBy: "end", indexAvailable: true, unreadable: 0, }; if (compiled.isEmpty) return empty; const db = openIndex(paths); if (!db) return { ...empty, indexAvailable: false }; const metaOnly = isMetadataOnly(compiled); const scanCap = metaOnly ? SCAN_CAP_METADATA : SCAN_CAP_CUES; const scope = input.channels.length > 0 ? input.allSlugs.filter((s) => input.channels.includes(s)) : [...input.allSlugs]; scope.sort(); const resume = decodeCursor(input.cursor ?? null); const matches: TagIndexMatch[] = []; let scanned = 0; let unreadable = 0; let stoppedBy: TagPreviewResult["stoppedBy"] = "end"; let lastKey: ChannelKey | null = null; try { outer: for (const slug of scope) { // A cursor names the slug it stopped inside; skip the channels before it. if (resume && slug < resume[0]) continue; const start: ChannelKey = resume && slug === resume[0] ? resume : [slug, "", ""]; for (const { key } of db.byChannel.getRange({ start, end: [slug, "￿", "￿"], })) { const k = key as ChannelKey; if (k[0] !== slug) break; // getRange's start is INCLUSIVE, so a resumed scan would re-examine the // key it stopped on. Skipping it is what makes "Scan more" progress. if (resume && encodeCursor(k) === encodeCursor(resume)) continue; lastKey = k; const indexKey: IndexKey = [k[1], k[0], k[2]]; // ONE RECORD'S FAILURE IS ONE RECORD. A value that will not decode — // a half-written record, a schema this build predates, a compression // mismatch — must cost its own row and nothing else: this function is // awaited by a page render, and a throw here would 500 the whole page // over one video. try { const summary = db.sums.get(indexKey); if (summary) { const wantsCaption = anyRuleInScope( compiled.caption, summary.channelSlug, summary.uploadDate, ); const wantsChat = anyRuleInScope( compiled.chatAuthor, summary.channelSlug, summary.uploadDate, ); const hits = evaluateCompiledRules( { channelSlug: summary.channelSlug, id: summary.id, title: summary.title, description: summary.description, tags: summary.tags, uploadDate: summary.uploadDate, // Decoded ONLY when a rule of that kind could fire on THIS // record — the whole cost of a caption preview is this decode. captionCues: wantsCaption ? db.cues.get(indexKey) : undefined, chatCues: wantsChat ? chatCuesOf(db.subs.get(indexKey)) : undefined, }, compiled, ); if (hits.length > 0) { matches.push({ channelSlug: summary.channelSlug, id: summary.id, title: summary.title, uploadDate: summary.uploadDate, }); } } } catch { unreadable += 1; } scanned += 1; if (matches.length >= MATCH_CAP) { stoppedBy = "match-cap"; break outer; } if (scanned >= scanCap) { stoppedBy = "scan-cap"; break outer; } } } } catch { // The iteration itself failed (a corrupt page, a mid-rebuild file). Keep // what was found and stop, exactly as the caps do — the alternative is a // preview that throws and a page that will not render. unreadable += 1; } finally { void db.close(); } return { matches, scanned, nextCursor: stoppedBy === "end" ? null : lastKey ? encodeCursor(lastKey) : null, stoppedBy, indexAvailable: true, unreadable, }; } // The tags one video's rules would produce — the video page's "rule hit" // provenance. Costs one summary decode (plus the cue decode a caption or chat // rule needs) for ONE key, so it is nothing like a preview. // // `uploadDate` is the index key's first element; when the caller knows it (the // video page reads metadata.info.json anyway) the lookup is a single get. When // it does not, the channel's key range is scanned for the id — key-only, no // value decoded. export function ruleHitsForVideo( paths: Paths, defs: CuratedTagDef[], channelSlug: string, id: string, uploadDate?: string, ): { hits: string[]; indexAvailable: boolean; indexed: boolean } { const compiled = compileTagRules(defs); if (compiled.isEmpty) { return { hits: [], indexAvailable: true, indexed: true }; } const db = openIndex(paths); if (!db) return { hits: [], indexAvailable: false, indexed: false }; // READ AS "not indexed", NEVER AS A THROWN PAGE. This is awaited inside the // video-detail render, so an unreadable value here would take the page down // for a video whose only sin is being large. The panel already knows how to // say "not in the index yet". try { let key: IndexKey | null = uploadDate ? [uploadDate, channelSlug, id] : null; if (key && !db.sums.get(key)) key = null; if (!key) { for (const { key: ck } of db.byChannel.getRange({ start: [channelSlug, "", ""], end: [channelSlug, "￿", "￿"], })) { const k = ck as ChannelKey; if (k[0] !== channelSlug) break; if (k[2] === id) { key = [k[1], k[0], k[2]]; break; } } } if (!key) return { hits: [], indexAvailable: true, indexed: false }; const summary = db.sums.get(key); if (!summary) return { hits: [], indexAvailable: true, indexed: false }; const hits = evaluateCompiledRules( { channelSlug: summary.channelSlug, id: summary.id, title: summary.title, description: summary.description, tags: summary.tags, uploadDate: summary.uploadDate, captionCues: anyRuleInScope( compiled.caption, summary.channelSlug, summary.uploadDate, ) ? db.cues.get(key) : undefined, chatCues: anyRuleInScope( compiled.chatAuthor, summary.channelSlug, summary.uploadDate, ) ? chatCuesOf(db.subs.get(key)) : undefined, }, compiled, ); return { hits, indexAvailable: true, indexed: true }; } catch { return { hits: [], indexAvailable: true, indexed: false }; } finally { void db.close(); } } // ─── Directory name → record id ────────────────────────────────────────────── // // A video's DATA DIRECTORY is usually named for its id, and sometimes is not: // summarize() takes `webpage_url_basename` for Odysee, a Rumble directory is // named for the URL slug while the record carries the embed id, and a Twitch // VOD's directory drops the leading "v". Measured on the live corpus // (2026-09-21): three of the thirty-odd channels here differ that way. // // That matters to anything keyed by the RECORD id — curated-tag assignments // are — while holding DIRECTORY names, which is what a channel's video list // holds. The index already stores the mapping: the `mtimes` sub-DB is keyed // [channelSlug, videoDir] and its value carries the record's indexKey. One // ranged read per channel, and no metadata.info.json re-read. // // Returns dirName -> recordId, empty when there is no index (every caller then // falls back to the identity mapping, which is right for the common case). type MtimeRecordLike = { indexKey: IndexKey }; export function recordIdsByVideoDir( paths: Paths, channelSlug: string, ): Map { const out = new Map(); if (!existsSync(paths.lmdbPath)) return out; let root: ReturnType; try { // compression: true for the same reason openIndex says it — the writer // compresses, and a reader without it throws on every large value. root = open({ path: paths.lmdbPath, readOnly: true, maxDbs: 18, compression: true, }); } catch { return out; } try { const mtimes = root.openDB({ name: "mtimes", encoding: "msgpack", }); for (const { key, value } of mtimes.getRange({ start: [channelSlug, ""], end: [channelSlug, "￿"], })) { const k = key as [string, string]; if (k[0] !== channelSlug) break; const id = (value as MtimeRecordLike)?.indexKey?.[2]; if (typeof id === "string" && id !== "") out.set(k[1], id); } } catch { // A mid-rebuild or missing sub-DB is "no mapping", never a failed page. } finally { void root.close().catch(() => {}); } return out; }