Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 403cdae39b1d27e3d8e2aed694a95221c7d0762e
parent a4f77c30bee93e9a25ede8b5082516408813b494
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 13:09:59 -0400

tags S1.3: derive curatedTags in the index, and notice when only a tag moved

Rule hits are not persisted anywhere, so nothing in the per-video mtime diff
can see that a rule was edited or a video pinned. common/controller/
curatedTagsIndex.ts is the answer: two hashes in the existing `meta` sub-DB —
curatedRulesHash (every def's id and order plus every ENABLED rule's kind,
pattern, channel scope and dates; presentation fields deliberately excluded, so
relabelling a tag does not re-page 30,000 videos) and curatedAssignHash — are
checked on EVERY build, with a per-video signature map beside them so an
assignment-only edit re-derives just the videos whose assignment moved, found
by a key-only byChannel scan per affected channel.

The per-video worker derives tags where the caption cues and the live_chat
track are already in hand; sums.put moved below the subs parse for that. The
re-apply pass re-derives straight out of LMDB — no video directory is read
twice — and reads cues only for videos a caption/chat rule is actually scoped
to. Whatever it changes flips sharedNeedsBuild, because the page writer's sha1
skip is what then keeps the untouched pages untouched.

SCHEMA_VERSION 14 (one-time re-derive; no new sub-DB, so no new clearAsync
line). toDisplaySummary carries curatedTags through, omitted when empty. Each
site's summaries stream now also accumulates per-tag counts into
tag-counts.json, and the site fingerprint includes the two hashes so a tag-only
edit cannot leave a site reporting "up to date" with yesterday's counts.

Measured on the real legal-mindset channel (518 videos, 1.37 M caption cues +
1.47 M chat cues) with the three seed rules, offline against the production
LMDB: 1.28 s warm — 1.03 s of it LMDB decode, 232 ms of actual regex. The
metadata rule alone is 18 ms.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Mcommon/controller/buildIndex.ts | 91++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Acommon/controller/curatedTagsIndex.test.ts | 379+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/curatedTagsIndex.ts | 463+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/transcripts-server.ts | 6++++++
4 files changed, 936 insertions(+), 3 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -66,6 +66,7 @@ import { getSettings } from "../lib/settings"; import { listSites, siteDigestsDir, + siteIndexDir, sitePostsDir, siteSummariesDir, siteSubsDir, @@ -127,6 +128,14 @@ import { } from "../lib/digests"; import { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME, effectiveDigest } from "../lib/digest"; import { loadDigest, loadDigestOverrides } from "../lib/digest-server"; +import { + TAG_COUNTS_FILENAME, + applyCuratedTagsToSummary, + createTagCounter, + deriveCuratedTags, + loadCuratedTagsRuntime, + reapplyCuratedTags, +} from "./curatedTagsIndex"; // v10: multi-site build. Shared per-channel transcript/subs pages are written // once; per-site summaries + subs manifests are filtered selections. Bumped to @@ -140,7 +149,12 @@ import { loadDigest, loadDigestOverrides } from "../lib/digest-server"; // (effectiveDigest of the machine sidecar + the human override file) plus a // shared /digests/<slug>/ page tree. Bumped so existing indexes populate the // new sub-DB — nothing re-derives it lazily. -const SCHEMA_VERSION = 13; +// v14: curated per-video tags. Summaries now carry `curatedTags` (derived from +// transcripts/tags.json: rule hits folded with the operator's pins and +// suppressions). Bumped so every existing record is re-derived once; after +// that the two `meta` hashes (curatedRulesHash / curatedAssignHash) carry +// invalidation, and no new sub-DB was added — see curatedTagsIndex.ts. +const SCHEMA_VERSION = 14; // Per-channel post stats, persisted so per-site aggregates survive a no-op // rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat. @@ -611,9 +625,18 @@ export async function buildIndex({ } return true; }; - const sharedNeedsBuild = + let sharedNeedsBuild = anyMutations || schemaBumped || !(await sharedManifestsPresent()); + // The curated-tag vocabulary + assignments, loaded and compiled ONCE for the + // whole build. Every per-video derivation below uses this; reapplyCuratedTags + // (after the worker loop) decides whether a tag-only edit has to re-derive + // records the mtime diff never looked at. + const curated = loadCuratedTagsRuntime(paths); + // indexKeyIds the worker derived fresh this build, so the re-apply pass does + // not do them a second time. + const curatedFresh = new Set<string>(); + log( `Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`, ); @@ -694,7 +717,6 @@ export async function buildIndex({ byChannel.remove(indexToChannelKey(prev.indexKey)); } - sums.put(indexKey, summary); if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); @@ -730,6 +752,29 @@ export async function buildIndex({ if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs); else subs.remove(indexKey); + // Curated tags. Derived here, with the caption cues and the live-chat + // track already in hand, and stored ON the summary — which is why the + // sums.put waited for the subs parse. Omitted when empty, so an + // untagged corpus's pages stay byte-identical. Costs nothing at all + // when the corpus has no tags (curated.isNoop short-circuits). + if (!curated.isNoop) { + applyCuratedTagsToSummary( + summary, + deriveCuratedTags(curated, { + channelSlug: s.channelSlug, + id: summary.id, + title: summary.title, + description: summary.description, + tags: summary.tags, + uploadDate: summary.uploadDate, + captionCues: cueList, + chatCues: parsedSubs.find((t) => t.track === "live_chat")?.cues, + }), + ); + } + curatedFresh.add(indexKeyId(indexKey)); + sums.put(indexKey, summary); + // The derived layer. Only opened when the scan saw a sidecar, so the // ~76k videos without one cost zero extra reads. What is stored is // effectiveDigest(machine, overrides) — human corrections applied, @@ -824,6 +869,26 @@ export async function buildIndex({ mtimes.remove(pathKey); } + // Curated tags can change with no video on disk moving at all — a rule + // edited, a video pinned — and nothing in the mtime diff above can see that. + // The two meta hashes are therefore checked on EVERY build, and the videos + // whose tags moved are re-derived straight out of LMDB (no disk re-read). + // Anything that changed has to be re-paged, which is what the + // sharedNeedsBuild flip below buys: the page writer's sha1 skip then keeps + // the untouched pages untouched, so compose still copies only real changes. + const curatedReapply = reapplyCuratedTags({ + runtime: curated, + sums, + cues, + subs, + byChannel, + meta, + alreadyFresh: curatedFresh, + allFresh: schemaBumped, + log, + }); + if (curatedReapply.changed.length > 0) sharedNeedsBuild = true; + await sums.flushed; await cues.flushed; await subs.flushed; @@ -1680,6 +1745,12 @@ export async function buildIndex({ groups: site.groups, defaultGroupId: site.defaultGroupId, channels: site.channels, + // A tag-only edit moves no site config and no video mtime, but it does + // change this site's tag-counts.json (and the curatedTags on its summary + // pages). Without these the site would report "up to date" and ship + // yesterday's counts. + curatedRules: curated.rulesHash, + curatedAssign: curated.assignHash, }); const fpKey = `siteFp:${site.siteId}`; if ( @@ -1739,12 +1810,18 @@ export async function buildIndex({ pageIndex++; }; + // Per-tag totals for THIS site, accumulated in the pass that is already + // streaming every member record — no second walk of the corpus. compose + // joins these to the site's effective vocabulary to write /tags.json. + const tagCounter = createTagCounter(); + for (const { key, value } of sums.getRange({ reverse: true })) { const ik = key as IndexKey; if (!slugSet.has(ik[1])) continue; const s = value as TranscriptSummary; const acc = chan.get(ik[1]); if (acc) acc.count++; + tagCounter.add(ik[1], s.curatedTags); buffer.push( toDisplaySummary(s, { state: stateByIndexKey.get(indexKeyId(ik)) }), ); @@ -1753,6 +1830,14 @@ export async function buildIndex({ } await flushPage(); + // Written even when empty (a tiny `{tags:{}}`), so a site that loses its + // last tagged video does not keep serving a stale count file. compose + // decides from this whether to publish /tags.json at all. + await writeJsonAtomic( + path.join(siteIndexDir(paths, site.siteId), TAG_COUNTS_FILENAME), + tagCounter.file(new Date().toISOString()), + ); + const expectedPages = new Set<string>(); for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i)); for (const name of await readdir(summariesOut).catch(() => [] as string[])) { diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts @@ -0,0 +1,379 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import type { Cue } from "../lib/vtt"; +import type { TranscriptSummary } from "../lib/transcripts"; +import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags"; +import { + applyCuratedTagsToSummary, + createTagCounter, + curatedTagsRuntime, + deriveCuratedTags, + hashCuratedAssignments, + hashCuratedRules, + publishedTagsFrom, + reapplyCuratedTags, + META_ASSIGN_HASH, + META_ASSIGN_SIGS, + META_RULES_HASH, +} from "./curatedTagsIndex"; + +type IndexKey = [string, string, string]; + +// Minimal stand-ins for the LMDB sub-DBs buildIndex hands the re-apply pass. +// Keys are the same tuples; range scans only need to be correct, not fast. +function makeDb<V>() { + const map = new Map<string, { key: unknown; value: V }>(); + const k = (key: unknown) => JSON.stringify(key); + return { + map, + get(key: unknown): V | undefined { + return map.get(k(key))?.value; + }, + put(key: unknown, value: V): void { + map.set(k(key), { key, value }); + }, + getRange(options?: { start?: unknown[] }): Iterable<{ key: unknown; value: V }> { + const start = options?.start as string[] | undefined; + const all = Array.from(map.values()).sort((a, b) => + k(a.key) < k(b.key) ? -1 : 1, + ); + if (!start) return all; + return all.filter((e) => (e.key as string[])[0] === start[0]); + }, + }; +} + +function makeMeta() { + const map = new Map<string, unknown>(); + return { + map, + get: (key: string) => map.get(key), + put: (key: string, value: unknown) => void map.set(key, value), + }; +} + +function summary( + channelSlug: string, + id: string, + uploadDate: string, + over: Partial<TranscriptSummary> = {}, +): TranscriptSummary { + return { + slug: `${channelSlug}/${id}`, + id, + channelSlug, + title: `Video ${id}`, + uploadDate, + duration: 100, + channel: channelSlug, + description: "", + tags: [], + isLivestream: false, + ageRestricted: false, + platform: "youtube", + webpageUrl: `https://example.com/${id}`, + ...over, + }; +} + +const COLLAB_RULE: CuratedTagDef = { + id: "eva-collab", + label: "Collab", + order: 1, + rules: [{ id: "meta", kind: "metadata", pattern: "elfpire", enabled: true }], +}; + +function fixture(config: CuratedTagsConfig) { + const sums = makeDb<TranscriptSummary>(); + const cues = makeDb<Cue[]>(); + const subs = makeDb<{ track: string; cues: Cue[] }[]>(); + const byChannel = makeDb<number>(); + const meta = makeMeta(); + const put = (s: TranscriptSummary) => { + const ik: IndexKey = [s.uploadDate, s.channelSlug, s.id]; + sums.put(ik, s); + byChannel.put([s.channelSlug, s.uploadDate, s.id], 1); + return ik; + }; + return { sums, cues, subs, byChannel, meta, put, runtime: curatedTagsRuntime(config) }; +} + +function cfg( + tags: CuratedTagDef[], + assignments: CuratedTagsConfig["assignments"] = {}, +): CuratedTagsConfig { + return { version: 1, tags, assignments }; +} + +// ─── hashes ─── + +test("hashCuratedRules ignores presentation, reacts to derivation", () => { + const base = hashCuratedRules([COLLAB_RULE]); + assert.equal( + hashCuratedRules([{ ...COLLAB_RULE, label: "On mic", color: "#fff", hidden: true }]), + base, + "relabelling must not re-page the corpus", + ); + assert.notEqual( + hashCuratedRules([ + { ...COLLAB_RULE, rules: [{ id: "meta", kind: "metadata", pattern: "eva", enabled: true }] }, + ]), + base, + ); + assert.notEqual(hashCuratedRules([{ ...COLLAB_RULE, order: 9 }]), base); + assert.notEqual( + hashCuratedRules([ + { + ...COLLAB_RULE, + rules: [{ ...COLLAB_RULE.rules![0], channels: ["legal-mindset"] }], + }, + ]), + base, + ); + // A disabled rule is not part of the derivation at all. + assert.equal( + hashCuratedRules([ + { + ...COLLAB_RULE, + rules: [COLLAB_RULE.rules![0], { id: "off", kind: "caption", pattern: "x", enabled: false }], + }, + ]), + base, + ); +}); + +test("hashCuratedAssignments is order-insensitive", () => { + const a = hashCuratedAssignments({ + "c/v1": { manual: ["b", "a"] }, + "c/v2": { suppressed: ["z"] }, + }); + const b = hashCuratedAssignments({ + "c/v2": { suppressed: ["z"] }, + "c/v1": { manual: ["a", "b"] }, + }); + assert.equal(a, b); + assert.notEqual(a, hashCuratedAssignments({ "c/v1": { manual: ["a"] } })); +}); + +// ─── the summary field ─── + +test("applyCuratedTagsToSummary omits the key when empty", () => { + const s = summary("c", "v", "20260101"); + assert.equal(applyCuratedTagsToSummary(s, []), false); + assert.equal("curatedTags" in s, false); + assert.equal(applyCuratedTagsToSummary(s, ["a"]), true); + assert.deepEqual(s.curatedTags, ["a"]); + assert.equal(applyCuratedTagsToSummary(s, ["a"]), false, "idempotent"); + assert.equal(applyCuratedTagsToSummary(s, []), true); + assert.equal("curatedTags" in s, false, "the key is deleted, not emptied"); +}); + +// ─── derivation ─── + +test("deriveCuratedTags folds rule hits with pins and suppressions", () => { + const runtime = curatedTagsRuntime( + cfg([COLLAB_RULE, { id: "manual-only", label: "Manual", order: 2 }], { + "lm/v2": { manual: ["manual-only"] }, + "lm/v3": { suppressed: ["eva-collab"] }, + }), + ); + assert.deepEqual( + deriveCuratedTags(runtime, { channelSlug: "lm", id: "v1", title: "with Elfpire" }), + ["eva-collab"], + ); + assert.deepEqual( + deriveCuratedTags(runtime, { channelSlug: "lm", id: "v2", title: "nothing" }), + ["manual-only"], + ); + assert.deepEqual( + deriveCuratedTags(runtime, { channelSlug: "lm", id: "v3", title: "with Elfpire" }), + [], + "a suppression kills a rule hit", + ); +}); + +// ─── reapplyCuratedTags ─── + +test("unchanged hashes do no work at all", () => { + const f = fixture(cfg([COLLAB_RULE])); + f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" })); + f.meta.put(META_RULES_HASH, f.runtime.rulesHash); + f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash); + const res = reapplyCuratedTags({ ...f }); + assert.deepEqual(res, { + rulesChanged: false, + assignmentsChanged: false, + changed: [], + examined: 0, + }); +}); + +test("a rule change re-derives the whole corpus from LMDB and reports the movers", () => { + const f = fixture(cfg([COLLAB_RULE])); + f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" })); + f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" })); + f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" })); + const res = reapplyCuratedTags({ ...f }); + assert.equal(res.rulesChanged, true); + assert.equal(res.examined, 3); + assert.deepEqual( + res.changed.map((k) => k[2]).sort(), + ["v1", "v3"], + ); + assert.deepEqual(f.sums.get(["20260101", "lm", "v1"])!.curatedTags, ["eva-collab"]); + assert.equal(f.sums.get(["20260102", "lm", "v2"])!.curatedTags, undefined); + // The hashes are recorded, so a second pass is a no-op. + assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash); + assert.equal(reapplyCuratedTags({ ...f }).examined, 0); +}); + +test("removing the last rule strips the tag back off every record", () => { + const f = fixture(cfg([COLLAB_RULE])); + const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" })); + reapplyCuratedTags({ ...f }); + assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]); + const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) }; + const res = reapplyCuratedTags(gone); + assert.deepEqual(res.changed.length, 1); + assert.equal(f.sums.get(ik)!.curatedTags, undefined); +}); + +test("an assignment-only change touches ONLY the videos whose assignment moved", () => { + const f = fixture(cfg([COLLAB_RULE])); + for (let i = 0; i < 20; i++) { + f.put(summary("lm", `v${i}`, `202601${String(i + 10).padStart(2, "0")}`)); + } + f.put(summary("other", "w1", "20260201")); + reapplyCuratedTags({ ...f }); // establish the baseline hashes + + const pinned = { + ...f, + runtime: curatedTagsRuntime( + cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }), + ), + }; + const res = reapplyCuratedTags(pinned); + assert.equal(res.rulesChanged, false); + assert.equal(res.assignmentsChanged, true); + // 21 videos in the corpus; exactly one was even looked at. + assert.equal(res.examined, 1); + assert.deepEqual(res.changed.map((k) => k[2]), ["v7"]); + assert.deepEqual(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, ["eva-collab"]); + + // Clearing it again is also a one-video pass, and the key comes back off. + const cleared = { ...f, runtime: curatedTagsRuntime(cfg([COLLAB_RULE])) }; + const res2 = reapplyCuratedTags(cleared); + assert.equal(res2.examined, 1); + assert.equal(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, undefined); +}); + +test("videos the worker already derived this build are not done twice", () => { + const f = fixture(cfg([COLLAB_RULE])); + const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" })); + const res = reapplyCuratedTags({ + ...f, + alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]), + }); + assert.equal(res.examined, 0); + assert.deepEqual(res.changed, []); +}); + +test("a schema bump records the hashes and re-applies nothing", () => { + const f = fixture(cfg([COLLAB_RULE])); + f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" })); + const res = reapplyCuratedTags({ ...f, allFresh: true }); + assert.equal(res.changed.length, 0); + assert.equal(res.examined, 0); + assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash); + assert.deepEqual(f.meta.get(META_ASSIGN_SIGS), {}); +}); + +test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => { + const capRule: CuratedTagDef = { + id: "eva-topic", + label: "Discussed", + rules: [ + { + id: "cap", + kind: "caption", + pattern: "elfpire", + channels: ["lm"], + enabled: true, + }, + ], + }; + const chatRule: CuratedTagDef = { + id: "eva-in-chat", + label: "In chat", + rules: [{ id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true }], + }; + const f = fixture(cfg([capRule, chatRule])); + const a = f.put(summary("lm", "v1", "20260101")); + const b = f.put(summary("other", "v2", "20260102")); + f.cues.put(a, [{ start: 0, end: 1, text: "we talked about Elfpire" }]); + f.cues.put(b, [{ start: 0, end: 1, text: "we talked about Elfpire" }]); + f.subs.put(a, [ + { track: "en", cues: [{ start: 0, end: 1, text: "ElfpireEva: not chat" }] }, + { track: "live_chat", cues: [{ start: 0, end: 1, text: "ElfpireEva: hi" }] }, + ]); + + reapplyCuratedTags({ ...f }); + // Order follows the vocabulary (neither def sets `order`, so it is the + // config's own order), not the alphabet. + assert.deepEqual(f.sums.get(a)!.curatedTags, ["eva-topic", "eva-in-chat"]); + // The caption rule is scoped to `lm`, so the other channel gets nothing even + // though its cues would match. + assert.equal(f.sums.get(b)!.curatedTags, undefined); +}); + +// ─── counts + publication ─── + +test("createTagCounter accumulates per tag and per channel", () => { + const c = createTagCounter(); + c.add("lm", ["eva-collab", "eva-topic"]); + c.add("lm", ["eva-collab"]); + c.add("nux", ["eva-collab"]); + c.add("nux", undefined); + c.add("nux", []); + const file = c.file("2026-09-21T00:00:00Z"); + assert.equal(file.version, 1); + assert.deepEqual(file.tags["eva-collab"], { + count: 3, + channels: { lm: 2, nux: 1 }, + }); + assert.deepEqual(file.tags["eva-topic"], { count: 1, channels: { lm: 1 } }); +}); + +test("publishedTagsFrom drops hidden and zero-count tags and sorts by order", () => { + const defs: CuratedTagDef[] = [ + { id: "b-tag", label: "B", order: 2, group: "eva", groupLabel: "Eva" }, + { id: "a-tag", label: "A", order: 1, color: "#fff" }, + { id: "hidden-tag", label: "Hidden", order: 0, hidden: true }, + { id: "unused", label: "Unused", order: 3 }, + ]; + const counts = { + version: 1, + generatedAt: "", + tags: { + "a-tag": { count: 2, channels: { lm: 2 } }, + "b-tag": { count: 1, channels: { nux: 1 } }, + "hidden-tag": { count: 5, channels: { lm: 5 } }, + }, + }; + const published = publishedTagsFrom(defs, counts); + assert.deepEqual( + published.tags.map((t) => t.id), + ["a-tag", "b-tag"], + ); + assert.deepEqual(published.tags[0], { + id: "a-tag", + label: "A", + color: "#fff", + order: 1, + count: 2, + channels: { lm: 2 }, + }); + assert.equal(published.tags[1].groupLabel, "Eva"); + // No counts at all → nothing to publish (compose then writes no file). + assert.deepEqual(publishedTagsFrom(defs, null).tags, []); +}); diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts @@ -0,0 +1,463 @@ +// Curated tags, as the index build sees them. +// +// A video's `curatedTags` is derived, never stored on disk beside the video: +// it is (rule hits ∪ operator pins) − suppressions, folded fresh at every +// build. Rule hits in particular are NOT persisted anywhere — edit a rule, +// rebuild, done — which is why this module exists: nothing in the per-video +// mtime diff can notice that a rule changed, so the build has to ask. +// +// Two hashes in the existing `meta` sub-DB answer that question: +// curatedRulesHash the derivation-relevant shape of the vocabulary — every +// def's id and order, and every ENABLED rule's kind, +// pattern, channel scope and date range +// curatedAssignHash every pin and suppression +// Both are checked unconditionally on every build (a skipped check would make +// a rule edit silently inert until some unrelated mtime moved), and alongside +// them `curatedAssignSigs` keeps a per-video signature so an assignment-only +// change can re-derive just the videos whose assignment actually moved instead +// of the whole corpus. +// +// **Only the CORPUS vocabulary is evaluated here.** A site-only rule +// (sites/<id>/tags.json) contributes its definition to that site's published +// /tags.json, but it does not tag records: records are shared by every site +// that carries the channel, so a per-site rule hit would have to live in a +// per-site record. Promote a rule to transcripts/tags.json to make it bind. + +import { createHash } from "node:crypto"; +import type { Cue } from "../lib/vtt"; +import type { Paths } from "../lib/paths"; +import type { TranscriptSummary } from "../lib/transcripts"; +import { + compileTagRules, + effectiveTagsFor, + evaluateCompiledRules, + type CompiledTagRule, + type CompiledTagRules, + type CuratedTagAssignment, + type CuratedTagDef, + type CuratedTagsConfig, + type PublishedTags, + type TagRuleInput, +} from "../lib/curatedTags"; +import { readGlobalTags } from "../lib/curatedTagsStore"; + +export const META_RULES_HASH = "curatedRulesHash"; +export const META_ASSIGN_HASH = "curatedAssignHash"; +export const META_ASSIGN_SIGS = "curatedAssignSigs"; + +export const TAG_COUNTS_FILENAME = "tag-counts.json"; +export const TAG_COUNTS_VERSION = 1; + +type IndexKey = [string, string, string]; +type ChannelKey = [string, string, string]; + +function sha1(s: string): string { + return createHash("sha1").update(s).digest("hex"); +} + +// The shape of the vocabulary that can change a record. Presentation fields +// (label, colour, group, hidden) are deliberately absent: relabelling a tag +// must not re-page 30,000 videos. +export function hashCuratedRules(defs: CuratedTagDef[]): string { + const shape = defs.map((def, i) => [ + def.id, + def.order ?? i, + (def.rules ?? []) + .filter((r) => r.enabled !== false) + .map((r) => [ + r.id, + r.kind, + r.pattern, + [...(r.channels ?? [])].sort(), + r.dateFrom ?? null, + r.dateTo ?? null, + ]), + ]); + return sha1(JSON.stringify(shape)); +} + +// One video's assignment, canonically. Also the per-key signature stored in +// meta, so the diff is a string compare. +export function assignmentSignature(a: CuratedTagAssignment): string { + return `${[...(a.manual ?? [])].sort().join(",")}|${[...(a.suppressed ?? [])].sort().join(",")}`; +} + +export function assignmentSignatures( + assignments: Record<string, CuratedTagAssignment>, +): Record<string, string> { + const out: Record<string, string> = {}; + for (const key of Object.keys(assignments).sort()) { + out[key] = assignmentSignature(assignments[key]); + } + return out; +} + +export function hashCuratedAssignments( + assignments: Record<string, CuratedTagAssignment>, +): string { + return sha1(JSON.stringify(assignmentSignatures(assignments))); +} + +// Everything the build needs, loaded and compiled ONCE per build. +export type CuratedTagsRuntime = { + config: CuratedTagsConfig; + defs: CuratedTagDef[]; + compiled: CompiledTagRules; + rulesHash: string; + assignHash: string; + sigs: Record<string, string>; + // True when the corpus has no rules and no assignments at all — the state of + // every untagged install, and the cue to do no work whatsoever. + isNoop: boolean; +}; + +export function loadCuratedTagsRuntime(paths: Paths): CuratedTagsRuntime { + return curatedTagsRuntime(readGlobalTags(paths)); +} + +export function curatedTagsRuntime(config: CuratedTagsConfig): CuratedTagsRuntime { + const compiled = compileTagRules(config.tags); + return { + config, + defs: config.tags, + compiled, + rulesHash: hashCuratedRules(config.tags), + assignHash: hashCuratedAssignments(config.assignments), + sigs: assignmentSignatures(config.assignments), + isNoop: + compiled.isEmpty && Object.keys(config.assignments).length === 0, + }; +} + +// Cheap pre-checks, so a video on a channel no rule scopes to never pays for a +// cue read. Mirrors the scoping evaluateCompiledRules does internally. +function scopeAllows( + rule: CompiledTagRule, + channelSlug: string, + uploadDate: string | undefined, +): boolean { + if (rule.channels && !rule.channels.has(channelSlug)) return false; + if (rule.dateFrom || rule.dateTo) { + const date = (uploadDate ?? "").replace(/\D/g, ""); + if (!date) return false; + if (rule.dateFrom && date < rule.dateFrom) return false; + if (rule.dateTo && date > rule.dateTo) return false; + } + return true; +} + +export function needsCaptionCues( + runtime: CuratedTagsRuntime, + channelSlug: string, + uploadDate?: string, +): boolean { + return runtime.compiled.caption.some((r) => scopeAllows(r, channelSlug, uploadDate)); +} + +export function needsChatCues( + runtime: CuratedTagsRuntime, + channelSlug: string, + uploadDate?: string, +): boolean { + return runtime.compiled.chatAuthor.some((r) => + scopeAllows(r, channelSlug, uploadDate), + ); +} + +// The whole fold for one video: rule hits, then the operator's pins and +// suppressions over the top. +export function deriveCuratedTags( + runtime: CuratedTagsRuntime, + input: TagRuleInput, +): string[] { + const hits = evaluateCompiledRules(input, runtime.compiled); + const assignment = runtime.config.assignments[`${input.channelSlug}/${input.id}`]; + if (hits.length === 0 && !assignment) return []; + return effectiveTagsFor(hits, assignment, runtime.defs); +} + +// Set/clear `curatedTags` on a summary in place. Returns true when the stored +// value actually changed — omitted-when-empty is the contract, so an untagged +// record must come out byte-identical to the one already on disk. +export function applyCuratedTagsToSummary( + summary: TranscriptSummary, + tags: string[], +): boolean { + const before = summary.curatedTags; + if (tags.length === 0) { + if (before === undefined) return false; + delete summary.curatedTags; + return true; + } + if ( + before && + before.length === tags.length && + before.every((t, i) => t === tags[i]) + ) { + return false; + } + summary.curatedTags = tags; + return true; +} + +// ─── re-derivation ─── + +type SumsDb = { + get(key: IndexKey): TranscriptSummary | undefined; + put(key: IndexKey, value: TranscriptSummary): unknown; + getRange(options?: unknown): Iterable<{ key: unknown; value: unknown }>; +}; +type CuesDb = { get(key: IndexKey): Cue[] | undefined }; +type SubsDb = { get(key: IndexKey): { track: string; cues: Cue[] }[] | undefined }; +type ByChannelDb = { + getRange(options: unknown): Iterable<{ key: unknown }>; +}; +type MetaDb = { + get(key: string): unknown; + put(key: string, value: unknown): unknown; +}; + +export type ReapplyOptions = { + runtime: CuratedTagsRuntime; + sums: SumsDb; + cues: CuesDb; + subs: SubsDb; + byChannel: ByChannelDb; + meta: MetaDb; + // indexKeyIds the per-video worker already derived this build. Skipped here + // so a rule edit landing in the same build as new videos doesn't do them + // twice. + alreadyFresh?: Set<string>; + // A schema bump cleared the cache and the worker re-derived everything, so + // there is nothing to re-apply — just record the hashes. + allFresh?: boolean; + log?: (msg: string) => void; +}; + +export type ReapplyResult = { + rulesChanged: boolean; + assignmentsChanged: boolean; + // Videos whose stored curatedTags moved. Their channels must be re-paged. + changed: IndexKey[]; + // Videos examined (the cost). + examined: number; +}; + +function indexKeyId(k: IndexKey): string { + return `${k[0]}\x00${k[1]}\x00${k[2]}`; +} + +function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[] | undefined { + if (!stored) return undefined; + for (const t of stored) if (t.track === "live_chat") return t.cues; + return undefined; +} + +// Re-derive `curatedTags` for the videos a tag change can reach, WITHOUT +// re-reading a single video directory: cues and live chat come back out of the +// LMDB sub-DBs the build already populated. +// +// rules changed every video is a candidate (a new pattern can match +// anything), but cues are read only for the videos a +// caption/chat rule is actually scoped to +// assignments changed only the videos whose assignment signature moved — +// found by a key-only range scan of `byChannel` per +// affected channel, which is the cheap idiom +// +// Returns the changed index keys so the caller can force those pages to be +// rewritten: nothing in the mtime diff knows this happened. +export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult { + const { runtime, sums, cues, subs, byChannel, meta } = opts; + const log = opts.log ?? (() => {}); + const prevRules = meta.get(META_RULES_HASH) as string | undefined; + const prevAssign = meta.get(META_ASSIGN_HASH) as string | undefined; + const prevSigs = (meta.get(META_ASSIGN_SIGS) as Record<string, string> | undefined) ?? {}; + const rulesChanged = prevRules !== runtime.rulesHash; + const assignmentsChanged = prevAssign !== runtime.assignHash; + + const record = () => { + meta.put(META_RULES_HASH, runtime.rulesHash); + meta.put(META_ASSIGN_HASH, runtime.assignHash); + meta.put(META_ASSIGN_SIGS, runtime.sigs); + }; + + if (!rulesChanged && !assignmentsChanged) { + return { rulesChanged, assignmentsChanged, changed: [], examined: 0 }; + } + if (opts.allFresh) { + record(); + log( + `curated tags: rules ${runtime.rulesHash.slice(0, 8)}, assignments ${runtime.assignHash.slice(0, 8)} (full rebuild, nothing to re-apply)`, + ); + return { rulesChanged, assignmentsChanged, changed: [], examined: 0 }; + } + + const changed: IndexKey[] = []; + let examined = 0; + const skip = opts.alreadyFresh; + + // `known` is the value the cursor already decoded, when there is one — a + // second sums.get() per video would double the msgpack decodes on the + // whole-corpus path. + const rederive = (indexKey: IndexKey, known?: TranscriptSummary): void => { + if (skip?.has(indexKeyId(indexKey))) return; + const summary = known ?? sums.get(indexKey); + if (!summary) return; + examined++; + const channelSlug = indexKey[1]; + const tags = deriveCuratedTags(runtime, { + channelSlug, + id: summary.id, + title: summary.title, + description: summary.description, + tags: summary.tags, + uploadDate: summary.uploadDate, + // Only read what a scoped rule can actually use. On an untagged corpus, + // and on every channel no rule names, this reads nothing. + captionCues: needsCaptionCues(runtime, channelSlug, summary.uploadDate) + ? cues.get(indexKey) + : undefined, + chatCues: needsChatCues(runtime, channelSlug, summary.uploadDate) + ? chatCuesOf(subs.get(indexKey)) + : undefined, + }); + if (applyCuratedTagsToSummary(summary, tags)) { + sums.put(indexKey, summary); + changed.push(indexKey); + } + }; + + if (rulesChanged) { + // Every video is a candidate. This walks `sums` keys only; the value comes + // from the same cursor, and cue reads are gated per video above. + for (const { key, value } of sums.getRange()) { + rederive(key as IndexKey, value as TranscriptSummary); + } + } else { + // Assignment-only change: the symmetric difference of the signatures. + const movedKeys = new Set<string>(); + for (const [key, sig] of Object.entries(runtime.sigs)) { + if (prevSigs[key] !== sig) movedKeys.add(key); + } + for (const key of Object.keys(prevSigs)) { + if (!(key in runtime.sigs)) movedKeys.add(key); + } + // Group by channel so each channel costs one key-only range scan. + const byChannelSlug = new Map<string, Set<string>>(); + for (const key of movedKeys) { + const at = key.indexOf("/"); + if (at <= 0) continue; + const slug = key.slice(0, at); + const id = key.slice(at + 1); + const set = byChannelSlug.get(slug) ?? new Set<string>(); + set.add(id); + byChannelSlug.set(slug, set); + } + for (const [slug, ids] of byChannelSlug) { + for (const { key } of byChannel.getRange({ + start: [slug], + end: [slug, "￿"], + values: false, + })) { + const ck = key as ChannelKey; + if (ck[0] !== slug) continue; + if (!ids.has(ck[2])) continue; + rederive([ck[1], ck[0], ck[2]]); + } + } + } + + record(); + const what = [ + rulesChanged ? `rules ${runtime.rulesHash.slice(0, 8)} (changed)` : null, + assignmentsChanged + ? `assignments ${runtime.assignHash.slice(0, 8)} (changed)` + : null, + ] + .filter(Boolean) + .join(", "); + log( + `curated tags: ${what}, examined ${examined}, re-derived ${changed.length}`, + ); + return { rulesChanged, assignmentsChanged, changed, examined }; +} + +// ─── per-site counts ─── + +export type TagCountsFile = { + version: number; + generatedAt: string; + // tag id -> total on this site + the per-channel breakdown. + tags: Record<string, { count: number; channels: Record<string, number> }>; +}; + +export function emptyTagCounts(): TagCountsFile { + return { version: TAG_COUNTS_VERSION, generatedAt: "", tags: {} }; +} + +// Accumulator for the per-site summaries stream: one add() per record, no +// second pass over the corpus. +export function createTagCounter() { + const tags = new Map<string, { count: number; channels: Map<string, number> }>(); + return { + add(channelSlug: string, curatedTags: string[] | undefined): void { + if (!curatedTags || curatedTags.length === 0) return; + for (const id of curatedTags) { + let entry = tags.get(id); + if (!entry) { + entry = { count: 0, channels: new Map() }; + tags.set(id, entry); + } + entry.count++; + entry.channels.set(channelSlug, (entry.channels.get(channelSlug) ?? 0) + 1); + } + }, + get size(): number { + return tags.size; + }, + file(generatedAt: string): TagCountsFile { + const out: TagCountsFile["tags"] = {}; + for (const id of Array.from(tags.keys()).sort()) { + const entry = tags.get(id)!; + const channels: Record<string, number> = {}; + for (const slug of Array.from(entry.channels.keys()).sort()) { + channels[slug] = entry.channels.get(slug)!; + } + out[id] = { count: entry.count, channels }; + } + return { version: TAG_COUNTS_VERSION, generatedAt, tags: out }; + }, + }; +} + +// The published /tags.json for one site: its effective vocabulary joined to its +// counts. Hidden tags and tags with no video ON THIS SITE are dropped — a +// global tag that matched nothing here does not get a chip here — and a site +// with nothing left ships no file at all (compose skips the write; a 404 is a +// legitimate empty state). +export function publishedTagsFrom( + defs: CuratedTagDef[], + counts: TagCountsFile | null, +): PublishedTags { + const tags = defs + .filter((def) => def.hidden !== true) + .map((def, i) => ({ def, i, entry: counts?.tags[def.id] })) + .filter((x) => (x.entry?.count ?? 0) > 0) + .sort((a, b) => { + const oa = a.def.order ?? a.i; + const ob = b.def.order ?? b.i; + if (oa !== ob) return oa - ob; + return a.def.id < b.def.id ? -1 : a.def.id > b.def.id ? 1 : 0; + }) + .map(({ def, entry }) => ({ + id: def.id, + label: def.label, + ...(def.group ? { group: def.group } : {}), + ...(def.groupLabel ? { groupLabel: def.groupLabel } : {}), + ...(def.color ? { color: def.color } : {}), + ...(def.order !== undefined ? { order: def.order } : {}), + count: entry!.count, + channels: entry!.channels, + })); + return { version: 1, tags }; +} diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -220,5 +220,11 @@ export function toDisplaySummary( ...(state === "available" ? {} : { state }), platform: t.platform, webpageUrl: t.webpageUrl, + // Curated tags ride through to the browse index, so a list can be narrowed + // to "streams where X collabs" without loading a transcript page. Omitted + // when empty for the same byte-identical reason as `state`. + ...(t.curatedTags && t.curatedTags.length > 0 + ? { curatedTags: t.curatedTags } + : {}), }; }