commit 403cdae39b1d27e3d8e2aed694a95221c7d0762e
parent a4f77c30bee93e9a25ede8b5082516408813b494
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 13:09:59 -0400
tags S1.3: derive curatedTags in the index, and notice when only a tag moved
Rule hits are not persisted anywhere, so nothing in the per-video mtime diff
can see that a rule was edited or a video pinned. common/controller/
curatedTagsIndex.ts is the answer: two hashes in the existing `meta` sub-DB —
curatedRulesHash (every def's id and order plus every ENABLED rule's kind,
pattern, channel scope and dates; presentation fields deliberately excluded, so
relabelling a tag does not re-page 30,000 videos) and curatedAssignHash — are
checked on EVERY build, with a per-video signature map beside them so an
assignment-only edit re-derives just the videos whose assignment moved, found
by a key-only byChannel scan per affected channel.
The per-video worker derives tags where the caption cues and the live_chat
track are already in hand; sums.put moved below the subs parse for that. The
re-apply pass re-derives straight out of LMDB — no video directory is read
twice — and reads cues only for videos a caption/chat rule is actually scoped
to. Whatever it changes flips sharedNeedsBuild, because the page writer's sha1
skip is what then keeps the untouched pages untouched.
SCHEMA_VERSION 14 (one-time re-derive; no new sub-DB, so no new clearAsync
line). toDisplaySummary carries curatedTags through, omitted when empty. Each
site's summaries stream now also accumulates per-tag counts into
tag-counts.json, and the site fingerprint includes the two hashes so a tag-only
edit cannot leave a site reporting "up to date" with yesterday's counts.
Measured on the real legal-mindset channel (518 videos, 1.37 M caption cues +
1.47 M chat cues) with the three seed rules, offline against the production
LMDB: 1.28 s warm — 1.03 s of it LMDB decode, 232 ms of actual regex. The
metadata rule alone is 18 ms.
Co-Authored-By: Claude Opus <noreply@anthropic.com>
Diffstat:
4 files changed, 936 insertions(+), 3 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -66,6 +66,7 @@ import { getSettings } from "../lib/settings";
import {
listSites,
siteDigestsDir,
+ siteIndexDir,
sitePostsDir,
siteSummariesDir,
siteSubsDir,
@@ -127,6 +128,14 @@ import {
} from "../lib/digests";
import { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME, effectiveDigest } from "../lib/digest";
import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
+import {
+ TAG_COUNTS_FILENAME,
+ applyCuratedTagsToSummary,
+ createTagCounter,
+ deriveCuratedTags,
+ loadCuratedTagsRuntime,
+ reapplyCuratedTags,
+} from "./curatedTagsIndex";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -140,7 +149,12 @@ import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
// (effectiveDigest of the machine sidecar + the human override file) plus a
// shared /digests/<slug>/ page tree. Bumped so existing indexes populate the
// new sub-DB — nothing re-derives it lazily.
-const SCHEMA_VERSION = 13;
+// v14: curated per-video tags. Summaries now carry `curatedTags` (derived from
+// transcripts/tags.json: rule hits folded with the operator's pins and
+// suppressions). Bumped so every existing record is re-derived once; after
+// that the two `meta` hashes (curatedRulesHash / curatedAssignHash) carry
+// invalidation, and no new sub-DB was added — see curatedTagsIndex.ts.
+const SCHEMA_VERSION = 14;
// Per-channel post stats, persisted so per-site aggregates survive a no-op
// rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat.
@@ -611,9 +625,18 @@ export async function buildIndex({
}
return true;
};
- const sharedNeedsBuild =
+ let sharedNeedsBuild =
anyMutations || schemaBumped || !(await sharedManifestsPresent());
+ // The curated-tag vocabulary + assignments, loaded and compiled ONCE for the
+ // whole build. Every per-video derivation below uses this; reapplyCuratedTags
+ // (after the worker loop) decides whether a tag-only edit has to re-derive
+ // records the mtime diff never looked at.
+ const curated = loadCuratedTagsRuntime(paths);
+ // indexKeyIds the worker derived fresh this build, so the re-apply pass does
+ // not do them a second time.
+ const curatedFresh = new Set<string>();
+
log(
`Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`,
);
@@ -694,7 +717,6 @@ export async function buildIndex({
byChannel.remove(indexToChannelKey(prev.indexKey));
}
- sums.put(indexKey, summary);
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
@@ -730,6 +752,29 @@ export async function buildIndex({
if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs);
else subs.remove(indexKey);
+ // Curated tags. Derived here, with the caption cues and the live-chat
+ // track already in hand, and stored ON the summary — which is why the
+ // sums.put waited for the subs parse. Omitted when empty, so an
+ // untagged corpus's pages stay byte-identical. Costs nothing at all
+ // when the corpus has no tags (curated.isNoop short-circuits).
+ if (!curated.isNoop) {
+ applyCuratedTagsToSummary(
+ summary,
+ deriveCuratedTags(curated, {
+ channelSlug: s.channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ captionCues: cueList,
+ chatCues: parsedSubs.find((t) => t.track === "live_chat")?.cues,
+ }),
+ );
+ }
+ curatedFresh.add(indexKeyId(indexKey));
+ sums.put(indexKey, summary);
+
// The derived layer. Only opened when the scan saw a sidecar, so the
// ~76k videos without one cost zero extra reads. What is stored is
// effectiveDigest(machine, overrides) — human corrections applied,
@@ -824,6 +869,26 @@ export async function buildIndex({
mtimes.remove(pathKey);
}
+ // Curated tags can change with no video on disk moving at all — a rule
+ // edited, a video pinned — and nothing in the mtime diff above can see that.
+ // The two meta hashes are therefore checked on EVERY build, and the videos
+ // whose tags moved are re-derived straight out of LMDB (no disk re-read).
+ // Anything that changed has to be re-paged, which is what the
+ // sharedNeedsBuild flip below buys: the page writer's sha1 skip then keeps
+ // the untouched pages untouched, so compose still copies only real changes.
+ const curatedReapply = reapplyCuratedTags({
+ runtime: curated,
+ sums,
+ cues,
+ subs,
+ byChannel,
+ meta,
+ alreadyFresh: curatedFresh,
+ allFresh: schemaBumped,
+ log,
+ });
+ if (curatedReapply.changed.length > 0) sharedNeedsBuild = true;
+
await sums.flushed;
await cues.flushed;
await subs.flushed;
@@ -1680,6 +1745,12 @@ export async function buildIndex({
groups: site.groups,
defaultGroupId: site.defaultGroupId,
channels: site.channels,
+ // A tag-only edit moves no site config and no video mtime, but it does
+ // change this site's tag-counts.json (and the curatedTags on its summary
+ // pages). Without these the site would report "up to date" and ship
+ // yesterday's counts.
+ curatedRules: curated.rulesHash,
+ curatedAssign: curated.assignHash,
});
const fpKey = `siteFp:${site.siteId}`;
if (
@@ -1739,12 +1810,18 @@ export async function buildIndex({
pageIndex++;
};
+ // Per-tag totals for THIS site, accumulated in the pass that is already
+ // streaming every member record — no second walk of the corpus. compose
+ // joins these to the site's effective vocabulary to write /tags.json.
+ const tagCounter = createTagCounter();
+
for (const { key, value } of sums.getRange({ reverse: true })) {
const ik = key as IndexKey;
if (!slugSet.has(ik[1])) continue;
const s = value as TranscriptSummary;
const acc = chan.get(ik[1]);
if (acc) acc.count++;
+ tagCounter.add(ik[1], s.curatedTags);
buffer.push(
toDisplaySummary(s, { state: stateByIndexKey.get(indexKeyId(ik)) }),
);
@@ -1753,6 +1830,14 @@ export async function buildIndex({
}
await flushPage();
+ // Written even when empty (a tiny `{tags:{}}`), so a site that loses its
+ // last tagged video does not keep serving a stale count file. compose
+ // decides from this whether to publish /tags.json at all.
+ await writeJsonAtomic(
+ path.join(siteIndexDir(paths, site.siteId), TAG_COUNTS_FILENAME),
+ tagCounter.file(new Date().toISOString()),
+ );
+
const expectedPages = new Set<string>();
for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i));
for (const name of await readdir(summariesOut).catch(() => [] as string[])) {
diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts
@@ -0,0 +1,379 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Cue } from "../lib/vtt";
+import type { TranscriptSummary } from "../lib/transcripts";
+import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags";
+import {
+ applyCuratedTagsToSummary,
+ createTagCounter,
+ curatedTagsRuntime,
+ deriveCuratedTags,
+ hashCuratedAssignments,
+ hashCuratedRules,
+ publishedTagsFrom,
+ reapplyCuratedTags,
+ META_ASSIGN_HASH,
+ META_ASSIGN_SIGS,
+ META_RULES_HASH,
+} from "./curatedTagsIndex";
+
+type IndexKey = [string, string, string];
+
+// Minimal stand-ins for the LMDB sub-DBs buildIndex hands the re-apply pass.
+// Keys are the same tuples; range scans only need to be correct, not fast.
+function makeDb<V>() {
+ const map = new Map<string, { key: unknown; value: V }>();
+ const k = (key: unknown) => JSON.stringify(key);
+ return {
+ map,
+ get(key: unknown): V | undefined {
+ return map.get(k(key))?.value;
+ },
+ put(key: unknown, value: V): void {
+ map.set(k(key), { key, value });
+ },
+ getRange(options?: { start?: unknown[] }): Iterable<{ key: unknown; value: V }> {
+ const start = options?.start as string[] | undefined;
+ const all = Array.from(map.values()).sort((a, b) =>
+ k(a.key) < k(b.key) ? -1 : 1,
+ );
+ if (!start) return all;
+ return all.filter((e) => (e.key as string[])[0] === start[0]);
+ },
+ };
+}
+
+function makeMeta() {
+ const map = new Map<string, unknown>();
+ return {
+ map,
+ get: (key: string) => map.get(key),
+ put: (key: string, value: unknown) => void map.set(key, value),
+ };
+}
+
+function summary(
+ channelSlug: string,
+ id: string,
+ uploadDate: string,
+ over: Partial<TranscriptSummary> = {},
+): TranscriptSummary {
+ return {
+ slug: `${channelSlug}/${id}`,
+ id,
+ channelSlug,
+ title: `Video ${id}`,
+ uploadDate,
+ duration: 100,
+ channel: channelSlug,
+ description: "",
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: `https://example.com/${id}`,
+ ...over,
+ };
+}
+
+const COLLAB_RULE: CuratedTagDef = {
+ id: "eva-collab",
+ label: "Collab",
+ order: 1,
+ rules: [{ id: "meta", kind: "metadata", pattern: "elfpire", enabled: true }],
+};
+
+function fixture(config: CuratedTagsConfig) {
+ const sums = makeDb<TranscriptSummary>();
+ const cues = makeDb<Cue[]>();
+ const subs = makeDb<{ track: string; cues: Cue[] }[]>();
+ const byChannel = makeDb<number>();
+ const meta = makeMeta();
+ const put = (s: TranscriptSummary) => {
+ const ik: IndexKey = [s.uploadDate, s.channelSlug, s.id];
+ sums.put(ik, s);
+ byChannel.put([s.channelSlug, s.uploadDate, s.id], 1);
+ return ik;
+ };
+ return { sums, cues, subs, byChannel, meta, put, runtime: curatedTagsRuntime(config) };
+}
+
+function cfg(
+ tags: CuratedTagDef[],
+ assignments: CuratedTagsConfig["assignments"] = {},
+): CuratedTagsConfig {
+ return { version: 1, tags, assignments };
+}
+
+// ─── hashes ───
+
+test("hashCuratedRules ignores presentation, reacts to derivation", () => {
+ const base = hashCuratedRules([COLLAB_RULE]);
+ assert.equal(
+ hashCuratedRules([{ ...COLLAB_RULE, label: "On mic", color: "#fff", hidden: true }]),
+ base,
+ "relabelling must not re-page the corpus",
+ );
+ assert.notEqual(
+ hashCuratedRules([
+ { ...COLLAB_RULE, rules: [{ id: "meta", kind: "metadata", pattern: "eva", enabled: true }] },
+ ]),
+ base,
+ );
+ assert.notEqual(hashCuratedRules([{ ...COLLAB_RULE, order: 9 }]), base);
+ assert.notEqual(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [{ ...COLLAB_RULE.rules![0], channels: ["legal-mindset"] }],
+ },
+ ]),
+ base,
+ );
+ // A disabled rule is not part of the derivation at all.
+ assert.equal(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [COLLAB_RULE.rules![0], { id: "off", kind: "caption", pattern: "x", enabled: false }],
+ },
+ ]),
+ base,
+ );
+});
+
+test("hashCuratedAssignments is order-insensitive", () => {
+ const a = hashCuratedAssignments({
+ "c/v1": { manual: ["b", "a"] },
+ "c/v2": { suppressed: ["z"] },
+ });
+ const b = hashCuratedAssignments({
+ "c/v2": { suppressed: ["z"] },
+ "c/v1": { manual: ["a", "b"] },
+ });
+ assert.equal(a, b);
+ assert.notEqual(a, hashCuratedAssignments({ "c/v1": { manual: ["a"] } }));
+});
+
+// ─── the summary field ───
+
+test("applyCuratedTagsToSummary omits the key when empty", () => {
+ const s = summary("c", "v", "20260101");
+ assert.equal(applyCuratedTagsToSummary(s, []), false);
+ assert.equal("curatedTags" in s, false);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), true);
+ assert.deepEqual(s.curatedTags, ["a"]);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), false, "idempotent");
+ assert.equal(applyCuratedTagsToSummary(s, []), true);
+ assert.equal("curatedTags" in s, false, "the key is deleted, not emptied");
+});
+
+// ─── derivation ───
+
+test("deriveCuratedTags folds rule hits with pins and suppressions", () => {
+ const runtime = curatedTagsRuntime(
+ cfg([COLLAB_RULE, { id: "manual-only", label: "Manual", order: 2 }], {
+ "lm/v2": { manual: ["manual-only"] },
+ "lm/v3": { suppressed: ["eva-collab"] },
+ }),
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v1", title: "with Elfpire" }),
+ ["eva-collab"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v2", title: "nothing" }),
+ ["manual-only"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v3", title: "with Elfpire" }),
+ [],
+ "a suppression kills a rule hit",
+ );
+});
+
+// ─── reapplyCuratedTags ───
+
+test("unchanged hashes do no work at all", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.meta.put(META_RULES_HASH, f.runtime.rulesHash);
+ f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash);
+ const res = reapplyCuratedTags({ ...f });
+ assert.deepEqual(res, {
+ rulesChanged: false,
+ assignmentsChanged: false,
+ changed: [],
+ examined: 0,
+ });
+});
+
+test("a rule change re-derives the whole corpus from LMDB and reports the movers", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" }));
+ f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" }));
+ const res = reapplyCuratedTags({ ...f });
+ assert.equal(res.rulesChanged, true);
+ assert.equal(res.examined, 3);
+ assert.deepEqual(
+ res.changed.map((k) => k[2]).sort(),
+ ["v1", "v3"],
+ );
+ assert.deepEqual(f.sums.get(["20260101", "lm", "v1"])!.curatedTags, ["eva-collab"]);
+ assert.equal(f.sums.get(["20260102", "lm", "v2"])!.curatedTags, undefined);
+ // The hashes are recorded, so a second pass is a no-op.
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+});
+
+test("removing the last rule strips the tag back off every record", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ reapplyCuratedTags({ ...f });
+ assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]);
+ const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) };
+ const res = reapplyCuratedTags(gone);
+ assert.deepEqual(res.changed.length, 1);
+ assert.equal(f.sums.get(ik)!.curatedTags, undefined);
+});
+
+test("an assignment-only change touches ONLY the videos whose assignment moved", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ for (let i = 0; i < 20; i++) {
+ f.put(summary("lm", `v${i}`, `202601${String(i + 10).padStart(2, "0")}`));
+ }
+ f.put(summary("other", "w1", "20260201"));
+ reapplyCuratedTags({ ...f }); // establish the baseline hashes
+
+ const pinned = {
+ ...f,
+ runtime: curatedTagsRuntime(
+ cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }),
+ ),
+ };
+ const res = reapplyCuratedTags(pinned);
+ assert.equal(res.rulesChanged, false);
+ assert.equal(res.assignmentsChanged, true);
+ // 21 videos in the corpus; exactly one was even looked at.
+ assert.equal(res.examined, 1);
+ assert.deepEqual(res.changed.map((k) => k[2]), ["v7"]);
+ assert.deepEqual(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, ["eva-collab"]);
+
+ // Clearing it again is also a one-video pass, and the key comes back off.
+ const cleared = { ...f, runtime: curatedTagsRuntime(cfg([COLLAB_RULE])) };
+ const res2 = reapplyCuratedTags(cleared);
+ assert.equal(res2.examined, 1);
+ assert.equal(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, undefined);
+});
+
+test("videos the worker already derived this build are not done twice", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({
+ ...f,
+ alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]),
+ });
+ assert.equal(res.examined, 0);
+ assert.deepEqual(res.changed, []);
+});
+
+test("a schema bump records the hashes and re-applies nothing", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({ ...f, allFresh: true });
+ assert.equal(res.changed.length, 0);
+ assert.equal(res.examined, 0);
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.deepEqual(f.meta.get(META_ASSIGN_SIGS), {});
+});
+
+test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => {
+ const capRule: CuratedTagDef = {
+ id: "eva-topic",
+ label: "Discussed",
+ rules: [
+ {
+ id: "cap",
+ kind: "caption",
+ pattern: "elfpire",
+ channels: ["lm"],
+ enabled: true,
+ },
+ ],
+ };
+ const chatRule: CuratedTagDef = {
+ id: "eva-in-chat",
+ label: "In chat",
+ rules: [{ id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true }],
+ };
+ const f = fixture(cfg([capRule, chatRule]));
+ const a = f.put(summary("lm", "v1", "20260101"));
+ const b = f.put(summary("other", "v2", "20260102"));
+ f.cues.put(a, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.cues.put(b, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.subs.put(a, [
+ { track: "en", cues: [{ start: 0, end: 1, text: "ElfpireEva: not chat" }] },
+ { track: "live_chat", cues: [{ start: 0, end: 1, text: "ElfpireEva: hi" }] },
+ ]);
+
+ reapplyCuratedTags({ ...f });
+ // Order follows the vocabulary (neither def sets `order`, so it is the
+ // config's own order), not the alphabet.
+ assert.deepEqual(f.sums.get(a)!.curatedTags, ["eva-topic", "eva-in-chat"]);
+ // The caption rule is scoped to `lm`, so the other channel gets nothing even
+ // though its cues would match.
+ assert.equal(f.sums.get(b)!.curatedTags, undefined);
+});
+
+// ─── counts + publication ───
+
+test("createTagCounter accumulates per tag and per channel", () => {
+ const c = createTagCounter();
+ c.add("lm", ["eva-collab", "eva-topic"]);
+ c.add("lm", ["eva-collab"]);
+ c.add("nux", ["eva-collab"]);
+ c.add("nux", undefined);
+ c.add("nux", []);
+ const file = c.file("2026-09-21T00:00:00Z");
+ assert.equal(file.version, 1);
+ assert.deepEqual(file.tags["eva-collab"], {
+ count: 3,
+ channels: { lm: 2, nux: 1 },
+ });
+ assert.deepEqual(file.tags["eva-topic"], { count: 1, channels: { lm: 1 } });
+});
+
+test("publishedTagsFrom drops hidden and zero-count tags and sorts by order", () => {
+ const defs: CuratedTagDef[] = [
+ { id: "b-tag", label: "B", order: 2, group: "eva", groupLabel: "Eva" },
+ { id: "a-tag", label: "A", order: 1, color: "#fff" },
+ { id: "hidden-tag", label: "Hidden", order: 0, hidden: true },
+ { id: "unused", label: "Unused", order: 3 },
+ ];
+ const counts = {
+ version: 1,
+ generatedAt: "",
+ tags: {
+ "a-tag": { count: 2, channels: { lm: 2 } },
+ "b-tag": { count: 1, channels: { nux: 1 } },
+ "hidden-tag": { count: 5, channels: { lm: 5 } },
+ },
+ };
+ const published = publishedTagsFrom(defs, counts);
+ assert.deepEqual(
+ published.tags.map((t) => t.id),
+ ["a-tag", "b-tag"],
+ );
+ assert.deepEqual(published.tags[0], {
+ id: "a-tag",
+ label: "A",
+ color: "#fff",
+ order: 1,
+ count: 2,
+ channels: { lm: 2 },
+ });
+ assert.equal(published.tags[1].groupLabel, "Eva");
+ // No counts at all → nothing to publish (compose then writes no file).
+ assert.deepEqual(publishedTagsFrom(defs, null).tags, []);
+});
diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts
@@ -0,0 +1,463 @@
+// Curated tags, as the index build sees them.
+//
+// A video's `curatedTags` is derived, never stored on disk beside the video:
+// it is (rule hits ∪ operator pins) − suppressions, folded fresh at every
+// build. Rule hits in particular are NOT persisted anywhere — edit a rule,
+// rebuild, done — which is why this module exists: nothing in the per-video
+// mtime diff can notice that a rule changed, so the build has to ask.
+//
+// Two hashes in the existing `meta` sub-DB answer that question:
+// curatedRulesHash the derivation-relevant shape of the vocabulary — every
+// def's id and order, and every ENABLED rule's kind,
+// pattern, channel scope and date range
+// curatedAssignHash every pin and suppression
+// Both are checked unconditionally on every build (a skipped check would make
+// a rule edit silently inert until some unrelated mtime moved), and alongside
+// them `curatedAssignSigs` keeps a per-video signature so an assignment-only
+// change can re-derive just the videos whose assignment actually moved instead
+// of the whole corpus.
+//
+// **Only the CORPUS vocabulary is evaluated here.** A site-only rule
+// (sites/<id>/tags.json) contributes its definition to that site's published
+// /tags.json, but it does not tag records: records are shared by every site
+// that carries the channel, so a per-site rule hit would have to live in a
+// per-site record. Promote a rule to transcripts/tags.json to make it bind.
+
+import { createHash } from "node:crypto";
+import type { Cue } from "../lib/vtt";
+import type { Paths } from "../lib/paths";
+import type { TranscriptSummary } from "../lib/transcripts";
+import {
+ compileTagRules,
+ effectiveTagsFor,
+ evaluateCompiledRules,
+ type CompiledTagRule,
+ type CompiledTagRules,
+ type CuratedTagAssignment,
+ type CuratedTagDef,
+ type CuratedTagsConfig,
+ type PublishedTags,
+ type TagRuleInput,
+} from "../lib/curatedTags";
+import { readGlobalTags } from "../lib/curatedTagsStore";
+
+export const META_RULES_HASH = "curatedRulesHash";
+export const META_ASSIGN_HASH = "curatedAssignHash";
+export const META_ASSIGN_SIGS = "curatedAssignSigs";
+
+export const TAG_COUNTS_FILENAME = "tag-counts.json";
+export const TAG_COUNTS_VERSION = 1;
+
+type IndexKey = [string, string, string];
+type ChannelKey = [string, string, string];
+
+function sha1(s: string): string {
+ return createHash("sha1").update(s).digest("hex");
+}
+
+// The shape of the vocabulary that can change a record. Presentation fields
+// (label, colour, group, hidden) are deliberately absent: relabelling a tag
+// must not re-page 30,000 videos.
+export function hashCuratedRules(defs: CuratedTagDef[]): string {
+ const shape = defs.map((def, i) => [
+ def.id,
+ def.order ?? i,
+ (def.rules ?? [])
+ .filter((r) => r.enabled !== false)
+ .map((r) => [
+ r.id,
+ r.kind,
+ r.pattern,
+ [...(r.channels ?? [])].sort(),
+ r.dateFrom ?? null,
+ r.dateTo ?? null,
+ ]),
+ ]);
+ return sha1(JSON.stringify(shape));
+}
+
+// One video's assignment, canonically. Also the per-key signature stored in
+// meta, so the diff is a string compare.
+export function assignmentSignature(a: CuratedTagAssignment): string {
+ return `${[...(a.manual ?? [])].sort().join(",")}|${[...(a.suppressed ?? [])].sort().join(",")}`;
+}
+
+export function assignmentSignatures(
+ assignments: Record<string, CuratedTagAssignment>,
+): Record<string, string> {
+ const out: Record<string, string> = {};
+ for (const key of Object.keys(assignments).sort()) {
+ out[key] = assignmentSignature(assignments[key]);
+ }
+ return out;
+}
+
+export function hashCuratedAssignments(
+ assignments: Record<string, CuratedTagAssignment>,
+): string {
+ return sha1(JSON.stringify(assignmentSignatures(assignments)));
+}
+
+// Everything the build needs, loaded and compiled ONCE per build.
+export type CuratedTagsRuntime = {
+ config: CuratedTagsConfig;
+ defs: CuratedTagDef[];
+ compiled: CompiledTagRules;
+ rulesHash: string;
+ assignHash: string;
+ sigs: Record<string, string>;
+ // True when the corpus has no rules and no assignments at all — the state of
+ // every untagged install, and the cue to do no work whatsoever.
+ isNoop: boolean;
+};
+
+export function loadCuratedTagsRuntime(paths: Paths): CuratedTagsRuntime {
+ return curatedTagsRuntime(readGlobalTags(paths));
+}
+
+export function curatedTagsRuntime(config: CuratedTagsConfig): CuratedTagsRuntime {
+ const compiled = compileTagRules(config.tags);
+ return {
+ config,
+ defs: config.tags,
+ compiled,
+ rulesHash: hashCuratedRules(config.tags),
+ assignHash: hashCuratedAssignments(config.assignments),
+ sigs: assignmentSignatures(config.assignments),
+ isNoop:
+ compiled.isEmpty && Object.keys(config.assignments).length === 0,
+ };
+}
+
+// Cheap pre-checks, so a video on a channel no rule scopes to never pays for a
+// cue read. Mirrors the scoping evaluateCompiledRules does internally.
+function scopeAllows(
+ rule: CompiledTagRule,
+ channelSlug: string,
+ uploadDate: string | undefined,
+): boolean {
+ if (rule.channels && !rule.channels.has(channelSlug)) return false;
+ if (rule.dateFrom || rule.dateTo) {
+ const date = (uploadDate ?? "").replace(/\D/g, "");
+ if (!date) return false;
+ if (rule.dateFrom && date < rule.dateFrom) return false;
+ if (rule.dateTo && date > rule.dateTo) return false;
+ }
+ return true;
+}
+
+export function needsCaptionCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.caption.some((r) => scopeAllows(r, channelSlug, uploadDate));
+}
+
+export function needsChatCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.chatAuthor.some((r) =>
+ scopeAllows(r, channelSlug, uploadDate),
+ );
+}
+
+// The whole fold for one video: rule hits, then the operator's pins and
+// suppressions over the top.
+export function deriveCuratedTags(
+ runtime: CuratedTagsRuntime,
+ input: TagRuleInput,
+): string[] {
+ const hits = evaluateCompiledRules(input, runtime.compiled);
+ const assignment = runtime.config.assignments[`${input.channelSlug}/${input.id}`];
+ if (hits.length === 0 && !assignment) return [];
+ return effectiveTagsFor(hits, assignment, runtime.defs);
+}
+
+// Set/clear `curatedTags` on a summary in place. Returns true when the stored
+// value actually changed — omitted-when-empty is the contract, so an untagged
+// record must come out byte-identical to the one already on disk.
+export function applyCuratedTagsToSummary(
+ summary: TranscriptSummary,
+ tags: string[],
+): boolean {
+ const before = summary.curatedTags;
+ if (tags.length === 0) {
+ if (before === undefined) return false;
+ delete summary.curatedTags;
+ return true;
+ }
+ if (
+ before &&
+ before.length === tags.length &&
+ before.every((t, i) => t === tags[i])
+ ) {
+ return false;
+ }
+ summary.curatedTags = tags;
+ return true;
+}
+
+// ─── re-derivation ───
+
+type SumsDb = {
+ get(key: IndexKey): TranscriptSummary | undefined;
+ put(key: IndexKey, value: TranscriptSummary): unknown;
+ getRange(options?: unknown): Iterable<{ key: unknown; value: unknown }>;
+};
+type CuesDb = { get(key: IndexKey): Cue[] | undefined };
+type SubsDb = { get(key: IndexKey): { track: string; cues: Cue[] }[] | undefined };
+type ByChannelDb = {
+ getRange(options: unknown): Iterable<{ key: unknown }>;
+};
+type MetaDb = {
+ get(key: string): unknown;
+ put(key: string, value: unknown): unknown;
+};
+
+export type ReapplyOptions = {
+ runtime: CuratedTagsRuntime;
+ sums: SumsDb;
+ cues: CuesDb;
+ subs: SubsDb;
+ byChannel: ByChannelDb;
+ meta: MetaDb;
+ // indexKeyIds the per-video worker already derived this build. Skipped here
+ // so a rule edit landing in the same build as new videos doesn't do them
+ // twice.
+ alreadyFresh?: Set<string>;
+ // A schema bump cleared the cache and the worker re-derived everything, so
+ // there is nothing to re-apply — just record the hashes.
+ allFresh?: boolean;
+ log?: (msg: string) => void;
+};
+
+export type ReapplyResult = {
+ rulesChanged: boolean;
+ assignmentsChanged: boolean;
+ // Videos whose stored curatedTags moved. Their channels must be re-paged.
+ changed: IndexKey[];
+ // Videos examined (the cost).
+ examined: number;
+};
+
+function indexKeyId(k: IndexKey): string {
+ return `${k[0]}\x00${k[1]}\x00${k[2]}`;
+}
+
+function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[] | undefined {
+ if (!stored) return undefined;
+ for (const t of stored) if (t.track === "live_chat") return t.cues;
+ return undefined;
+}
+
+// Re-derive `curatedTags` for the videos a tag change can reach, WITHOUT
+// re-reading a single video directory: cues and live chat come back out of the
+// LMDB sub-DBs the build already populated.
+//
+// rules changed every video is a candidate (a new pattern can match
+// anything), but cues are read only for the videos a
+// caption/chat rule is actually scoped to
+// assignments changed only the videos whose assignment signature moved —
+// found by a key-only range scan of `byChannel` per
+// affected channel, which is the cheap idiom
+//
+// Returns the changed index keys so the caller can force those pages to be
+// rewritten: nothing in the mtime diff knows this happened.
+export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
+ const { runtime, sums, cues, subs, byChannel, meta } = opts;
+ const log = opts.log ?? (() => {});
+ const prevRules = meta.get(META_RULES_HASH) as string | undefined;
+ const prevAssign = meta.get(META_ASSIGN_HASH) as string | undefined;
+ const prevSigs = (meta.get(META_ASSIGN_SIGS) as Record<string, string> | undefined) ?? {};
+ const rulesChanged = prevRules !== runtime.rulesHash;
+ const assignmentsChanged = prevAssign !== runtime.assignHash;
+
+ const record = () => {
+ meta.put(META_RULES_HASH, runtime.rulesHash);
+ meta.put(META_ASSIGN_HASH, runtime.assignHash);
+ meta.put(META_ASSIGN_SIGS, runtime.sigs);
+ };
+
+ if (!rulesChanged && !assignmentsChanged) {
+ return { rulesChanged, assignmentsChanged, changed: [], examined: 0 };
+ }
+ if (opts.allFresh) {
+ record();
+ log(
+ `curated tags: rules ${runtime.rulesHash.slice(0, 8)}, assignments ${runtime.assignHash.slice(0, 8)} (full rebuild, nothing to re-apply)`,
+ );
+ return { rulesChanged, assignmentsChanged, changed: [], examined: 0 };
+ }
+
+ const changed: IndexKey[] = [];
+ let examined = 0;
+ const skip = opts.alreadyFresh;
+
+ // `known` is the value the cursor already decoded, when there is one — a
+ // second sums.get() per video would double the msgpack decodes on the
+ // whole-corpus path.
+ const rederive = (indexKey: IndexKey, known?: TranscriptSummary): void => {
+ if (skip?.has(indexKeyId(indexKey))) return;
+ const summary = known ?? sums.get(indexKey);
+ if (!summary) return;
+ examined++;
+ const channelSlug = indexKey[1];
+ const tags = deriveCuratedTags(runtime, {
+ channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ // Only read what a scoped rule can actually use. On an untagged corpus,
+ // and on every channel no rule names, this reads nothing.
+ captionCues: needsCaptionCues(runtime, channelSlug, summary.uploadDate)
+ ? cues.get(indexKey)
+ : undefined,
+ chatCues: needsChatCues(runtime, channelSlug, summary.uploadDate)
+ ? chatCuesOf(subs.get(indexKey))
+ : undefined,
+ });
+ if (applyCuratedTagsToSummary(summary, tags)) {
+ sums.put(indexKey, summary);
+ changed.push(indexKey);
+ }
+ };
+
+ if (rulesChanged) {
+ // Every video is a candidate. This walks `sums` keys only; the value comes
+ // from the same cursor, and cue reads are gated per video above.
+ for (const { key, value } of sums.getRange()) {
+ rederive(key as IndexKey, value as TranscriptSummary);
+ }
+ } else {
+ // Assignment-only change: the symmetric difference of the signatures.
+ const movedKeys = new Set<string>();
+ for (const [key, sig] of Object.entries(runtime.sigs)) {
+ if (prevSigs[key] !== sig) movedKeys.add(key);
+ }
+ for (const key of Object.keys(prevSigs)) {
+ if (!(key in runtime.sigs)) movedKeys.add(key);
+ }
+ // Group by channel so each channel costs one key-only range scan.
+ const byChannelSlug = new Map<string, Set<string>>();
+ for (const key of movedKeys) {
+ const at = key.indexOf("/");
+ if (at <= 0) continue;
+ const slug = key.slice(0, at);
+ const id = key.slice(at + 1);
+ const set = byChannelSlug.get(slug) ?? new Set<string>();
+ set.add(id);
+ byChannelSlug.set(slug, set);
+ }
+ for (const [slug, ids] of byChannelSlug) {
+ for (const { key } of byChannel.getRange({
+ start: [slug],
+ end: [slug, ""],
+ values: false,
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== slug) continue;
+ if (!ids.has(ck[2])) continue;
+ rederive([ck[1], ck[0], ck[2]]);
+ }
+ }
+ }
+
+ record();
+ const what = [
+ rulesChanged ? `rules ${runtime.rulesHash.slice(0, 8)} (changed)` : null,
+ assignmentsChanged
+ ? `assignments ${runtime.assignHash.slice(0, 8)} (changed)`
+ : null,
+ ]
+ .filter(Boolean)
+ .join(", ");
+ log(
+ `curated tags: ${what}, examined ${examined}, re-derived ${changed.length}`,
+ );
+ return { rulesChanged, assignmentsChanged, changed, examined };
+}
+
+// ─── per-site counts ───
+
+export type TagCountsFile = {
+ version: number;
+ generatedAt: string;
+ // tag id -> total on this site + the per-channel breakdown.
+ tags: Record<string, { count: number; channels: Record<string, number> }>;
+};
+
+export function emptyTagCounts(): TagCountsFile {
+ return { version: TAG_COUNTS_VERSION, generatedAt: "", tags: {} };
+}
+
+// Accumulator for the per-site summaries stream: one add() per record, no
+// second pass over the corpus.
+export function createTagCounter() {
+ const tags = new Map<string, { count: number; channels: Map<string, number> }>();
+ return {
+ add(channelSlug: string, curatedTags: string[] | undefined): void {
+ if (!curatedTags || curatedTags.length === 0) return;
+ for (const id of curatedTags) {
+ let entry = tags.get(id);
+ if (!entry) {
+ entry = { count: 0, channels: new Map() };
+ tags.set(id, entry);
+ }
+ entry.count++;
+ entry.channels.set(channelSlug, (entry.channels.get(channelSlug) ?? 0) + 1);
+ }
+ },
+ get size(): number {
+ return tags.size;
+ },
+ file(generatedAt: string): TagCountsFile {
+ const out: TagCountsFile["tags"] = {};
+ for (const id of Array.from(tags.keys()).sort()) {
+ const entry = tags.get(id)!;
+ const channels: Record<string, number> = {};
+ for (const slug of Array.from(entry.channels.keys()).sort()) {
+ channels[slug] = entry.channels.get(slug)!;
+ }
+ out[id] = { count: entry.count, channels };
+ }
+ return { version: TAG_COUNTS_VERSION, generatedAt, tags: out };
+ },
+ };
+}
+
+// The published /tags.json for one site: its effective vocabulary joined to its
+// counts. Hidden tags and tags with no video ON THIS SITE are dropped — a
+// global tag that matched nothing here does not get a chip here — and a site
+// with nothing left ships no file at all (compose skips the write; a 404 is a
+// legitimate empty state).
+export function publishedTagsFrom(
+ defs: CuratedTagDef[],
+ counts: TagCountsFile | null,
+): PublishedTags {
+ const tags = defs
+ .filter((def) => def.hidden !== true)
+ .map((def, i) => ({ def, i, entry: counts?.tags[def.id] }))
+ .filter((x) => (x.entry?.count ?? 0) > 0)
+ .sort((a, b) => {
+ const oa = a.def.order ?? a.i;
+ const ob = b.def.order ?? b.i;
+ if (oa !== ob) return oa - ob;
+ return a.def.id < b.def.id ? -1 : a.def.id > b.def.id ? 1 : 0;
+ })
+ .map(({ def, entry }) => ({
+ id: def.id,
+ label: def.label,
+ ...(def.group ? { group: def.group } : {}),
+ ...(def.groupLabel ? { groupLabel: def.groupLabel } : {}),
+ ...(def.color ? { color: def.color } : {}),
+ ...(def.order !== undefined ? { order: def.order } : {}),
+ count: entry!.count,
+ channels: entry!.channels,
+ }));
+ return { version: 1, tags };
+}
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -220,5 +220,11 @@ export function toDisplaySummary(
...(state === "available" ? {} : { state }),
platform: t.platform,
webpageUrl: t.webpageUrl,
+ // Curated tags ride through to the browse index, so a list can be narrowed
+ // to "streams where X collabs" without loading a transcript page. Omitted
+ // when empty for the same byte-identical reason as `state`.
+ ...(t.curatedTags && t.curatedTags.length > 0
+ ? { curatedTags: t.curatedTags }
+ : {}),
};
}