commit 5d3464634ab3bc50a4bff3e71353e514ab0de7aa
parent 6bb8e59468c31cbd636463027dc04ad20fc36601
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 13:44:07 -0400
Merge branch 'main' into tags/site
Diffstat:
25 files changed, 2468 insertions(+), 27 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -63,6 +63,7 @@ yarn-error.log*
/export/public/chart-templates.json
/export/public/search-aliases.json
/export/public/duplicates.json
+/export/public/tags.json
/export/public/site.json
/export/public/_headers
/export/public/sw.js
diff --git a/AGENTS.md b/AGENTS.md
@@ -198,6 +198,19 @@ apps*, not [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md), which is about *building sites*
| `transcripts/index.mdb` | The LMDB transcript index. Key-only range scans over its `byChannel` sub-DB are cheap; see `common/controller/recencyIndex.ts`. |
| `transcripts/saved-videos/` | Persisted source-video store. |
| `transcripts/search-aliases.json`, `duplicates*.json` | Corpus-wide curated data. |
+| `transcripts/tags.json` | **Curated per-video tags** — the cross-channel vocabulary AND every assignment. Curated data: never hand-edited, never a scratch file. |
+
+**Curated tags are written through ONE path, and it is not your text editor.**
+`transcripts/tags.json` holds the operator's vocabulary (`eva-collab`, rules that
+re-evaluate at index build) and every pin/suppression with its provenance; a site's
+`sites/<id>/tags.json` is presentation only. Every writer — editor UI, `pnpm ops`, umtool —
+goes through `applyTagAssignments` in `common/lib/curatedTagsStore.ts`, which validates,
+records who claimed what, and writes the file once, atomically. Hand-editing it loses
+provenance and races whatever is running. Tests use temp dirs.
+
+Three unrelated things in this repo are called "tags": the yt-dlp keywords on a record
+(`TranscriptSummary.tags`), AI digest topic tags, and these. The record field for these is
+**`curatedTags`**, never `tags` — see `plans/FACTS.md`, "Naming hazards".
**The roots a channel's media may be moved to are named entities**, `settings.storage.locations`
(id, label, root, `autoRepoint`, and the volume UUID learned at the last probe) — managed on
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -35,6 +35,13 @@ import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescr
import { shipsPwa } from "../lib/archive/contract";
import { renderHeadersFile } from "../lib/archive/headers";
import { effectiveSiteAliases } from "../lib/aliasesStore";
+import { effectiveSiteTags } from "../lib/curatedTagsStore";
+import { TAGS_FILENAME } from "../lib/curatedTags";
+import {
+ TAG_COUNTS_FILENAME,
+ publishedTagsFrom,
+ type TagCountsFile,
+} from "../controller/curatedTagsIndex";
import {
buildSiteCorpus,
renderSiteLlmsTxt,
@@ -137,10 +144,16 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
/* no digests manifest for this site */
}
+ // Whether this site published a curated vocabulary. Presence of the file is
+ // the fact — it was written (or removed) just above, by the same rule that
+ // decides whether any tag means anything on this site.
+ const hasTags = await exists(path.join(paths.exportPublicDir, TAGS_FILENAME));
+
const corpus = buildSiteCorpus(descriptor, {
hasArchives,
postCounts,
digestCounts,
+ hasTags,
});
await writeFile(
path.join(paths.exportPublicDir, "corpus.json"),
@@ -751,6 +764,37 @@ async function main(): Promise<void> {
JSON.stringify({ aliases }),
);
+ // --- curated tags (corpus vocabulary + this site's overlay + its counts) ---
+ // The vocabulary is read from source at compose time, like the aliases; the
+ // counts come from the per-site tag-counts.json build:index wrote while it
+ // was already streaming this site's summaries.
+ //
+ // What ships is only what means something HERE: hidden tags are dropped, and
+ // so is any tag with no video on this site — a corpus-wide tag that matched
+ // nothing here gets no chip here. Nothing left → no file at all, and the 404
+ // is the empty state (the same contract duplicates.json has). The stale copy
+ // is removed in that case, so a site that loses its last tagged video stops
+ // advertising tags on the very next build.
+ const tagDefs = effectiveSiteTags(paths, siteId);
+ let tagCounts: TagCountsFile | null = null;
+ try {
+ tagCounts = JSON.parse(
+ await readFile(
+ path.join(paths.exportSitesIndexDir, siteId, TAG_COUNTS_FILENAME),
+ "utf8",
+ ),
+ ) as TagCountsFile;
+ } catch {
+ /* no counts staged for this site yet — treated as "nothing tagged" */
+ }
+ const publishedTags = publishedTagsFrom(tagDefs, tagCounts);
+ const tagsDest = path.join(paths.exportPublicDir, TAGS_FILENAME);
+ if (publishedTags.tags.length > 0) {
+ await writeFile(tagsDest, JSON.stringify(publishedTags));
+ } else {
+ await rm(tagsDest, { force: true });
+ }
+
// --- duplicate-shorts report (global → site-filtered) ---
// The detector writes one global duplicates.json over the whole channel pool;
// each site only serves its own channels, so filter clusters to the site's
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -66,6 +66,7 @@ import { getSettings } from "../lib/settings";
import {
listSites,
siteDigestsDir,
+ siteIndexDir,
sitePostsDir,
siteSummariesDir,
siteSubsDir,
@@ -127,6 +128,15 @@ import {
} from "../lib/digests";
import { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME, effectiveDigest } from "../lib/digest";
import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
+import {
+ TAG_COUNTS_FILENAME,
+ applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
+ createTagCounter,
+ deriveCuratedTags,
+ loadCuratedTagsRuntime,
+ reapplyCuratedTags,
+} from "./curatedTagsIndex";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -140,6 +150,18 @@ import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
// (effectiveDigest of the machine sidecar + the human override file) plus a
// shared /digests/<slug>/ page tree. Bumped so existing indexes populate the
// new sub-DB — nothing re-derives it lazily.
+// NOT a v14 — curated tags (`curatedTags` on every summary, derived from
+// transcripts/tags.json) deliberately did NOT bump this, and the reason is
+// worth keeping: every prior bump existed because something would otherwise
+// never be derived at all (v13 populates a new sub-DB; nothing re-derives it
+// lazily). Curated tags DO re-derive lazily, by design. `curatedRulesHash` is
+// absent on an index built before them, so it can never equal the current hash
+// and `reapplyCuratedTags` re-derives every record on the first build after
+// this ships — measured at ~80 s across 30k videos, straight out of LMDB, and
+// zero page writes on an untagged corpus because an empty derivation leaves
+// each summary byte-identical. A bump would instead wipe the cache and re-read
+// 76k video directories to reach exactly the same state. No sub-DB was added,
+// so the clearAsync() enumeration above is unchanged too.
const SCHEMA_VERSION = 13;
// Per-channel post stats, persisted so per-site aggregates survive a no-op
@@ -611,9 +633,18 @@ export async function buildIndex({
}
return true;
};
- const sharedNeedsBuild =
+ let sharedNeedsBuild =
anyMutations || schemaBumped || !(await sharedManifestsPresent());
+ // The curated-tag vocabulary + assignments, loaded and compiled ONCE for the
+ // whole build. Every per-video derivation below uses this; reapplyCuratedTags
+ // (after the worker loop) decides whether a tag-only edit has to re-derive
+ // records the mtime diff never looked at.
+ const curated = loadCuratedTagsRuntime(paths);
+ // indexKeyIds the worker derived fresh this build, so the re-apply pass does
+ // not do them a second time.
+ const curatedFresh = new Set<string>();
+
log(
`Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`,
);
@@ -694,7 +725,6 @@ export async function buildIndex({
byChannel.remove(indexToChannelKey(prev.indexKey));
}
- sums.put(indexKey, summary);
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
@@ -730,6 +760,29 @@ export async function buildIndex({
if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs);
else subs.remove(indexKey);
+ // Curated tags. Derived here, with the caption cues and the live-chat
+ // track already in hand, and stored ON the summary — which is why the
+ // sums.put waited for the subs parse. Omitted when empty, so an
+ // untagged corpus's pages stay byte-identical. Costs nothing at all
+ // when the corpus has no tags (curated.isNoop short-circuits).
+ if (!curated.isNoop) {
+ applyCuratedTagsToSummary(
+ summary,
+ deriveCuratedTags(curated, {
+ channelSlug: s.channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ captionCues: cueList,
+ chatCues: parsedSubs.find((t) => t.track === "live_chat")?.cues,
+ }),
+ );
+ }
+ curatedFresh.add(indexKeyId(indexKey));
+ sums.put(indexKey, summary);
+
// The derived layer. Only opened when the scan saw a sidecar, so the
// ~76k videos without one cost zero extra reads. What is stored is
// effectiveDigest(machine, overrides) — human corrections applied,
@@ -824,6 +877,34 @@ export async function buildIndex({
mtimes.remove(pathKey);
}
+ // Curated tags can change with no video on disk moving at all — a rule
+ // edited, a video pinned — and nothing in the mtime diff above can see that.
+ // The two meta hashes are therefore checked on EVERY build, and the videos
+ // whose tags moved are re-derived straight out of LMDB (no disk re-read).
+ // Anything that changed has to be re-paged, which is what the
+ // sharedNeedsBuild flip below buys: the page writer's sha1 skip then keeps
+ // the untouched pages untouched, so compose still copies only real changes.
+ const curatedReapply = reapplyCuratedTags({
+ runtime: curated,
+ sums,
+ cues,
+ subs,
+ byChannel,
+ meta,
+ alreadyFresh: curatedFresh,
+ allFresh: schemaBumped,
+ log,
+ });
+ // `pagesPending` also covers a PREVIOUS build that re-derived records and was
+ // then interrupted before writing the pages: its hashes are already stored,
+ // so without the flag nothing would ever ask for those shards again and they
+ // would ship stale curatedTags for ever.
+ if (curatedReapply.pagesPending) sharedNeedsBuild = true;
+ // Commit the flag (and the re-derived summaries) BEFORE the page build, so an
+ // interrupt during it finds the debt recorded rather than lost in a pending
+ // write batch.
+ await meta.flushed;
+
await sums.flushed;
await cues.flushed;
await subs.flushed;
@@ -1285,6 +1366,13 @@ export async function buildIndex({
log(`Shared sub pages up to date; ${channelStats.size} channels with subs.`);
}
+ // Both shared trees are now written (or were already current), so the
+ // curated-tag page debt is settled. Deliberately AFTER the page build and not
+ // beside the hashes: an interrupt anywhere above must leave the flag standing
+ // so the next build rewrites the shards.
+ clearCuratedPagesPending(meta);
+ await meta.flushed;
+
// ---------------------------------------------------------------------------
// The social-post corpus: a parallel per-channel page tree beside transcripts
// and subs. Social channels have no `data/` dir and so never appear in the
@@ -1680,6 +1768,12 @@ export async function buildIndex({
groups: site.groups,
defaultGroupId: site.defaultGroupId,
channels: site.channels,
+ // A tag-only edit moves no site config and no video mtime, but it does
+ // change this site's tag-counts.json (and the curatedTags on its summary
+ // pages). Without these the site would report "up to date" and ship
+ // yesterday's counts.
+ curatedRules: curated.rulesHash,
+ curatedAssign: curated.assignHash,
});
const fpKey = `siteFp:${site.siteId}`;
if (
@@ -1739,12 +1833,18 @@ export async function buildIndex({
pageIndex++;
};
+ // Per-tag totals for THIS site, accumulated in the pass that is already
+ // streaming every member record — no second walk of the corpus. compose
+ // joins these to the site's effective vocabulary to write /tags.json.
+ const tagCounter = createTagCounter();
+
for (const { key, value } of sums.getRange({ reverse: true })) {
const ik = key as IndexKey;
if (!slugSet.has(ik[1])) continue;
const s = value as TranscriptSummary;
const acc = chan.get(ik[1]);
if (acc) acc.count++;
+ tagCounter.add(ik[1], s.curatedTags);
buffer.push(
toDisplaySummary(s, { state: stateByIndexKey.get(indexKeyId(ik)) }),
);
@@ -1753,6 +1853,14 @@ export async function buildIndex({
}
await flushPage();
+ // Written even when empty (a tiny `{tags:{}}`), so a site that loses its
+ // last tagged video does not keep serving a stale count file. compose
+ // decides from this whether to publish /tags.json at all.
+ await writeJsonAtomic(
+ path.join(siteIndexDir(paths, site.siteId), TAG_COUNTS_FILENAME),
+ tagCounter.file(new Date().toISOString()),
+ );
+
const expectedPages = new Set<string>();
for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i));
for (const name of await readdir(summariesOut).catch(() => [] as string[])) {
diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts
@@ -0,0 +1,287 @@
+// Integration: a curated-tag edit reaching the published files, through the
+// REAL buildIndex and the real compose script, over a temp corpus.
+//
+// The unit tests around reapplyCuratedTags prove the derivation. This proves
+// the thing that actually breaks in production: a tag edit moves no video and
+// no mtime, so every incremental short-circuit in the build has a chance to
+// decide nothing happened and ship yesterday's shards.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/curatedTagsBuild.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { execFile } from "node:child_process";
+import { mkdtempSync, writeFileSync, mkdirSync, readFileSync, existsSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { promisify } from "node:util";
+import { open } from "lmdb";
+
+// getPaths() is lazy and cached, and nothing above calls it at import time, so
+// pointing the whole path graph at a temp root here is enough to isolate this
+// file's process from the real corpus. Every path the build touches is derived
+// from these three.
+const ROOT = mkdtempSync(path.join(tmpdir(), "curated-tags-build-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public");
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { writeGlobalTags } = await import("../lib/curatedTagsStore");
+const {
+ curatedTagsRuntime,
+ reapplyCuratedTags,
+ META_PAGES_PENDING,
+} = await import("./curatedTagsIndex");
+
+const paths = getPaths();
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+const exec = promisify(execFile);
+
+// compose-site.ts is a SCRIPT (it runs main() on import), so it is exercised
+// the way the build runs it: as a child process with SITE_ID, inheriting this
+// file's temp-root env. `.bin/tsx` is a shell wrapper, so the CLI entry is
+// invoked directly.
+const composeSite = () =>
+ exec(
+ process.execPath,
+ [
+ path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"),
+ path.join(REPO, "common", "bin", "compose-site.ts"),
+ ],
+ { env: { ...process.env, SITE_ID: SITE }, cwd: REPO },
+ );
+
+const CHANNEL = "test-channel";
+const SITE = "testsite";
+const HIT = "vid-hit";
+const MISS = "vid-miss";
+
+function writeJson(file: string, value: unknown): void {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+}
+
+function videoDir(id: string): string {
+ return path.join(paths.channelsDir, CHANNEL, "data", id);
+}
+
+function seedCorpus(): void {
+ writeFileSync(paths.settingsFile, JSON.stringify({}));
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Test Channel",
+ url: "https://www.youtube.com/@example/videos",
+ });
+ const video = (id: string, title: string, uploadDate: string) => {
+ writeJson(path.join(videoDir(id), "metadata.info.json"), {
+ id,
+ title,
+ channel: "Test Channel",
+ upload_date: uploadDate,
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(
+ path.join(videoDir(id), "transcript.en.vtt"),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nA line of transcript.\n",
+ );
+ };
+ video(HIT, "Stream with Elfpire Eva", "20260102");
+ video(MISS, "An ordinary stream", "20260101");
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+}
+
+const build = () => buildIndex({ paths, onLog: () => {} });
+
+function transcriptPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSharedTranscriptsDir, CHANNEL, "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function summariesPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "summaries", "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function tagCounts(): { tags: Record<string, { count: number; channels: Record<string, number> }> } {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "tag-counts.json"),
+ "utf8",
+ ),
+ );
+}
+
+const recordIn = (page: { id: string; curatedTags?: string[] }[], id: string) =>
+ page.find((r) => r.id === id)!;
+
+const EVA = {
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [
+ {
+ id: "meta",
+ kind: "metadata" as const,
+ pattern: "elfpire",
+ enabled: true,
+ },
+ ],
+ },
+ ],
+ assignments: {},
+};
+
+// The whole slice, in one sequence, because each step's fixture is the previous
+// step's output. node:test runs top-level tests in order within a file.
+
+test("build 1: an untagged corpus ships no curatedTags key at all", async () => {
+ seedCorpus();
+ const res = await build();
+ assert.equal(res.totalCount, 2);
+ for (const page of [transcriptPage(), summariesPage()]) {
+ for (const record of page) {
+ assert.equal(
+ "curatedTags" in record,
+ false,
+ "omitted-when-empty, or every untagged corpus re-pages on upgrade",
+ );
+ }
+ }
+ assert.deepEqual(tagCounts().tags, {});
+});
+
+test("build 2: a rule edit with NO mtime change reaches the shards AND the site pages", async () => {
+ // The only thing that changes between build 1 and build 2. No video
+ // directory is touched, so the mtime diff sees +0 ~0 -0.
+ writeGlobalTags(paths, EVA);
+
+ const res = await build();
+ assert.equal(res.added, 0);
+ assert.equal(res.changed, 0);
+ assert.equal(res.removed, 0);
+
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(transcriptPage(), MISS), false);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(summariesPage(), MISS), false);
+ assert.deepEqual(tagCounts().tags["eva-collab"], {
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ });
+});
+
+test("compose 1: the site publishes /tags.json and corpus.json points at it", async () => {
+ await composeSite();
+ const tagsFile = path.join(paths.exportPublicDir, "tags.json");
+ const published = JSON.parse(readFileSync(tagsFile, "utf8"));
+ assert.deepEqual(published.tags, [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ },
+ ]);
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.spec, 4);
+ assert.equal(corpus.tags.videoField, "curatedTags");
+ assert.match(corpus.tags.url, /\/tags\.json$/);
+ assert.match(
+ readFileSync(path.join(paths.exportPublicDir, "llms.txt"), "utf8"),
+ /tags\.json/,
+ );
+});
+
+test("build 3 + compose 2: dropping the vocabulary removes the file and the pointer", async () => {
+ writeGlobalTags(paths, { version: 1, tags: [], assignments: {} });
+ await build();
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+ assert.deepEqual(tagCounts().tags, {});
+
+ await composeSite();
+ assert.equal(
+ existsSync(path.join(paths.exportPublicDir, "tags.json")),
+ false,
+ "a site with nothing to say must stop serving the file, not serve an empty one",
+ );
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.tags, undefined);
+});
+
+test("an interrupted build's page debt is paid on the next build", async () => {
+ // Reconstruct exactly what a Ctrl-C between the re-derivation and the page
+ // build leaves behind: records re-derived in LMDB, hashes recorded,
+ // curatedPagesPending set — and pages still holding the OLD content.
+ writeGlobalTags(paths, EVA);
+ const runtime = curatedTagsRuntime(EVA);
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const sums = root.openDB<unknown, [string, string, string]>({ name: "sums", encoding: "msgpack" });
+ const cues = root.openDB<unknown, [string, string, string]>({ name: "cues", encoding: "msgpack" });
+ const subs = root.openDB<unknown, [string, string, string]>({ name: "subs", encoding: "msgpack" });
+ const byChannel = root.openDB<number, [string, string, string]>({ name: "byChannel", encoding: "msgpack" });
+ const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ const res = reapplyCuratedTags({
+ runtime,
+ // The sub-DB handles are structurally what the pass needs.
+ sums: sums as never,
+ cues: cues as never,
+ subs: subs as never,
+ byChannel: byChannel as never,
+ meta: meta as never,
+ });
+ assert.equal(res.changedCount, 1, "the interrupted build did re-derive a record");
+ await sums.flushed;
+ await meta.flushed;
+ assert.equal(meta.get(META_PAGES_PENDING), true);
+ await root.close();
+
+ // The shards are stale: the interrupted build never wrote them.
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+
+ // Next build: no mtime moved AND the hashes now match, so nothing but the
+ // pending flag can save these pages.
+ await build();
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+
+ const after = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const meta2 = after.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ assert.equal(meta2.get(META_PAGES_PENDING), false, "the debt is settled");
+ await after.close();
+});
diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts
@@ -0,0 +1,462 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Cue } from "../lib/vtt";
+import type { TranscriptSummary } from "../lib/transcripts";
+import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags";
+import {
+ applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
+ createTagCounter,
+ curatedTagsRuntime,
+ deriveCuratedTags,
+ hashCuratedAssignments,
+ hashCuratedRules,
+ publishedTagsFrom,
+ reapplyCuratedTags,
+ META_ASSIGN_HASH,
+ META_ASSIGN_SIGS,
+ META_PAGES_PENDING,
+ META_RULES_HASH,
+} from "./curatedTagsIndex";
+
+type IndexKey = [string, string, string];
+
+// Minimal stand-ins for the LMDB sub-DBs buildIndex hands the re-apply pass.
+// Keys are the same tuples; range scans only need to be correct, not fast.
+function makeDb<V>() {
+ const map = new Map<string, { key: unknown; value: V }>();
+ const k = (key: unknown) => JSON.stringify(key);
+ return {
+ map,
+ get(key: unknown): V | undefined {
+ return map.get(k(key))?.value;
+ },
+ put(key: unknown, value: V): void {
+ map.set(k(key), { key, value });
+ },
+ getRange(options?: { start?: unknown[] }): Iterable<{ key: unknown; value: V }> {
+ const start = options?.start as string[] | undefined;
+ const all = Array.from(map.values()).sort((a, b) =>
+ k(a.key) < k(b.key) ? -1 : 1,
+ );
+ if (!start) return all;
+ return all.filter((e) => (e.key as string[])[0] === start[0]);
+ },
+ };
+}
+
+function makeMeta() {
+ const map = new Map<string, unknown>();
+ return {
+ map,
+ get: (key: string) => map.get(key),
+ put: (key: string, value: unknown) => void map.set(key, value),
+ };
+}
+
+function summary(
+ channelSlug: string,
+ id: string,
+ uploadDate: string,
+ over: Partial<TranscriptSummary> = {},
+): TranscriptSummary {
+ return {
+ slug: `${channelSlug}/${id}`,
+ id,
+ channelSlug,
+ title: `Video ${id}`,
+ uploadDate,
+ duration: 100,
+ channel: channelSlug,
+ description: "",
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: `https://example.com/${id}`,
+ ...over,
+ };
+}
+
+const COLLAB_RULE: CuratedTagDef = {
+ id: "eva-collab",
+ label: "Collab",
+ order: 1,
+ rules: [{ id: "meta", kind: "metadata", pattern: "elfpire", enabled: true }],
+};
+
+function fixture(config: CuratedTagsConfig) {
+ const sums = makeDb<TranscriptSummary>();
+ const cues = makeDb<Cue[]>();
+ const subs = makeDb<{ track: string; cues: Cue[] }[]>();
+ const byChannel = makeDb<number>();
+ const meta = makeMeta();
+ const put = (s: TranscriptSummary) => {
+ const ik: IndexKey = [s.uploadDate, s.channelSlug, s.id];
+ sums.put(ik, s);
+ byChannel.put([s.channelSlug, s.uploadDate, s.id], 1);
+ return ik;
+ };
+ return { sums, cues, subs, byChannel, meta, put, runtime: curatedTagsRuntime(config) };
+}
+
+function cfg(
+ tags: CuratedTagDef[],
+ assignments: CuratedTagsConfig["assignments"] = {},
+): CuratedTagsConfig {
+ return { version: 1, tags, assignments };
+}
+
+// ─── hashes ───
+
+test("hashCuratedRules ignores presentation, reacts to derivation", () => {
+ const base = hashCuratedRules([COLLAB_RULE]);
+ assert.equal(
+ hashCuratedRules([{ ...COLLAB_RULE, label: "On mic", color: "#fff", hidden: true }]),
+ base,
+ "relabelling must not re-page the corpus",
+ );
+ assert.notEqual(
+ hashCuratedRules([
+ { ...COLLAB_RULE, rules: [{ id: "meta", kind: "metadata", pattern: "eva", enabled: true }] },
+ ]),
+ base,
+ );
+ assert.notEqual(hashCuratedRules([{ ...COLLAB_RULE, order: 9 }]), base);
+ assert.notEqual(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [{ ...COLLAB_RULE.rules![0], channels: ["legal-mindset"] }],
+ },
+ ]),
+ base,
+ );
+ // A disabled rule is not part of the derivation at all.
+ assert.equal(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [COLLAB_RULE.rules![0], { id: "off", kind: "caption", pattern: "x", enabled: false }],
+ },
+ ]),
+ base,
+ );
+});
+
+test("hashCuratedAssignments is order-insensitive", () => {
+ const a = hashCuratedAssignments({
+ "c/v1": { manual: ["b", "a"] },
+ "c/v2": { suppressed: ["z"] },
+ });
+ const b = hashCuratedAssignments({
+ "c/v2": { suppressed: ["z"] },
+ "c/v1": { manual: ["a", "b"] },
+ });
+ assert.equal(a, b);
+ assert.notEqual(a, hashCuratedAssignments({ "c/v1": { manual: ["a"] } }));
+});
+
+// ─── the summary field ───
+
+test("applyCuratedTagsToSummary omits the key when empty", () => {
+ const s = summary("c", "v", "20260101");
+ assert.equal(applyCuratedTagsToSummary(s, []), false);
+ assert.equal("curatedTags" in s, false);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), true);
+ assert.deepEqual(s.curatedTags, ["a"]);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), false, "idempotent");
+ assert.equal(applyCuratedTagsToSummary(s, []), true);
+ assert.equal("curatedTags" in s, false, "the key is deleted, not emptied");
+});
+
+// ─── derivation ───
+
+test("deriveCuratedTags folds rule hits with pins and suppressions", () => {
+ const runtime = curatedTagsRuntime(
+ cfg([COLLAB_RULE, { id: "manual-only", label: "Manual", order: 2 }], {
+ "lm/v2": { manual: ["manual-only"] },
+ "lm/v3": { suppressed: ["eva-collab"] },
+ }),
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v1", title: "with Elfpire" }),
+ ["eva-collab"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v2", title: "nothing" }),
+ ["manual-only"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v3", title: "with Elfpire" }),
+ [],
+ "a suppression kills a rule hit",
+ );
+});
+
+// ─── reapplyCuratedTags ───
+
+test("unchanged hashes do no work at all", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.meta.put(META_RULES_HASH, f.runtime.rulesHash);
+ f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash);
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.deepEqual(res, {
+ rulesChanged: false,
+ assignmentsChanged: false,
+ changedCount: 0,
+ changed: [],
+ examined: 0,
+ pagesPending: false,
+ });
+});
+
+// THE COLD START, and the reason curated tags need no SCHEMA_VERSION bump: an
+// index built before they existed carries no hash at all, which can never equal
+// the current one, so the first build re-derives every record by itself.
+test("an index with no stored hashes re-derives every record", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.put(summary("lm", "v2", "20260102", { title: "ordinary" }));
+ assert.equal(f.meta.get(META_RULES_HASH), undefined);
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.equal(res.rulesChanged, true);
+ assert.equal(res.examined, 2, "every record is a candidate on a cold index");
+ assert.equal(res.changedCount, 1);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v1"]);
+});
+
+test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () => {
+ // The state of every existing install the day this ships: no vocabulary, no
+ // assignments. The pass still runs (the hashes have to be recorded), but
+ // every record derives [] and comes out byte-identical, so nothing is written
+ // and the caller never flips sharedNeedsBuild. This is what makes the absent
+ // SCHEMA_VERSION bump free rather than merely cheap.
+ const f = fixture(cfg([]));
+ for (let i = 0; i < 10; i++) f.put(summary("lm", `v${i}`, `2026010${i}`));
+ const before = JSON.stringify(Array.from(f.sums.map.values()));
+ const res = reapplyCuratedTags({ ...f });
+ assert.equal(res.changedCount, 0);
+ assert.equal(res.pagesPending, false, "nothing moved, so nothing to re-page");
+ assert.equal(res.examined, 10);
+ // The keys are not collected unless asked for — a 30k-record first build must
+ // not hold 30k tuples to answer a boolean.
+ assert.equal(res.changed, undefined);
+ assert.equal(JSON.stringify(Array.from(f.sums.map.values())), before);
+ for (const { value } of f.sums.map.values()) {
+ assert.equal("curatedTags" in value, false);
+ }
+ // And the hashes are now recorded, so every later build is a true no-op.
+ assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+});
+
+test("a rule change re-derives the whole corpus from LMDB and reports the movers", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" }));
+ f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" }));
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.equal(res.rulesChanged, true);
+ assert.equal(res.examined, 3);
+ assert.equal(res.changedCount, 2);
+ assert.deepEqual(
+ res.changed?.map((k) => k[2]).sort(),
+ ["v1", "v3"],
+ );
+ assert.deepEqual(f.sums.get(["20260101", "lm", "v1"])!.curatedTags, ["eva-collab"]);
+ assert.equal(f.sums.get(["20260102", "lm", "v2"])!.curatedTags, undefined);
+ // The hashes are recorded, so a second pass is a no-op.
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+});
+
+test("removing the last rule strips the tag back off every record", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ reapplyCuratedTags({ ...f });
+ assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]);
+ const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) };
+ const res = reapplyCuratedTags(gone);
+ assert.equal(res.changedCount, 1);
+ assert.equal(f.sums.get(ik)!.curatedTags, undefined);
+});
+
+test("an assignment-only change touches ONLY the videos whose assignment moved", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ for (let i = 0; i < 20; i++) {
+ f.put(summary("lm", `v${i}`, `202601${String(i + 10).padStart(2, "0")}`));
+ }
+ f.put(summary("other", "w1", "20260201"));
+ reapplyCuratedTags({ ...f }); // establish the baseline hashes
+
+ const pinned = {
+ ...f,
+ runtime: curatedTagsRuntime(
+ cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }),
+ ),
+ };
+ const res = reapplyCuratedTags({ ...pinned, collectChanged: true });
+ assert.equal(res.rulesChanged, false);
+ assert.equal(res.assignmentsChanged, true);
+ // 21 videos in the corpus; exactly one was even looked at.
+ assert.equal(res.examined, 1);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v7"]);
+ assert.deepEqual(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, ["eva-collab"]);
+
+ // Clearing it again is also a one-video pass, and the key comes back off.
+ const cleared = { ...f, runtime: curatedTagsRuntime(cfg([COLLAB_RULE])) };
+ const res2 = reapplyCuratedTags(cleared);
+ assert.equal(res2.examined, 1);
+ assert.equal(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, undefined);
+});
+
+test("videos the worker already derived this build are not done twice", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({
+ ...f,
+ alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]),
+ collectChanged: true,
+ });
+ assert.equal(res.examined, 0);
+ assert.deepEqual(res.changed, []);
+});
+
+test("a schema bump records the hashes and re-applies nothing", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({ ...f, allFresh: true });
+ assert.equal(res.changedCount, 0);
+ assert.equal(res.examined, 0);
+ // The pages ARE rebuilt after a wipe, so the debt is recorded all the same —
+ // an interrupt mid-rebuild must not be forgotten.
+ assert.equal(res.pagesPending, true);
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.deepEqual(f.meta.get(META_ASSIGN_SIGS), {});
+});
+
+// ─── the interrupted-build flag ───
+
+test("a change sets curatedPagesPending, and only clearing it settles the debt", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const first = reapplyCuratedTags({ ...f });
+ assert.equal(first.changedCount, 1);
+ assert.equal(first.pagesPending, true);
+ assert.equal(f.meta.get(META_PAGES_PENDING), true);
+
+ // THE INTERRUPT: the hashes are stored but the page build never ran, so the
+ // flag was never cleared. The next build finds the hashes equal — it
+ // re-derives nothing — and must STILL report that the pages owe a rewrite.
+ const afterCrash = reapplyCuratedTags({ ...f });
+ assert.equal(afterCrash.rulesChanged, false);
+ assert.equal(afterCrash.changedCount, 0);
+ assert.equal(
+ afterCrash.pagesPending,
+ true,
+ "hashes match but the shards are still stale",
+ );
+
+ // The page build completed this time.
+ clearCuratedPagesPending(f.meta);
+ assert.equal(f.meta.get(META_PAGES_PENDING), false);
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+});
+
+test("no change means no debt — an untagged corpus never sets the flag", () => {
+ const f = fixture(cfg([]));
+ f.put(summary("lm", "v1", "20260101"));
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+ assert.notEqual(f.meta.get(META_PAGES_PENDING), true);
+});
+
+test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => {
+ const capRule: CuratedTagDef = {
+ id: "eva-topic",
+ label: "Discussed",
+ rules: [
+ {
+ id: "cap",
+ kind: "caption",
+ pattern: "elfpire",
+ channels: ["lm"],
+ enabled: true,
+ },
+ ],
+ };
+ const chatRule: CuratedTagDef = {
+ id: "eva-in-chat",
+ label: "In chat",
+ rules: [{ id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true }],
+ };
+ const f = fixture(cfg([capRule, chatRule]));
+ const a = f.put(summary("lm", "v1", "20260101"));
+ const b = f.put(summary("other", "v2", "20260102"));
+ f.cues.put(a, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.cues.put(b, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.subs.put(a, [
+ { track: "en", cues: [{ start: 0, end: 1, text: "ElfpireEva: not chat" }] },
+ { track: "live_chat", cues: [{ start: 0, end: 1, text: "ElfpireEva: hi" }] },
+ ]);
+
+ reapplyCuratedTags({ ...f });
+ // Order follows the vocabulary (neither def sets `order`, so it is the
+ // config's own order), not the alphabet.
+ assert.deepEqual(f.sums.get(a)!.curatedTags, ["eva-topic", "eva-in-chat"]);
+ // The caption rule is scoped to `lm`, so the other channel gets nothing even
+ // though its cues would match.
+ assert.equal(f.sums.get(b)!.curatedTags, undefined);
+});
+
+// ─── counts + publication ───
+
+test("createTagCounter accumulates per tag and per channel", () => {
+ const c = createTagCounter();
+ c.add("lm", ["eva-collab", "eva-topic"]);
+ c.add("lm", ["eva-collab"]);
+ c.add("nux", ["eva-collab"]);
+ c.add("nux", undefined);
+ c.add("nux", []);
+ const file = c.file("2026-09-21T00:00:00Z");
+ assert.equal(file.version, 1);
+ assert.deepEqual(file.tags["eva-collab"], {
+ count: 3,
+ channels: { lm: 2, nux: 1 },
+ });
+ assert.deepEqual(file.tags["eva-topic"], { count: 1, channels: { lm: 1 } });
+});
+
+test("publishedTagsFrom drops hidden and zero-count tags and sorts by order", () => {
+ const defs: CuratedTagDef[] = [
+ { id: "b-tag", label: "B", order: 2, group: "eva", groupLabel: "Eva" },
+ { id: "a-tag", label: "A", order: 1, color: "#fff" },
+ { id: "hidden-tag", label: "Hidden", order: 0, hidden: true },
+ { id: "unused", label: "Unused", order: 3 },
+ ];
+ const counts = {
+ version: 1,
+ generatedAt: "",
+ tags: {
+ "a-tag": { count: 2, channels: { lm: 2 } },
+ "b-tag": { count: 1, channels: { nux: 1 } },
+ "hidden-tag": { count: 5, channels: { lm: 5 } },
+ },
+ };
+ const published = publishedTagsFrom(defs, counts);
+ assert.deepEqual(
+ published.tags.map((t) => t.id),
+ ["a-tag", "b-tag"],
+ );
+ assert.deepEqual(published.tags[0], {
+ id: "a-tag",
+ label: "A",
+ color: "#fff",
+ order: 1,
+ count: 2,
+ channels: { lm: 2 },
+ });
+ assert.equal(published.tags[1].groupLabel, "Eva");
+ // No counts at all → nothing to publish (compose then writes no file).
+ assert.deepEqual(publishedTagsFrom(defs, null).tags, []);
+});
diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts
@@ -0,0 +1,510 @@
+// Curated tags, as the index build sees them.
+//
+// A video's `curatedTags` is derived, never stored on disk beside the video:
+// it is (rule hits ∪ operator pins) − suppressions, folded fresh at every
+// build. Rule hits in particular are NOT persisted anywhere — edit a rule,
+// rebuild, done — which is why this module exists: nothing in the per-video
+// mtime diff can notice that a rule changed, so the build has to ask.
+//
+// Two hashes in the existing `meta` sub-DB answer that question:
+// curatedRulesHash the derivation-relevant shape of the vocabulary — every
+// def's id and order, and every ENABLED rule's kind,
+// pattern, channel scope and date range
+// curatedAssignHash every pin and suppression
+// Both are checked unconditionally on every build (a skipped check would make
+// a rule edit silently inert until some unrelated mtime moved), and alongside
+// them `curatedAssignSigs` keeps a per-video signature so an assignment-only
+// change can re-derive just the videos whose assignment actually moved instead
+// of the whole corpus.
+//
+// **Only the CORPUS vocabulary is evaluated here.** A site-only rule
+// (sites/<id>/tags.json) contributes its definition to that site's published
+// /tags.json, but it does not tag records: records are shared by every site
+// that carries the channel, so a per-site rule hit would have to live in a
+// per-site record. Promote a rule to transcripts/tags.json to make it bind.
+
+import { createHash } from "node:crypto";
+import type { Cue } from "../lib/vtt";
+import type { Paths } from "../lib/paths";
+import type { TranscriptSummary } from "../lib/transcripts";
+import {
+ compileTagRules,
+ effectiveTagsFor,
+ evaluateCompiledRules,
+ ruleApplies,
+ type CompiledTagRules,
+ type CuratedTagAssignment,
+ type CuratedTagDef,
+ type CuratedTagsConfig,
+ type PublishedTags,
+ type TagRuleInput,
+} from "../lib/curatedTags";
+import { readGlobalTags } from "../lib/curatedTagsStore";
+
+export const META_RULES_HASH = "curatedRulesHash";
+export const META_ASSIGN_HASH = "curatedAssignHash";
+export const META_ASSIGN_SIGS = "curatedAssignSigs";
+// "Records were re-derived; the shared pages do not reflect them yet." Set when
+// the hashes are recorded with changes, cleared only once the shared page build
+// has finished. Without it, a Ctrl-C between the two would leave the hashes
+// stored and the shards stale for ever: the next build would compare equal,
+// re-derive nothing, and never dirty the pages.
+export const META_PAGES_PENDING = "curatedPagesPending";
+
+export const TAG_COUNTS_FILENAME = "tag-counts.json";
+export const TAG_COUNTS_VERSION = 1;
+
+type IndexKey = [string, string, string];
+type ChannelKey = [string, string, string];
+
+function sha1(s: string): string {
+ return createHash("sha1").update(s).digest("hex");
+}
+
+// The shape of the vocabulary that can change a record. Presentation fields
+// (label, colour, group, hidden) are deliberately absent: relabelling a tag
+// must not re-page 30,000 videos.
+export function hashCuratedRules(defs: CuratedTagDef[]): string {
+ const shape = defs.map((def, i) => [
+ def.id,
+ def.order ?? i,
+ (def.rules ?? [])
+ .filter((r) => r.enabled !== false)
+ .map((r) => [
+ r.id,
+ r.kind,
+ r.pattern,
+ [...(r.channels ?? [])].sort(),
+ r.dateFrom ?? null,
+ r.dateTo ?? null,
+ ]),
+ ]);
+ return sha1(JSON.stringify(shape));
+}
+
+// One video's assignment, canonically. Also the per-key signature stored in
+// meta, so the diff is a string compare.
+export function assignmentSignature(a: CuratedTagAssignment): string {
+ return `${[...(a.manual ?? [])].sort().join(",")}|${[...(a.suppressed ?? [])].sort().join(",")}`;
+}
+
+export function assignmentSignatures(
+ assignments: Record<string, CuratedTagAssignment>,
+): Record<string, string> {
+ const out: Record<string, string> = {};
+ for (const key of Object.keys(assignments).sort()) {
+ out[key] = assignmentSignature(assignments[key]);
+ }
+ return out;
+}
+
+export function hashCuratedAssignments(
+ assignments: Record<string, CuratedTagAssignment>,
+): string {
+ return sha1(JSON.stringify(assignmentSignatures(assignments)));
+}
+
+// Everything the build needs, loaded and compiled ONCE per build.
+export type CuratedTagsRuntime = {
+ config: CuratedTagsConfig;
+ defs: CuratedTagDef[];
+ compiled: CompiledTagRules;
+ rulesHash: string;
+ assignHash: string;
+ sigs: Record<string, string>;
+ // True when the corpus has no rules and no assignments at all — the state of
+ // every untagged install, and the cue to do no work whatsoever.
+ isNoop: boolean;
+};
+
+export function loadCuratedTagsRuntime(paths: Paths): CuratedTagsRuntime {
+ return curatedTagsRuntime(readGlobalTags(paths));
+}
+
+export function curatedTagsRuntime(config: CuratedTagsConfig): CuratedTagsRuntime {
+ const compiled = compileTagRules(config.tags);
+ return {
+ config,
+ defs: config.tags,
+ compiled,
+ rulesHash: hashCuratedRules(config.tags),
+ assignHash: hashCuratedAssignments(config.assignments),
+ sigs: assignmentSignatures(config.assignments),
+ isNoop:
+ compiled.isEmpty && Object.keys(config.assignments).length === 0,
+ };
+}
+
+// Cheap pre-checks, so a video on a channel no rule is scoped to never pays for
+// a cue read. They ask `ruleApplies` — the SAME predicate
+// evaluateCompiledRules uses — so this can never answer "no cues needed" for a
+// rule that would then have fired.
+export function needsCaptionCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.caption.some((r) =>
+ ruleApplies(r, { channelSlug, uploadDate }),
+ );
+}
+
+export function needsChatCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.chatAuthor.some((r) =>
+ ruleApplies(r, { channelSlug, uploadDate }),
+ );
+}
+
+// The whole fold for one video: rule hits, then the operator's pins and
+// suppressions over the top.
+export function deriveCuratedTags(
+ runtime: CuratedTagsRuntime,
+ input: TagRuleInput,
+): string[] {
+ const hits = evaluateCompiledRules(input, runtime.compiled);
+ const assignment = runtime.config.assignments[`${input.channelSlug}/${input.id}`];
+ if (hits.length === 0 && !assignment) return [];
+ return effectiveTagsFor(hits, assignment, runtime.defs);
+}
+
+// Set/clear `curatedTags` on a summary in place. Returns true when the stored
+// value actually changed — omitted-when-empty is the contract, so an untagged
+// record must come out byte-identical to the one already on disk.
+export function applyCuratedTagsToSummary(
+ summary: TranscriptSummary,
+ tags: string[],
+): boolean {
+ const before = summary.curatedTags;
+ if (tags.length === 0) {
+ if (before === undefined) return false;
+ delete summary.curatedTags;
+ return true;
+ }
+ if (
+ before &&
+ before.length === tags.length &&
+ before.every((t, i) => t === tags[i])
+ ) {
+ return false;
+ }
+ summary.curatedTags = tags;
+ return true;
+}
+
+// ─── re-derivation ───
+
+type SumsDb = {
+ get(key: IndexKey): TranscriptSummary | undefined;
+ put(key: IndexKey, value: TranscriptSummary): unknown;
+ getRange(options?: unknown): Iterable<{ key: unknown; value: unknown }>;
+};
+type CuesDb = { get(key: IndexKey): Cue[] | undefined };
+type SubsDb = { get(key: IndexKey): { track: string; cues: Cue[] }[] | undefined };
+type ByChannelDb = {
+ getRange(options: unknown): Iterable<{ key: unknown }>;
+};
+type MetaDb = {
+ get(key: string): unknown;
+ put(key: string, value: unknown): unknown;
+};
+
+export type ReapplyOptions = {
+ runtime: CuratedTagsRuntime;
+ sums: SumsDb;
+ cues: CuesDb;
+ subs: SubsDb;
+ byChannel: ByChannelDb;
+ meta: MetaDb;
+ // indexKeyIds the per-video worker already derived this build. Skipped here
+ // so a rule edit landing in the same build as new videos doesn't do them
+ // twice.
+ alreadyFresh?: Set<string>;
+ // A schema bump cleared the cache and the worker re-derived everything, so
+ // there is nothing to re-apply — just record the hashes.
+ allFresh?: boolean;
+ // Return the changed index keys as well as the count. Off by default: the
+ // build only needs the boolean, and a first build over 30k records would
+ // otherwise hold 30k three-element tuples for no reason. Tests and any future
+ // per-channel dirtying ask for them.
+ collectChanged?: boolean;
+ log?: (msg: string) => void;
+};
+
+export type ReapplyResult = {
+ rulesChanged: boolean;
+ assignmentsChanged: boolean;
+ // How many videos' stored curatedTags moved. Non-zero means the shared pages
+ // must be rewritten.
+ changedCount: number;
+ // Which ones — only when `collectChanged` was asked for.
+ changed?: IndexKey[];
+ // Videos examined (the cost).
+ examined: number;
+ // True when the shared pages still owe a rewrite for curated tags: either
+ // this call changed records, or an earlier build did and was interrupted
+ // before its pages were written. The caller MUST OR this into whatever
+ // decides to rebuild the shared trees, and clear it (clearCuratedPagesPending)
+ // only once that build has finished.
+ pagesPending: boolean;
+};
+
+function indexKeyId(k: IndexKey): string {
+ return `${k[0]}\x00${k[1]}\x00${k[2]}`;
+}
+
+function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[] | undefined {
+ if (!stored) return undefined;
+ for (const t of stored) if (t.track === "live_chat") return t.cues;
+ return undefined;
+}
+
+// Re-derive `curatedTags` for the videos a tag change can reach, WITHOUT
+// re-reading a single video directory: cues and live chat come back out of the
+// LMDB sub-DBs the build already populated.
+//
+// rules changed every video is a candidate (a new pattern can match
+// anything), but cues are read only for the videos a
+// caption/chat rule is actually scoped to
+// assignments changed only the videos whose assignment signature moved —
+// found by a key-only range scan of `byChannel` per
+// affected channel, which is the cheap idiom
+//
+// Returns how many videos changed (and which, on request) so the caller can
+// force those pages to be rewritten: nothing in the mtime diff knows this
+// happened. It also sets `curatedPagesPending` — see the flag's own note.
+export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
+ const { runtime, sums, cues, subs, byChannel, meta } = opts;
+ const log = opts.log ?? (() => {});
+ const prevRules = meta.get(META_RULES_HASH) as string | undefined;
+ const prevAssign = meta.get(META_ASSIGN_HASH) as string | undefined;
+ const prevSigs = (meta.get(META_ASSIGN_SIGS) as Record<string, string> | undefined) ?? {};
+ const rulesChanged = prevRules !== runtime.rulesHash;
+ const assignmentsChanged = prevAssign !== runtime.assignHash;
+ // A previous build re-derived records and was then interrupted before it
+ // finished writing the shared pages. The hashes are already stored, so this
+ // build would otherwise see "nothing changed" and ship stale shards forever.
+ const pendingBefore = meta.get(META_PAGES_PENDING) === true;
+
+ const record = (pages: boolean) => {
+ meta.put(META_RULES_HASH, runtime.rulesHash);
+ meta.put(META_ASSIGN_HASH, runtime.assignHash);
+ meta.put(META_ASSIGN_SIGS, runtime.sigs);
+ // Set BEFORE the pages are written and cleared only after they are; never
+ // un-set here, or an interrupted build's debt would be forgotten.
+ if (pages) meta.put(META_PAGES_PENDING, true);
+ };
+ const result = (over: Partial<ReapplyResult> & { changedCount: number }): ReapplyResult => ({
+ rulesChanged,
+ assignmentsChanged,
+ examined: 0,
+ pagesPending: pendingBefore || over.changedCount > 0,
+ ...over,
+ });
+
+ if (!rulesChanged && !assignmentsChanged) {
+ return result({ changedCount: 0, ...(opts.collectChanged ? { changed: [] } : {}) });
+ }
+ if (opts.allFresh) {
+ // A schema bump: the worker re-derived every record, and the pages are
+ // being rebuilt anyway — but the debt is recorded all the same, so an
+ // interrupt mid-rebuild is not forgotten either.
+ record(true);
+ log(
+ `curated tags: rules ${runtime.rulesHash.slice(0, 8)}, assignments ${runtime.assignHash.slice(0, 8)} (full rebuild, nothing to re-apply)`,
+ );
+ return {
+ rulesChanged,
+ assignmentsChanged,
+ changedCount: 0,
+ ...(opts.collectChanged ? { changed: [] } : {}),
+ examined: 0,
+ pagesPending: true,
+ };
+ }
+
+ const changed: IndexKey[] = [];
+ let changedCount = 0;
+ let examined = 0;
+ const skip = opts.alreadyFresh;
+
+ // `known` is the value the cursor already decoded, when there is one — a
+ // second sums.get() per video would double the msgpack decodes on the
+ // whole-corpus path.
+ const rederive = (indexKey: IndexKey, known?: TranscriptSummary): void => {
+ if (skip?.has(indexKeyId(indexKey))) return;
+ const summary = known ?? sums.get(indexKey);
+ if (!summary) return;
+ examined++;
+ const channelSlug = indexKey[1];
+ const tags = deriveCuratedTags(runtime, {
+ channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ // Only read what a scoped rule can actually use. On an untagged corpus,
+ // and on every channel no rule names, this reads nothing.
+ captionCues: needsCaptionCues(runtime, channelSlug, summary.uploadDate)
+ ? cues.get(indexKey)
+ : undefined,
+ chatCues: needsChatCues(runtime, channelSlug, summary.uploadDate)
+ ? chatCuesOf(subs.get(indexKey))
+ : undefined,
+ });
+ if (applyCuratedTagsToSummary(summary, tags)) {
+ sums.put(indexKey, summary);
+ changedCount++;
+ if (opts.collectChanged) changed.push(indexKey);
+ }
+ };
+
+ if (rulesChanged) {
+ // Every video is a candidate. This walks `sums` keys only; the value comes
+ // from the same cursor, and cue reads are gated per video above.
+ for (const { key, value } of sums.getRange()) {
+ rederive(key as IndexKey, value as TranscriptSummary);
+ }
+ } else {
+ // Assignment-only change: the symmetric difference of the signatures.
+ const movedKeys = new Set<string>();
+ for (const [key, sig] of Object.entries(runtime.sigs)) {
+ if (prevSigs[key] !== sig) movedKeys.add(key);
+ }
+ for (const key of Object.keys(prevSigs)) {
+ if (!(key in runtime.sigs)) movedKeys.add(key);
+ }
+ // Group by channel so each channel costs one key-only range scan.
+ const byChannelSlug = new Map<string, Set<string>>();
+ for (const key of movedKeys) {
+ const at = key.indexOf("/");
+ if (at <= 0) continue;
+ const slug = key.slice(0, at);
+ const id = key.slice(at + 1);
+ const set = byChannelSlug.get(slug) ?? new Set<string>();
+ set.add(id);
+ byChannelSlug.set(slug, set);
+ }
+ for (const [slug, ids] of byChannelSlug) {
+ for (const { key } of byChannel.getRange({
+ start: [slug],
+ end: [slug, ""],
+ values: false,
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== slug) continue;
+ if (!ids.has(ck[2])) continue;
+ rederive([ck[1], ck[0], ck[2]]);
+ }
+ }
+ }
+
+ record(changedCount > 0);
+ const what = [
+ rulesChanged ? `rules ${runtime.rulesHash.slice(0, 8)} (changed)` : null,
+ assignmentsChanged
+ ? `assignments ${runtime.assignHash.slice(0, 8)} (changed)`
+ : null,
+ ]
+ .filter(Boolean)
+ .join(", ");
+ log(
+ `curated tags: ${what}, examined ${examined}, re-derived ${changedCount}`,
+ );
+ return result({
+ changedCount,
+ ...(opts.collectChanged ? { changed } : {}),
+ examined,
+ });
+}
+
+// Clear the "the shared pages do not yet reflect the current tags" debt. Call
+// ONLY after the shared transcript/subs page build has completed: until then an
+// interrupt must leave the flag standing, which is the whole point of it.
+export function clearCuratedPagesPending(meta: MetaDb): void {
+ meta.put(META_PAGES_PENDING, false);
+}
+
+// ─── per-site counts ───
+
+export type TagCountsFile = {
+ version: number;
+ generatedAt: string;
+ // tag id -> total on this site + the per-channel breakdown.
+ tags: Record<string, { count: number; channels: Record<string, number> }>;
+};
+
+export function emptyTagCounts(): TagCountsFile {
+ return { version: TAG_COUNTS_VERSION, generatedAt: "", tags: {} };
+}
+
+// Accumulator for the per-site summaries stream: one add() per record, no
+// second pass over the corpus.
+export function createTagCounter() {
+ const tags = new Map<string, { count: number; channels: Map<string, number> }>();
+ return {
+ add(channelSlug: string, curatedTags: string[] | undefined): void {
+ if (!curatedTags || curatedTags.length === 0) return;
+ for (const id of curatedTags) {
+ let entry = tags.get(id);
+ if (!entry) {
+ entry = { count: 0, channels: new Map() };
+ tags.set(id, entry);
+ }
+ entry.count++;
+ entry.channels.set(channelSlug, (entry.channels.get(channelSlug) ?? 0) + 1);
+ }
+ },
+ get size(): number {
+ return tags.size;
+ },
+ file(generatedAt: string): TagCountsFile {
+ const out: TagCountsFile["tags"] = {};
+ for (const id of Array.from(tags.keys()).sort()) {
+ const entry = tags.get(id)!;
+ const channels: Record<string, number> = {};
+ for (const slug of Array.from(entry.channels.keys()).sort()) {
+ channels[slug] = entry.channels.get(slug)!;
+ }
+ out[id] = { count: entry.count, channels };
+ }
+ return { version: TAG_COUNTS_VERSION, generatedAt, tags: out };
+ },
+ };
+}
+
+// The published /tags.json for one site: its effective vocabulary joined to its
+// counts. Hidden tags and tags with no video ON THIS SITE are dropped — a
+// global tag that matched nothing here does not get a chip here — and a site
+// with nothing left ships no file at all (compose skips the write; a 404 is a
+// legitimate empty state).
+export function publishedTagsFrom(
+ defs: CuratedTagDef[],
+ counts: TagCountsFile | null,
+): PublishedTags {
+ const tags = defs
+ .filter((def) => def.hidden !== true)
+ .map((def, i) => ({ def, i, entry: counts?.tags[def.id] }))
+ .filter((x) => (x.entry?.count ?? 0) > 0)
+ .sort((a, b) => {
+ const oa = a.def.order ?? a.i;
+ const ob = b.def.order ?? b.i;
+ if (oa !== ob) return oa - ob;
+ return a.def.id < b.def.id ? -1 : a.def.id > b.def.id ? 1 : 0;
+ })
+ .map(({ def, entry }) => ({
+ id: def.id,
+ label: def.label,
+ ...(def.group ? { group: def.group } : {}),
+ ...(def.groupLabel ? { groupLabel: def.groupLabel } : {}),
+ ...(def.color ? { color: def.color } : {}),
+ ...(def.order !== undefined ? { order: def.order } : {}),
+ count: entry!.count,
+ channels: entry!.channels,
+ }));
+ return { version: 1, tags };
+}
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -77,12 +77,15 @@ test("per-channel trees take a slug, flat trees do not", () => {
assert.throws(() => pageUrl("posts", undefined, 1), /needs a channel slug/);
});
-test("root files: the four JSON documents a reader fetches by name", () => {
+test("root files: the five JSON documents a reader fetches by name", () => {
assert.deepEqual([...ROOT_FILES], [
"corpus.json",
"site.json",
"search-aliases.json",
"duplicates.json",
+ // Curated tags. Legitimately absent (like duplicates.json) — but unlike it,
+ // DECLARED in corpus.json when present, which is what corpusSpec 4 says.
+ "tags.json",
]);
assert.equal(corpusUrl(), "/corpus.json");
assert.equal(corpusUrl("https://x.example"), "https://x.example/corpus.json");
@@ -161,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => {
test("CONTRACT is frozen where it is published", () => {
// These are on the wire. A change here is a change to every deployed archive's
// machine contract, so it belongs in a slice that says so, not in a refactor.
- assert.equal(CONTRACT.corpusSpec, 3);
+ assert.equal(CONTRACT.corpusSpec, 4);
assert.equal(CONTRACT.siteDescriptor, 1);
assert.equal(CONTRACT.manifest, 3);
assert.equal(CONTRACT.transcriptsManifest, 1);
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -24,6 +24,7 @@
// dependency graph is a DAG.
import { DUPLICATES_FILENAME } from "../duplicates";
+import { TAGS_FILENAME } from "../curatedTags";
// The machine-readable versions and constants of the published contract. Every
// value here appears on the wire, so a change is a wire change — see the
@@ -31,7 +32,7 @@ import { DUPLICATES_FILENAME } from "../duplicates";
export const CONTRACT = {
// /corpus.json's own `spec`. `generator` is deliberately unversioned; see the
// note in lib/corpus.ts for why a credit line did not bump the spec.
- corpusSpec: 3,
+ corpusSpec: 4,
// /site.json's `contract` (siteDescriptor.ts).
siteDescriptor: 1,
// The four manifest versions (manifest.ts). Each is the version field of one
@@ -118,11 +119,20 @@ export function archiveUrl(base: string | undefined, p: string): string {
// duplicates.json is the one that is legitimately absent: compose-site only
// writes it when there is at least one publishable cluster, and corpus.json
// does not declare it, so a 404 here is data, not an error.
+//
+// tags.json is absent the same way — the curated vocabulary, with per-site
+// counts, written only when this site has at least one visible tag with a
+// non-zero count. Unlike duplicates.json it IS declared in corpus.json (as
+// `tags`) when present, which is what the spec-4 bump announces: a reader that
+// does not know about it misses a document it could have fetched. An archive
+// built before spec 4 simply 404s here, and a tag filter over it must say so
+// rather than silently return nothing.
export const ROOT_FILES = [
"corpus.json",
"site.json",
"search-aliases.json",
DUPLICATES_FILENAME,
+ TAGS_FILENAME,
] as const;
export type RootFile = (typeof ROOT_FILES)[number];
diff --git a/common/lib/archive/headers.test.ts b/common/lib/archive/headers.test.ts
@@ -18,6 +18,8 @@ const SITE_HEADERS = `# Generated by compose-site.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/duplicates.json
Access-Control-Allow-Origin: *
+/tags.json
+ Access-Control-Allow-Origin: *
/summaries/*
Access-Control-Allow-Origin: *
/subs/*
diff --git a/common/lib/archive/headers.ts b/common/lib/archive/headers.ts
@@ -37,6 +37,7 @@ export const SITE_CORS_PATHS: readonly string[] = [
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
"/summaries/*",
"/subs/*",
"/transcripts/*",
diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts
@@ -95,6 +95,7 @@ test("the site list is the FLAT trees plus the root files, and nothing per-chann
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
const flat = ARCHIVE_TREES.filter(isFlatTree);
@@ -146,6 +147,7 @@ test("a tree the site does not ship costs one probe and contributes nothing", as
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
});
@@ -174,5 +176,6 @@ test("a federated origin prefixes both lists and nothing else changes", async ()
`${O}/site.json`,
`${O}/search-aliases.json`,
`${O}/duplicates.json`,
+ `${O}/tags.json`,
]);
});
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -159,16 +159,41 @@ test("both builders stamp the project generator, and llms.txt trails it", () =>
}
});
-test("adding `generator` did NOT bump the corpus spec", () => {
- // Guard on the reasoning, not just the number: prior bumps announced a new
- // fetchable layer. An informational credit string breaks no reader, so a
- // client pinned to spec 3 must keep validating. If this assertion is ever
- // updated, the version-history comment in corpus.ts must justify why.
- assert.equal(CORPUS_SPEC_VERSION, 3);
- assert.equal(
- buildSiteCorpus(descriptor(), { hasArchives: false }).spec,
- 3,
+test("the spec is 4, and `generator` is still not why", () => {
+ // Guard on the reasoning, not just the number: a bump announces a new
+ // FETCHABLE document. spec 4 is /tags.json. The informational credit string
+ // added at spec 3's time breaks no reader and did not bump anything — if
+ // either assertion is ever updated, the version-history comment in corpus.ts
+ // must justify why.
+ assert.equal(CORPUS_SPEC_VERSION, 4);
+ assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4);
+});
+
+test("the tags pointer is present only when the site published one", () => {
+ // No /tags.json → corpus.json is shaped exactly as before the bump.
+ const none = buildSiteCorpus(descriptor(), { hasArchives: false });
+ assert.equal(none.tags, undefined);
+ assert.doesNotMatch(renderSiteLlmsTxt(none), /tags\.json/);
+
+ const tagged = buildSiteCorpus(
+ descriptor({ siteUrl: "https://demo.example" }),
+ { hasArchives: false, hasTags: true },
);
+ assert.equal(tagged.tags?.url, "https://demo.example/tags.json");
+ // The field name is the whole point: NOT `tags`, which is yt-dlp keywords.
+ assert.equal(tagged.tags?.videoField, "curatedTags");
+ assert.match(tagged.tags?.description ?? "", /curatedTags/);
+ const txt = renderSiteLlmsTxt(tagged);
+ assert.match(txt, /https:\/\/demo\.example\/tags\.json/);
+ assert.match(txt, /curatedTags/);
+});
+
+test("a root-relative build still names tags.json correctly", () => {
+ const corpus = buildSiteCorpus(descriptor(), {
+ hasArchives: false,
+ hasTags: true,
+ });
+ assert.equal(corpus.tags?.url, "/tags.json");
});
test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => {
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -1,5 +1,6 @@
import type { PublicSiteDescriptor } from "./siteDescriptor";
import { PROJECT_GENERATOR } from "./project";
+import { TAGS_FILENAME } from "./curatedTags";
import {
CONTRACT,
archiveUrl,
@@ -26,6 +27,12 @@ import {
// (postScheme + per-channel posts manifest pointers).
// v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under
// the same shard scheme (digestScheme + per-channel digests manifest pointers).
+// v4: curated tags — an operator-authored, cross-channel vocabulary published
+// at /tags.json, with the per-record key `curatedTags` on summaries and
+// transcript shards. A new FETCHABLE DOCUMENT, which is exactly what a bump
+// announces: a reader that skips it cannot narrow a corpus to "every stream
+// where X collabs". The per-record key itself is additive and bumps no manifest
+// version — an older reader ignores an unknown field, as it always has.
//
// NOT a v4: the `generator` field added below is deliberately unversioned. Every
// prior bump announced a new FETCHABLE LAYER — a reader that ignored it would
@@ -176,6 +183,11 @@ export type SiteCorpus = {
postScheme?: typeof POST_SCHEME;
// Present when this site ships at least one digested video.
digestScheme?: typeof DIGEST_SCHEME;
+ // Present when this site publishes at least one curated tag. `url` is
+ // /tags.json (the vocabulary + this site's per-tag counts) and `videoField`
+ // names the per-record key those ids appear in — deliberately NOT `tags`,
+ // which on a transcript record is the platform's own keywords.
+ tags?: { url: string; videoField: "curatedTags"; description: string };
// Present when this build ships bulk-download archives (whole-channel zips).
bulkArchives?: { manifest: string; note: string };
// Pointer to the human page and BYO-key chat.
@@ -211,6 +223,10 @@ export function buildSiteCorpus(
// slug -> digested video count. Absent / empty means the site ships no
// derived layer and digestScheme is omitted.
digestCounts?: Record<string, number>;
+ // Whether this site published a /tags.json (compose writes one only when a
+ // visible tag has a non-zero count here). Absent/false leaves corpus.json
+ // shaped as before apart from the spec bump.
+ hasTags?: boolean;
},
): SiteCorpus {
const base = descriptor.siteUrl;
@@ -266,6 +282,19 @@ export function buildSiteCorpus(
if (Object.values(digestCounts).some((n) => n > 0)) {
corpus.digestScheme = DIGEST_SCHEME;
}
+ // The curated vocabulary, when this site ships one.
+ if (opts.hasTags) {
+ corpus.tags = {
+ url: rootFileUrl(TAGS_FILENAME, base),
+ videoField: "curatedTags",
+ description:
+ "Curated cross-channel tags, applied per video by the archive's " +
+ "operator. Fetch tags.json for the vocabulary (id, label, group and " +
+ "the number of videos carrying it on this site), then filter records " +
+ "by their `curatedTags` array — which is NOT the same field as `tags`, " +
+ "the platform's own keywords. Absent on archives built before spec 4.",
+ };
+ }
if (opts.hasArchives) {
corpus.bulkArchives = {
manifest: join(base, "/archives/manifest.json"),
@@ -353,6 +382,14 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
`see corpus.json's digestScheme. Coverage is partial and growing.`,
);
}
+ if (corpus.tags) {
+ out.push(
+ `- [tags.json](${corpus.tags.url}): curated cross-channel tags applied ` +
+ `per video by this archive's operator — the vocabulary plus how many ` +
+ `videos carry each one here. Records name them in \`curatedTags\` ` +
+ `(not \`tags\`, which is the platform's own keywords).`,
+ );
+ }
if (corpus.bulkArchives) {
out.push(
`- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` +
diff --git a/common/lib/curatedTags.ts b/common/lib/curatedTags.ts
@@ -469,7 +469,14 @@ export function compileTagRules(defs: CuratedTagDef[]): CompiledTagRules {
// Channel scope and date range are cheap string work and are applied BEFORE any
// regex runs — the whole point of scoping a rule to one channel is not paying
// for it on the other twenty-nine.
-function ruleApplies(rule: CompiledTagRule, input: TagRuleInput): boolean {
+// Exported because the index build asks the same question one step earlier — to
+// decide whether a video's cues are worth decoding AT ALL
+// (curatedTagsIndex.ts). A second copy of this predicate could disagree with
+// the evaluation it is supposed to be predicting.
+export function ruleApplies(
+ rule: CompiledTagRule,
+ input: { channelSlug: string; uploadDate?: string },
+): boolean {
if (rule.channels && !rule.channels.has(input.channelSlug)) return false;
if (rule.dateFrom || rule.dateTo) {
const date = (input.uploadDate ?? "").replace(/\D/g, "");
diff --git a/common/lib/curatedTagsStore.test.ts b/common/lib/curatedTagsStore.test.ts
@@ -0,0 +1,456 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtempSync, readFileSync, rmSync, writeFileSync, mkdirSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "./paths";
+import {
+ applyTagAssignments,
+ assignmentFor,
+ effectiveSiteTags,
+ emptyTagsConfig,
+ readGlobalTags,
+ readSiteTags,
+ writeGlobalTags,
+ writeSiteTags,
+} from "./curatedTagsStore";
+import { siteTagsFile } from "./site";
+import type { CuratedTagsConfig } from "./curatedTags";
+
+function tempPaths(): { paths: Paths; dir: string; cleanup: () => void } {
+ const dir = mkdtempSync(path.join(tmpdir(), "curated-tags-store-"));
+ // Only the fields curatedTagsStore touches need to be real.
+ const paths = {
+ sitesDir: path.join(dir, "sites"),
+ globalTagsFile: path.join(dir, "tags.json"),
+ } as Paths;
+ return {
+ paths,
+ dir,
+ cleanup: () => rmSync(dir, { recursive: true, force: true }),
+ };
+}
+
+function withPaths(fn: (paths: Paths, dir: string) => void): void {
+ const { paths, dir, cleanup } = tempPaths();
+ try {
+ fn(paths, dir);
+ } finally {
+ cleanup();
+ }
+}
+
+const COLLAB: CuratedTagsConfig = {
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [
+ { id: "meta", kind: "metadata", pattern: "elfpire", enabled: true },
+ ],
+ },
+ ],
+ assignments: {},
+};
+
+// applyTagAssignments refuses to pin or suppress a tag the vocabulary does not
+// define, so most of these tests need one. Kept explicit per test rather than
+// seeded globally — the refusal itself is a test below.
+function seedVocab(paths: Paths, ...ids: string[]): void {
+ writeGlobalTags(paths, {
+ version: 1,
+ tags: ids.map((id) => ({ id, label: id })),
+ assignments: {},
+ });
+}
+
+const V1 = { channelSlug: "legal-mindset", id: "XZqL6k9IHGA" };
+const V2 = { channelSlug: "legal-mindset", id: "AAAAAAAAAAA" };
+
+// ─── read / write ───
+
+test("readGlobalTags of an absent file is EMPTY — no seeded defaults", () => {
+ withPaths((paths) => {
+ assert.deepEqual(readGlobalTags(paths), emptyTagsConfig());
+ assert.deepEqual(readGlobalTags(paths).tags, []);
+ });
+});
+
+test("readGlobalTags of unreadable junk is empty, not a throw", () => {
+ withPaths((paths) => {
+ writeFileSync(paths.globalTagsFile, "{not json");
+ assert.deepEqual(readGlobalTags(paths).tags, []);
+ });
+});
+
+test("global tags round-trip and are sanitized on the way out", () => {
+ withPaths((paths) => {
+ writeGlobalTags(paths, {
+ ...COLLAB,
+ tags: [
+ ...COLLAB.tags,
+ { id: "BAD ID", label: "dropped" } as never,
+ ],
+ });
+ const back = readGlobalTags(paths);
+ assert.deepEqual(
+ back.tags.map((t) => t.id),
+ ["eva-collab"],
+ );
+ // Written pretty so the file stays hand-readable (never hand-EDITED).
+ assert.match(readFileSync(paths.globalTagsFile, "utf8"), /\n {2}"tags"/);
+ });
+});
+
+test("site tags land beside site.json and default to empty", () => {
+ withPaths((paths) => {
+ assert.deepEqual(readSiteTags(paths, "anilyzer").tags, []);
+ writeSiteTags(paths, "anilyzer", {
+ version: 1,
+ tags: [{ id: "eva-collab", label: "On mic" }],
+ assignments: {},
+ });
+ assert.equal(
+ siteTagsFile(paths, "anilyzer"),
+ path.join(paths.sitesDir, "anilyzer", "tags.json"),
+ );
+ assert.equal(readSiteTags(paths, "anilyzer").tags[0].label, "On mic");
+ });
+});
+
+// ─── layering ───
+
+test("effectiveSiteTags layers the site overlay over the corpus vocabulary", () => {
+ withPaths((paths) => {
+ writeGlobalTags(paths, COLLAB);
+ writeSiteTags(paths, "anilyzer", {
+ version: 1,
+ tags: [
+ { id: "eva-collab", label: "On mic", color: "#b48ead" },
+ { id: "site-only", label: "Site only" },
+ ],
+ assignments: {},
+ });
+ const defs = effectiveSiteTags(paths, "anilyzer");
+ assert.deepEqual(
+ defs.map((d) => d.id),
+ ["eva-collab", "site-only"],
+ );
+ assert.equal(defs[0].label, "On mic");
+ assert.equal(defs[0].color, "#b48ead");
+ // Untouched global fields survive the overlay.
+ assert.equal(defs[0].group, "eva");
+ assert.deepEqual(
+ (defs[0].rules ?? []).map((r) => r.id),
+ ["meta"],
+ );
+ });
+});
+
+test("a site with no overlay ships the corpus vocabulary unchanged", () => {
+ withPaths((paths) => {
+ writeGlobalTags(paths, COLLAB);
+ assert.deepEqual(effectiveSiteTags(paths, "anilyzer"), COLLAB.tags);
+ });
+});
+
+test("the site layer cannot delete a global rule, tag or assignment", () => {
+ withPaths((paths) => {
+ writeGlobalTags(paths, COLLAB);
+ applyTagAssignments(paths, {
+ op: "add",
+ tag: "eva-collab",
+ videos: [V1],
+ source: "operator",
+ at: "2026-09-21T18:00:00Z",
+ });
+ // A site file that tries to empty the rules, and to carry an assignment.
+ mkdirSync(path.join(paths.sitesDir, "anilyzer"), { recursive: true });
+ writeFileSync(
+ siteTagsFile(paths, "anilyzer"),
+ JSON.stringify({
+ version: 1,
+ tags: [{ id: "eva-collab", label: "x", rules: [] }],
+ assignments: { "legal-mindset/XZqL6k9IHGA": { suppressed: ["eva-collab"] } },
+ }),
+ );
+ const defs = effectiveSiteTags(paths, "anilyzer");
+ assert.deepEqual(
+ (defs[0].rules ?? []).map((r) => r.id),
+ ["meta"],
+ );
+ // Assignments are global facts: the site file's copy is simply not consulted.
+ const global = readGlobalTags(paths);
+ assert.deepEqual(assignmentFor(global, V1.channelSlug, V1.id)?.manual, [
+ "eva-collab",
+ ]);
+ assert.equal(
+ assignmentFor(global, V1.channelSlug, V1.id)?.suppressed,
+ undefined,
+ );
+ });
+});
+
+// ─── applyTagAssignments: validation ───
+
+test("applyTagAssignments rejects a bad op, tag id or video ref", () => {
+ withPaths((paths) => {
+ assert.throws(
+ () => applyTagAssignments(paths, { op: "nope" as never, tag: "t", videos: [V1] }),
+ /unknown tag op/,
+ );
+ assert.throws(
+ () => applyTagAssignments(paths, { op: "add", tag: "Bad Id", videos: [V1] }),
+ /invalid tag id/,
+ );
+ assert.throws(
+ () =>
+ applyTagAssignments(paths, {
+ op: "add",
+ tag: "t",
+ videos: [{ channelSlug: "a/b", id: "v" }],
+ }),
+ /invalid video ref/,
+ );
+ assert.throws(
+ () => applyTagAssignments(paths, { op: "add", tag: "t", videos: "x" as never }),
+ /videos must be an array/,
+ );
+ // Nothing was written by any of those.
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
+test("add and suppress refuse a tag the vocabulary does not define", () => {
+ // The typo guard. Without it `eva-collabs` would ride on records for ever:
+ // no /tags.json can show a tag that has no definition, so nothing would
+ // surface the mistake.
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ for (const op of ["add", "suppress"] as const) {
+ assert.throws(
+ () => applyTagAssignments(paths, { op, tag: "eva-collabs", videos: [V1] }),
+ /unknown tag "eva-collabs" — define it first/,
+ op,
+ );
+ }
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
+test("remove and unsuppress still work for a tag whose def was deleted", () => {
+ // The other half of the guard: those two only ever take a claim away, and an
+ // operator must be able to clean up after deleting a definition.
+ withPaths((paths) => {
+ writeGlobalTags(paths, {
+ version: 1,
+ tags: [],
+ assignments: {
+ "legal-mindset/XZqL6k9IHGA": { manual: ["gone"] },
+ "legal-mindset/AAAAAAAAAAA": { suppressed: ["gone"] },
+ },
+ });
+ applyTagAssignments(paths, { op: "remove", tag: "gone", videos: [V1] });
+ applyTagAssignments(paths, { op: "unsuppress", tag: "gone", videos: [V2] });
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
+test("applyTagAssignments lowercases the tag id it stores", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ applyTagAssignments(paths, { op: "add", tag: " EVA-Collab ", videos: [V1] });
+ assert.deepEqual(readGlobalTags(paths).assignments["legal-mindset/XZqL6k9IHGA"].manual, [
+ "eva-collab",
+ ]);
+ });
+});
+
+// ─── applyTagAssignments: the four ops ───
+
+test("add pins with provenance and is idempotent", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ const first = applyTagAssignments(paths, {
+ op: "add",
+ tag: "eva-collab",
+ videos: [V1, V2, V1], // duplicate ref collapses
+ source: "umtool:elfpire-eva",
+ at: "2026-09-21T18:04:11Z",
+ });
+ assert.deepEqual(first.changed, [
+ "legal-mindset/XZqL6k9IHGA",
+ "legal-mindset/AAAAAAAAAAA",
+ ]);
+ assert.equal(first.touched.length, 2);
+ const cfg = readGlobalTags(paths);
+ const a = assignmentFor(cfg, V1.channelSlug, V1.id)!;
+ assert.deepEqual(a.manual, ["eva-collab"]);
+ assert.deepEqual(a.sources?.["eva-collab"], {
+ source: "umtool:elfpire-eva",
+ setAt: "2026-09-21T18:04:11Z",
+ });
+ // Re-adding changes nothing and does not re-stamp provenance.
+ const again = applyTagAssignments(paths, {
+ op: "add",
+ tag: "eva-collab",
+ videos: [V1],
+ source: "operator",
+ at: "2026-09-22T00:00:00Z",
+ });
+ assert.deepEqual(again.changed, []);
+ assert.equal(
+ readGlobalTags(paths).assignments["legal-mindset/XZqL6k9IHGA"].sources?.[
+ "eva-collab"
+ ].source,
+ "umtool:elfpire-eva",
+ );
+ });
+});
+
+test("add clears a suppression of the same tag", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
+ const a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
+ assert.deepEqual(a.manual, ["eva-collab"]);
+ assert.equal(a.suppressed, undefined);
+ });
+});
+
+test("remove unpins only — a rule hit survives it, which is why suppress exists", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
+ applyTagAssignments(paths, { op: "remove", tag: "eva-collab", videos: [V1] });
+ const cfg = readGlobalTags(paths);
+ // No pin, and crucially NO suppression: the rule may still tag this video.
+ assert.equal(assignmentFor(cfg, V1.channelSlug, V1.id), undefined);
+ assert.deepEqual(cfg.assignments, {});
+ });
+});
+
+test("suppress rejects a rule-derived tag and records provenance", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-topic");
+ const res = applyTagAssignments(paths, {
+ op: "suppress",
+ tag: "eva-topic",
+ videos: [V1],
+ source: "operator",
+ at: "2026-09-21T18:05:02Z",
+ });
+ assert.deepEqual(res.changed, ["legal-mindset/XZqL6k9IHGA"]);
+ const a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
+ assert.deepEqual(a.suppressed, ["eva-topic"]);
+ assert.equal(a.manual, undefined);
+ assert.deepEqual(a.sources?.["eva-topic"], {
+ source: "operator",
+ setAt: "2026-09-21T18:05:02Z",
+ });
+ });
+});
+
+test("suppress also unpins, and unsuppress clears the rejection", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
+ applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
+ let a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
+ assert.equal(a.manual, undefined);
+ assert.deepEqual(a.suppressed, ["eva-collab"]);
+ applyTagAssignments(paths, { op: "unsuppress", tag: "eva-collab", videos: [V1] });
+ // Both claims gone: the key itself is dropped rather than left as a husk.
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
+test("an op that changes nothing writes no file at all", () => {
+ withPaths((paths) => {
+ const res = applyTagAssignments(paths, {
+ op: "remove",
+ tag: "eva-collab",
+ videos: [V1],
+ });
+ assert.deepEqual(res.changed, []);
+ assert.deepEqual(res.touched, ["legal-mindset/XZqL6k9IHGA"]);
+ assert.throws(() => readFileSync(paths.globalTagsFile, "utf8"));
+ });
+});
+
+test("other tags on the same video, and other videos, are untouched", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab", "eva-topic");
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1, V2] });
+ applyTagAssignments(paths, { op: "add", tag: "eva-topic", videos: [V1] });
+ applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
+ const cfg = readGlobalTags(paths);
+ const a = assignmentFor(cfg, V1.channelSlug, V1.id)!;
+ assert.deepEqual(a.manual, ["eva-topic"]);
+ assert.deepEqual(a.suppressed, ["eva-collab"]);
+ assert.deepEqual(Object.keys(a.sources ?? {}).sort(), ["eva-collab", "eva-topic"]);
+ assert.deepEqual(assignmentFor(cfg, V2.channelSlug, V2.id)?.manual, ["eva-collab"]);
+ });
+});
+
+test("the tag vocabulary survives an assignment write", () => {
+ withPaths((paths) => {
+ writeGlobalTags(paths, COLLAB);
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
+ const cfg = readGlobalTags(paths);
+ assert.deepEqual(cfg.tags, COLLAB.tags);
+ });
+});
+
+test("a bulk call is ONE write, and reports exactly the keys it changed", () => {
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ const many = Array.from({ length: 50 }, (_, i) => ({
+ channelSlug: "legal-mindset",
+ id: `v${i}`,
+ }));
+ applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: many });
+ assert.equal(Object.keys(readGlobalTags(paths).assignments).length, 50);
+ // Half already pinned: only the new half is reported as changed.
+ const mixed = applyTagAssignments(paths, {
+ op: "add",
+ tag: "eva-collab",
+ videos: [...many.slice(0, 25), { channelSlug: "other", id: "z1" }],
+ });
+ assert.deepEqual(mixed.changed, ["other/z1"]);
+ assert.equal(mixed.touched.length, 26);
+ });
+});
+
+test("provenance for a tag that is neither pinned nor suppressed is pruned", () => {
+ withPaths((paths) => {
+ // A hand-authored file with a stale source entry for an unassigned tag.
+ writeFileSync(
+ paths.globalTagsFile,
+ JSON.stringify({
+ version: 1,
+ tags: [
+ { id: "eva-collab", label: "Collab" },
+ { id: "eva-topic", label: "Discussed" },
+ ],
+ assignments: {
+ "legal-mindset/XZqL6k9IHGA": {
+ manual: ["eva-collab"],
+ sources: {
+ "eva-collab": { source: "operator", setAt: "2026-01-01T00:00:00Z" },
+ },
+ },
+ },
+ }),
+ );
+ applyTagAssignments(paths, { op: "add", tag: "eva-topic", videos: [V1] });
+ const a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
+ assert.deepEqual(a.manual, ["eva-collab", "eva-topic"]);
+ assert.deepEqual(Object.keys(a.sources ?? {}).sort(), ["eva-collab", "eva-topic"]);
+ });
+});
diff --git a/common/lib/curatedTagsStore.ts b/common/lib/curatedTagsStore.ts
@@ -0,0 +1,305 @@
+// Server-side persistence for curated per-video tags. Two source files:
+// - global: paths.globalTagsFile (<transcriptsDir>/tags.json)
+// - per-site: sites/<siteId>/tags.json (siteTagsFile)
+// The global file is AUTHORITATIVE for both the vocabulary and the assignments
+// (an assignment is a fact about a video, not a presentation choice). The
+// per-site file is PRESENTATION — relabel/recolour/reorder/hide — plus
+// site-only tags. See common/lib/curatedTags.ts for the pure model, the
+// coercion and mergeTagDefs.
+//
+// **A site-layer RULE never tags a record.** mergeTagDefs will append one (it
+// is harmless on a published /tags.json def), but rules are evaluated once at
+// index time over records SHARED by every site that carries the channel —
+// there is no per-site record for a per-site hit to live in. Rules bind only
+// from transcripts/tags.json; promote one there to make it fire. This is why
+// the editor offers no site-side rule editor.
+//
+// Unlike aliasesStore there are NO seeded defaults: a fresh install has no
+// tags, and an absent global file reads as an empty config.
+//
+// ─── The four ops of applyTagAssignments ───
+//
+// A tag lands on a video two ways: a RULE matched (re-evaluated at every index
+// build, never persisted) or the operator PINNED it. So "take this tag off this
+// video" has two distinct meanings and they are different ops:
+//
+// add pin the tag, and clear any suppression of it
+// remove unpin only — a rule that matches this video still tags it
+// suppress reject the tag: unpin it AND record a suppression, which is the
+// ONLY way a rule-derived tag can be made to go away, because rule
+// hits are not stored anywhere to delete
+// unsuppress clear the rejection (the rule, if any, tags the video again)
+//
+// The editor's per-video toggle uses add/suppress; the bulk list actions use
+// add/remove/suppress; umtool uses add. Every pin AND every suppression records
+// provenance (`operator`, `agent:<label>`, `umtool:<project>`), so a later
+// reader can always say who claimed what and when.
+//
+// One call = ONE atomic write of the whole file, however many videos it
+// touches. A per-video loop would be N read-modify-writes and could interleave
+// with another writer.
+
+import path from "node:path";
+import { readFileSync, writeFileSync, renameSync, mkdirSync } from "node:fs";
+import type { Paths } from "./paths";
+import { siteTagsFile } from "./site";
+import {
+ assignmentKey,
+ mergeTagDefs,
+ sanitizeTagsConfig,
+ CURATED_TAGS_VERSION,
+ TAG_ID_RE,
+ type CuratedTagAssignment,
+ type CuratedTagDef,
+ type CuratedTagsConfig,
+} from "./curatedTags";
+
+function writeJsonAtomic(filePath: string, value: unknown): void {
+ mkdirSync(path.dirname(filePath), { recursive: true });
+ const tmp = `${filePath}.tmp-${process.pid}`;
+ writeFileSync(tmp, JSON.stringify(value, null, 2));
+ renameSync(tmp, filePath);
+}
+
+export function emptyTagsConfig(): CuratedTagsConfig {
+ return { version: CURATED_TAGS_VERSION, tags: [], assignments: {} };
+}
+
+// The corpus-wide vocabulary + every assignment. Absent/unreadable → empty (a
+// fresh install has no tags; there is nothing sensible to seed).
+export function readGlobalTags(paths: Paths): CuratedTagsConfig {
+ try {
+ return sanitizeTagsConfig(
+ JSON.parse(readFileSync(paths.globalTagsFile, "utf8")),
+ );
+ } catch {
+ return emptyTagsConfig();
+ }
+}
+
+export function writeGlobalTags(paths: Paths, config: CuratedTagsConfig): void {
+ writeJsonAtomic(paths.globalTagsFile, sanitizeTagsConfig(config));
+}
+
+// Per-site presentation overlay + site-only rules/tags. Absent/unreadable →
+// empty (no overlay).
+export function readSiteTags(paths: Paths, siteId: string): CuratedTagsConfig {
+ try {
+ return sanitizeTagsConfig(
+ JSON.parse(readFileSync(siteTagsFile(paths, siteId), "utf8")),
+ );
+ } catch {
+ return emptyTagsConfig();
+ }
+}
+
+export function writeSiteTags(
+ paths: Paths,
+ siteId: string,
+ config: CuratedTagsConfig,
+): void {
+ writeJsonAtomic(siteTagsFile(paths, siteId), sanitizeTagsConfig(config));
+}
+
+// The tag definitions a site actually ships: the corpus vocabulary with the
+// site's overlay applied. Read directly by compose-site.ts at build time (the
+// source files are available there, like duplicates.json), so there is no
+// separate staging step.
+//
+// Assignments are deliberately NOT part of this: they are global facts and are
+// consumed by the index build, not by compose.
+export function effectiveSiteTags(
+ paths: Paths,
+ siteId: string,
+): CuratedTagDef[] {
+ return mergeTagDefs(readGlobalTags(paths).tags, readSiteTags(paths, siteId).tags);
+}
+
+// ─── Assignments ───
+
+export type TagAssignmentOp = "add" | "remove" | "suppress" | "unsuppress";
+
+const OPS: readonly TagAssignmentOp[] = [
+ "add",
+ "remove",
+ "suppress",
+ "unsuppress",
+];
+
+export type TagVideoRef = { channelSlug: string; id: string };
+
+export type ApplyTagAssignmentsInput = {
+ op: TagAssignmentOp;
+ tag: string;
+ videos: TagVideoRef[];
+ // Provenance, recorded on every pin and every suppression this call makes.
+ // `operator` | `agent:<label>` | `umtool:<project>` | `rule:<ruleId>`.
+ source?: string;
+ // ISO-8601 instant; defaults to now. Injectable so tests are deterministic.
+ at?: string;
+};
+
+export type ApplyTagAssignmentsResult = {
+ op: TagAssignmentOp;
+ tag: string;
+ // Assignment keys whose stored state actually changed — the caller's dirty
+ // set. A no-op (pinning what is already pinned) is not reported here.
+ changed: string[];
+ // Every key the call addressed, changed or not.
+ touched: string[];
+};
+
+// Order-preserving de-dupe, Set-backed: a bulk "tag selected" can carry
+// thousands of refs, and `keys.includes` would make that quadratic.
+function dedupeRefs(videos: TagVideoRef[]): string[] {
+ const keys: string[] = [];
+ const seen = new Set<string>();
+ for (const v of videos) {
+ if (!v || typeof v.channelSlug !== "string" || typeof v.id !== "string") {
+ throw new Error("each video needs a channelSlug and an id");
+ }
+ const slug = v.channelSlug.trim();
+ const id = v.id.trim();
+ if (!slug || !id || slug.includes("/") || id.includes("/")) {
+ throw new Error(
+ `invalid video ref ${JSON.stringify(v)}: channelSlug and id must be non-empty and contain no "/"`,
+ );
+ }
+ const key = assignmentKey(slug, id);
+ if (seen.has(key)) continue;
+ seen.add(key);
+ keys.push(key);
+ }
+ return keys;
+}
+
+function withoutTag(list: string[] | undefined, tag: string): string[] {
+ return (list ?? []).filter((t) => t !== tag);
+}
+
+// Apply one op for one tag to any number of videos, in a SINGLE atomic write of
+// the global file. Validates the tag id, the op and the video refs, checks the
+// tag EXISTS in the vocabulary for the two ops that add a claim, and records
+// provenance for every pin and suppression it writes. It throws rather than
+// silently storing junk a sanitize pass would drop later: this is the one write
+// path all three writers (editor UI, ops route, umtool) funnel through, and a
+// mistyped id here would put a tag on records that no /tags.json can ever show.
+//
+// `remove` and `unsuppress` deliberately skip the vocabulary check — they only
+// ever take a claim away, and an operator must be able to clean up after a tag
+// whose definition has already been deleted.
+//
+// The read-modify-write below is atomic against other writers in THIS process
+// only because this function is synchronous end to end — there is no await
+// between readGlobalTags and writeGlobalTags for another handler to interleave
+// in. Introducing one would silently make concurrent bulk tagging lossy.
+//
+// Returns which keys changed, so the caller can dirty exactly those videos.
+export function applyTagAssignments(
+ paths: Paths,
+ input: ApplyTagAssignmentsInput,
+): ApplyTagAssignmentsResult {
+ const op = input.op;
+ if (!OPS.includes(op)) {
+ throw new Error(`unknown tag op ${JSON.stringify(op)}; expected one of ${OPS.join(", ")}`);
+ }
+ const tag = typeof input.tag === "string" ? input.tag.trim().toLowerCase() : "";
+ if (!TAG_ID_RE.test(tag)) {
+ throw new Error(
+ `invalid tag id ${JSON.stringify(input.tag)}; must match ${TAG_ID_RE}`,
+ );
+ }
+ if (!Array.isArray(input.videos)) {
+ throw new Error("videos must be an array");
+ }
+ const keys = dedupeRefs(input.videos);
+ const source = (input.source ?? "operator").trim() || "operator";
+ const at = input.at ?? new Date().toISOString();
+
+ const config = readGlobalTags(paths);
+ if (
+ (op === "add" || op === "suppress") &&
+ !config.tags.some((def) => def.id === tag)
+ ) {
+ throw new Error(
+ `unknown tag "${tag}" — define it first (pnpm ops tags / the /tags page)`,
+ );
+ }
+ const changed: string[] = [];
+
+ for (const key of keys) {
+ const before = config.assignments[key];
+ const manual = withoutTag(before?.manual, tag);
+ const suppressed = withoutTag(before?.suppressed, tag);
+ const sources: Record<string, { source: string; setAt: string }> = {
+ ...(before?.sources ?? {}),
+ };
+ const wasManual = (before?.manual ?? []).includes(tag);
+ const wasSuppressed = (before?.suppressed ?? []).includes(tag);
+
+ let nowManual = wasManual;
+ let nowSuppressed = wasSuppressed;
+ switch (op) {
+ case "add":
+ nowManual = true;
+ nowSuppressed = false; // a pin clears a rejection of the same tag
+ break;
+ case "remove":
+ // Unpin only. A rule that matches this video still tags it — use
+ // `suppress` to reject a rule-derived tag.
+ nowManual = false;
+ break;
+ case "suppress":
+ nowManual = false;
+ nowSuppressed = true;
+ break;
+ case "unsuppress":
+ nowSuppressed = false;
+ break;
+ }
+
+ if (nowManual === wasManual && nowSuppressed === wasSuppressed) {
+ continue; // nothing to write for this video
+ }
+
+ if (nowManual) manual.push(tag);
+ if (nowSuppressed) suppressed.push(tag);
+ if (nowManual || nowSuppressed) {
+ // Provenance is re-stamped whenever the claim changes: the recorded
+ // source is the one that put the tag in its CURRENT state.
+ sources[tag] = { source, setAt: at };
+ } else {
+ delete sources[tag];
+ }
+ // Provenance for tags this video no longer carries is dead weight.
+ for (const id of Object.keys(sources)) {
+ if (!manual.includes(id) && !suppressed.includes(id)) delete sources[id];
+ }
+
+ const next: CuratedTagAssignment = {
+ ...(manual.length > 0 ? { manual } : {}),
+ ...(suppressed.length > 0 ? { suppressed } : {}),
+ ...(Object.keys(sources).length > 0 ? { sources } : {}),
+ };
+ if (manual.length === 0 && suppressed.length === 0) {
+ // The last claim about this video is gone; drop the key entirely rather
+ // than leaving an empty husk in the file.
+ delete config.assignments[key];
+ } else {
+ config.assignments[key] = next;
+ }
+ changed.push(key);
+ }
+
+ if (changed.length > 0) writeGlobalTags(paths, config);
+ return { op, tag, changed, touched: keys };
+}
+
+// Every assignment for one video, or undefined when it carries none.
+export function assignmentFor(
+ config: CuratedTagsConfig,
+ channelSlug: string,
+ id: string,
+): CuratedTagAssignment | undefined {
+ return config.assignments[assignmentKey(channelSlug, id)];
+}
diff --git a/common/lib/paths.ts b/common/lib/paths.ts
@@ -1,6 +1,8 @@
import fs from "node:fs";
import path from "node:path";
import os from "node:os";
+// curatedTags.ts is pure (no imports of its own), so this cannot cycle.
+import { TAGS_FILENAME } from "./curatedTags";
export type Paths = {
monorepoRoot: string;
@@ -103,6 +105,11 @@ export type Paths = {
// data dir (not monorepoRoot — the legacy chartsConfigFile location there is
// migration-only). See common/lib/aliasesStore.ts.
globalAliasesFile: string;
+ // Curated per-video tags: the corpus-wide vocabulary AND every assignment
+ // (<transcriptsDir>/tags.json). Per-site presentation overlays live at
+ // sitesDir/<siteId>/tags.json (see common/lib/site.ts). Authoritative, and
+ // real curated data — never hand-edited. See common/lib/curatedTagsStore.ts.
+ globalTagsFile: string;
ytdlpBin: string;
whisperBin: string;
whisperModel: string;
@@ -219,6 +226,9 @@ export function getPaths(): Paths {
globalAliasesFile:
process.env.SEARCH_ALIASES_FILE ??
path.join(transcriptsDir, "search-aliases.json"),
+ globalTagsFile:
+ process.env.CURATED_TAGS_FILE ??
+ path.join(transcriptsDir, TAGS_FILENAME),
ytdlpBin: process.env.YTDLP_BIN ?? "yt-dlp",
whisperBin: process.env.WHISPER_BIN ?? "whisper-cli",
whisperModel:
diff --git a/common/lib/site.ts b/common/lib/site.ts
@@ -8,6 +8,7 @@ import {
type ChannelGroup,
} from "./channelGroups";
import { getPaths, type Paths } from "./paths";
+import { TAGS_FILENAME } from "./curatedTags";
import type { SiteChannelIndex } from "./channelPriority";
import { parseAccent } from "./accent";
import {
@@ -133,6 +134,14 @@ export function siteAliasesFile(paths: Paths, siteId: string): string {
return path.join(siteDir(paths, siteId), "search-aliases.json");
}
+// Per-site curated-tag overlay: presentation fields (label/groupLabel/colour/
+// order/hidden) and site-only rules or tags, layered over the corpus
+// vocabulary (paths.globalTagsFile) at compose time. It carries no assignments
+// — those are global facts. See common/lib/curatedTagsStore.ts.
+export function siteTagsFile(paths: Paths, siteId: string): string {
+ return path.join(siteDir(paths, siteId), TAGS_FILENAME);
+}
+
// Staging output dirs (NOT served) where per-site aggregates are built before
// composition into export/public. See common/lib/paths.ts.
export function siteIndexDir(paths: Paths, siteId: string): string {
diff --git a/common/lib/transcripts-server.test.ts b/common/lib/transcripts-server.test.ts
@@ -0,0 +1,55 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { toDisplaySummary } from "./transcripts-server";
+import type { TranscriptSummary } from "./transcripts";
+
+// Run with: node_modules/.bin/tsx --test common/lib/transcripts-server.test.ts
+
+function summary(over: Partial<TranscriptSummary> = {}): TranscriptSummary {
+ return {
+ slug: "test-channel/v1",
+ id: "v1",
+ channelSlug: "test-channel",
+ title: "A stream",
+ uploadDate: "20260101",
+ duration: 3600,
+ channel: "Test Channel",
+ description: "",
+ tags: ["keyword-one", "keyword-two"],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: "https://example.com/v1",
+ ...over,
+ };
+}
+
+test("toDisplaySummary omits curatedTags entirely when there are none", () => {
+ // The byte-identical-pages contract: an untagged corpus's summaries pages
+ // must not gain a key, or every one of them is rewritten on upgrade.
+ const display = toDisplaySummary(summary());
+ assert.equal("curatedTags" in display, false);
+ assert.equal("curatedTags" in toDisplaySummary(summary({ curatedTags: [] })), false);
+ // And the platform's keywords do NOT leak in under the new name.
+ assert.equal(JSON.stringify(display).includes("keyword-one"), false);
+});
+
+test("toDisplaySummary passes curatedTags through in order", () => {
+ const display = toDisplaySummary(
+ summary({ curatedTags: ["eva-collab", "eva-topic"] }),
+ );
+ assert.deepEqual(display.curatedTags, ["eva-collab", "eva-topic"]);
+});
+
+test("toDisplaySummary still derives the state fields it always did", () => {
+ const available = toDisplaySummary(summary());
+ assert.equal(available.isDeleted, false);
+ assert.equal(available.isUnlisted, false);
+ assert.equal("state" in available, false);
+ const deleted = toDisplaySummary(summary({ curatedTags: ["x"] }), {
+ state: "deleted",
+ });
+ assert.equal(deleted.state, "deleted");
+ assert.equal(deleted.isDeleted, true);
+ assert.deepEqual(deleted.curatedTags, ["x"]);
+});
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -220,5 +220,11 @@ export function toDisplaySummary(
...(state === "available" ? {} : { state }),
platform: t.platform,
webpageUrl: t.webpageUrl,
+ // Curated tags ride through to the browse index, so a list can be narrowed
+ // to "streams where X collabs" without loading a transcript page. Omitted
+ // when empty for the same byte-identical reason as `state`.
+ ...(t.curatedTags && t.curatedTags.length > 0
+ ? { curatedTags: t.curatedTags }
+ : {}),
};
}
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -11,8 +11,8 @@
* PAGES — every published JSON document: the per-channel shard trees
* (manifest.json + page-NNNN.json), the site-wide flat trees
* (summaries, stats) and the root documents (corpus.json, site.json,
- * search-aliases.json, duplicates.json). Per-channel pages are
- * cache-first; per-channel manifests are network-first so a rebuild is
+ * search-aliases.json, duplicates.json, tags.json). Per-channel
+ * pages are cache-first; per-channel manifests are network-first so a rebuild is
* seen, with invalidation keyed off manifest.generatedAt — when a
* channel's manifest generatedAt changes, that channel's cached page
* shards are evicted (the shards have stable, non-content-hashed URLs,
@@ -40,7 +40,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js
@@ -33,7 +33,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
self.skipWaiting();
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -139,6 +139,17 @@ track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
**`report` already means three things**: `reportDebouncePreset` in settings, the
`refresh-report` job kind, and `/ask`'s `ReportPanel`. Use `feedback` / `review` instead.
+**`tags` means three unrelated things, and the record field is `curatedTags`.**
+
+| Which | Where | What it is |
+| --- | --- | --- |
+| yt-dlp keywords | `common/lib/transcripts.ts:21` — `TranscriptSummary.tags` | The platform's own keywords from `metadata.info.json`, searchable via the `"tags"` scope. The UI label for this scope is being changed to **"Keywords"** (`QueryLeafView.tsx:46,156-159`); the code token, the URL and the MCP `scopes` enum stay `"tags"`. |
+| AI digest topic tags | `common/lib/digests.ts` — `VideoDigest.tags` | Model-generated topics on a digested video. |
+| **Curated tags** | `common/lib/curatedTags.ts`; record field **`curatedTags`** | The operator's cross-channel vocabulary ("every stream where X collabs"), published at `/tags.json`. |
+
+Never name the curated field `tags`. Never assume a `tags.json` is the keyword list:
+`transcripts/tags.json` and `sites/<id>/tags.json` are curated tags.
+
---
## Reusable helpers (do not rewrite these)
@@ -160,6 +171,9 @@ track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
| ENOENT → friendly message | `common/social/xGalleryDlFetcher.ts:207-213` | `/ENOENT/.test(message) ? "X not found (set X_BIN)" : message`. |
| ETA estimator | `editor/app/jobs/active/buildActiveJobs.ts:38` | `computeEtaSeconds`. |
| Binary search over cue starts | `common/components/TranscriptModal.tsx:473-491` | `findActiveIndex`. |
+| Curated-tag model (pure) | `common/lib/curatedTags.ts` | `sanitizeTagsConfig` / `mergeTagDefs` / `effectiveTagsFor` / `compileTagRules` + `evaluateCompiledRules`. Compile ONCE per build, evaluate per video. |
+| Curated-tag persistence | `common/lib/curatedTagsStore.ts:198` | `applyTagAssignments` — the ONE write path for every tag writer (editor, ops, umtool); one atomic write per call, and it refuses to pin a tag the vocabulary does not define. |
+| A chat cue's author | `common/lib/curatedTags.ts:495` — `chatAuthorOf` | The `<author>: ` prefix; there is no `author` field on `Cue`. |
---
@@ -187,9 +201,9 @@ Exposed to the settings form via `listTranscriptionApps()` (`:272`), consumed at
| Constant | Where | Value |
| --- | --- | --- |
-| `SCHEMA_VERSION` | `buildIndex.ts:116` | 12 |
+| `SCHEMA_VERSION` | `buildIndex.ts:165` | **13** — and curated tags deliberately did NOT bump it (2026-09-21); see below |
| `CUES_FILE_VERSION` | `normalizeTranscript.ts:33` | 2 |
-| `CORPUS_SPEC_VERSION` | `common/lib/corpus.ts:16` | 2 |
+| `CORPUS_SPEC_VERSION` | `common/lib/archive/contract.ts:35` (re-exported `corpus.ts:59`) | **4** (3 → 4 for `/tags.json`, 2026-09-21) |
| `MANIFEST_VERSION` (site summaries) | `common/lib/manifest.ts:36` | 3 |
| `TRANSCRIPTS_MANIFEST_VERSION` | `manifest.ts:43` | 1 |
| `SUBS_MANIFEST_VERSION` | `manifest.ts:59` | 4 |
@@ -210,6 +224,75 @@ three times (`:667` transcripts, `:675` subs, `:686` posts); it passes
Transcript shards ship the **full `cues` array inline** — `buildIndex.ts:830-832`.
+### Curated tags invalidate through `meta`, not through mtimes (2026-09-21)
+
+`curatedTags` is DERIVED at build time — `(rule hits ∪ pins) − suppressions` — and **rule
+hits are never persisted**. So nothing in the per-video mtime diff can see that a rule was
+edited or a video pinned, and the mechanism is four keys in the existing `meta` sub-DB
+(`common/controller/curatedTagsIndex.ts:44-52`), checked **unconditionally on every build**:
+
+| Key | Covers | Deliberately excludes |
+| --- | --- | --- |
+| `curatedRulesHash` | every def's id + `order`, and every **enabled** rule's kind, pattern, channel scope and date range | label / colour / group / hidden — relabelling a tag must not re-page 30,000 videos |
+| `curatedAssignHash` | every pin and suppression | — |
+| `curatedAssignSigs` | per-key `manual|suppressed` signature | not a hash: it is what makes an assignment-only edit re-derive just the videos that moved |
+| `curatedPagesPending` | "records were re-derived, the shared pages have not been written yet" | set when the hashes are recorded WITH changes, cleared only after the shared page build — without it a Ctrl-C between the two would store the hashes and leave the shards stale for ever, since the next build would compare equal and dirty nothing |
+
+`reapplyCuratedTags` (`:279`) runs after the mtime diff: rules changed → every video is a
+candidate (cue reads gated per video by channel/date scope); assignments changed → only the
+symmetric difference of the signatures, located by a key-only `byChannel` range scan per
+affected channel. It re-derives out of LMDB — **no video directory is read twice** — and
+whatever it changes flips `sharedNeedsBuild`, because the page writer's sha1 skip is what
+then leaves the untouched pages untouched — and `curatedPagesPending` is what covers the
+build that was interrupted between the two. The per-site fingerprint (`buildIndex.ts:1765`)
+includes both hashes, or a tag-only edit would leave a site reporting "up to date" with
+yesterday's counts. **No new sub-DB**, so the `clearAsync()` enumeration is unchanged.
+
+**And no `SCHEMA_VERSION` bump — on purpose.** Every prior bump existed because something
+would otherwise never be derived at all (v13 populates a new sub-DB; nothing re-derives it
+lazily). Curated tags re-derive lazily *by design*: `curatedRulesHash` is absent on an index
+built before them, so it can never equal the current hash and the first build re-derives
+every record on its own — ~80 s across 30k videos, out of LMDB, and **zero page writes on an
+untagged corpus** because an empty derivation leaves each summary byte-identical (tested:
+"an UNTAGGED corpus cold-starts to zero changes"). A bump would instead wipe the cache and
+re-read 76k video directories to reach exactly the same state.
+
+**The site layer is presentation-only.** `sites/<id>/tags.json` relabels, recolours,
+reorders, hides and may add site-only tags; it can never delete a corpus rule or an
+assignment. A site-layer RULE is accepted by `mergeTagDefs` (harmless on a published def)
+but **never tags a record**: rules are evaluated once at index time over records SHARED by
+every site carrying the channel, so there is no per-site record for a per-site hit to live
+in. Rules bind only from `transcripts/tags.json` — promote one there to make it fire. The
+editor therefore offers no site-side rule editor.
+
+**A chat cue's author is a string prefix, not a field.** `common/lib/liveChat.ts:100-105`
+emits every live-chat cue as ``text: author ? `${author}: ${text}` : text`` (`:104`) — `Cue` is
+`{start, end, text}` (`vtt.ts:1`) and has no `author`. So a `chat-author` rule matches
+`chatAuthorOf(text)` (`curatedTags.ts:495`): everything before the FIRST `": "`, and `null`
+when there is none (a cue with no author can never match). Anything that wants the author
+of a chat line must use this, not a second copy of the split.
+
+### Measured: what a curated rule costs (2026-09-21, real corpus)
+
+`legal-mindset`, 518 videos / 1.37 M caption cues / 1.47 M live-chat cues, the three seed
+rules (metadata + chat-author + caption), run offline against the production LMDB (warm):
+
+| Rules | Wall | LMDB decode | Regex |
+| --- | --- | --- | --- |
+| metadata only | 18 ms | 0 | 4 ms |
+| chat-author only | 848 ms | 727 ms | 110 ms |
+| caption only | 421 ms | 302 ms | 109 ms |
+| all three | 1.28 s | 1.03 s | 232 ms |
+
+**The cost is the cue decode, not the regex** (4:1). A metadata-only vocabulary is free; a
+cue-backed rule costs about 2.5 ms per video with cues, so a whole-corpus re-derive at
+Jeralyzer's scale (30,886 videos) projects to ~80 s — a rule edit, not a rebuild. Scoping a
+rule with `channels`/`dateFrom` skips the decode entirely for everything out of scope.
+
+Hits on that channel: 7 `eva-collab` (metadata), 96 `eva-in-chat` (chat author), 1
+`eva-topic` (caption `\belf ?pire\b|elfpyre|legal loli`) — the caption number is a pattern
+problem, not a plumbing one: a spoken name rarely appears spelled that way in ASR output.
+
---
## Digest bake-off (measured 2026-07-26)
diff --git a/plans/curated-tags.md b/plans/curated-tags.md
@@ -21,7 +21,9 @@ with her in chat.
(re-evaluated at index build; operator pins/suppresses exceptions), by **import from a umtool project**.
2. **One tag per kind**: `eva-collab` (on mic), `eva-in-chat`, `eva-topic`; filter is OR across chips.
3. **Vocabulary both corpus-wide and per site, like aliases** — corpus file authoritative for
- definitions AND assignments; the site file is presentation (relabel/hide/order/colour) + site-only rules.
+ definitions AND assignments; the site file is presentation ONLY (relabel/hide/order/colour). **Amended
+ 2026-09-21 during S1**: a site-layer rule can never tag a record — a record is shared by every site carrying
+ its channel — so rules bind only from the corpus file and the site tab offers no rule editor.
4. **Surfaces — all four**: export lists (channel + all-videos), export search, MCP filter, editor lists.
5. **Universal + flexible, three writers**: free-form tag ids for any subject, optional `group` so chips
read "Eva: Collab · In chat · Discussed"; operator UI, `pnpm ops` (agents; MCP stays read-only per
@@ -90,7 +92,8 @@ scope and date range are applied before compiling; each kind compiles once per b
video from the `cues`/`subs` sub-DBs (no disk re-read). Returns the changed `indexKey`s, which are unioned
into the dirty-channel set so transcript/subs/summaries pages rewrite — `common/lib/channelSignature.ts`
hashes only mtimes (FACTS ~:203-207), so this explicit dirtying is what keeps compose from skipping them.
- `SCHEMA_VERSION` 13 → 14 (`:143`). Build log prints `curated tags: rules <hash8> (changed), re-derived N`.
+ `SCHEMA_VERSION` stays **13** (amended during S1: an absent `curatedRulesHash` can never match, so the first
+ build re-derives every record from LMDB in ~80 s; a bump would re-read 76k video dirs for the same state). Build log prints `curated tags: rules <hash8> (changed), re-derived N`.
- **Counts**: while streaming `sums` into per-site summaries pages (`:1712`) accumulate per-tag totals and
per-channel breakdown → `tag-counts.json` in `siteIndexDir(paths, siteId)` (`common/lib/site.ts:138`).
- **Compose** (`common/bin/compose-site.ts:744-753`, beside aliases): `effectiveSiteTags(paths, siteId)` +
@@ -109,16 +112,16 @@ scope and date range are applied before compiling; each kind compiles once per b
`export/e2e/fixtures/data.ts` emits `curatedTags` on two summaries and a `/tags.json` fixture. Merge to
main immediately — S2 and S3 branch from it.
2. `common/lib/curatedTagsStore.ts` (mirror `common/lib/aliasesStore.ts`): `readGlobalTags/writeGlobalTags/
- readSiteTags/writeSiteTags/effectiveSiteTags` + **`applyTagAssignments(paths, {op: add|remove|replace, tag,
+ readSiteTags/writeSiteTags/effectiveSiteTags` + **`applyTagAssignments(paths, {op: add|remove|suppress|unsuppress, tag,
videos, source, at})`**; `paths.globalTagsFile` (`common/lib/paths.ts:105,219`), `siteTagsFile`
(`common/lib/site.ts:132`); tests: sanitize, layering, both-lists conflict, provenance.
3. `curatedTagsIndex.ts` + `buildIndex.ts` hook at `:697`, meta hashes, `reapplyCuratedTags`,
- `SCHEMA_VERSION` 14, per-site counts at `:1712`; `toDisplaySummary`. Measure rule cost on legal-mindset
+ NO schema bump (see Derivation), per-site counts at `:1712`; `toDisplaySummary`. Measure rule cost on legal-mindset
(478 chat streams) and record it in FACTS.
4. Contract + compose: `contract.ts` spec 4 + `ROOT_FILES`; `corpus.ts:204,:320`; `compose-site.ts:744-753`
writes `/tags.json`; `corpus.test.ts`.
5. `plans/FACTS.md` (the three meanings of "tags" in the naming-hazards table; chat-author prefix contract;
- the two meta hashes; SCHEMA_VERSION 14) + `AGENTS.md` (tags.json is curated data; never hand-edit).
+ the two meta hashes; SCHEMA_VERSION unchanged) + `AGENTS.md` (tags.json is curated data; never hand-edit).
### S2 — export UI + MCP (branch `tags/site`, needs only S1.1)
6. `common/lib/search/evalTree.ts`: `curatedTags?` on `SearchFilters` (`:311`) and `FilterableRecord`
@@ -146,7 +149,8 @@ scope and date range are applied before compiling; each kind compiles once per b
`previewTagRuleAction` = dry-run `evaluateTagRules` over LMDB `sums/cues/subs`, listing title/date/channel
with Pin all / Pin / Unpin, `applyTagAssignmentsAction`) — pattern `editor/app/sites/[siteId]/aliases/`.
Nav entry (twelve → thirteen; `editor/scripts/measure-nav.mjs`).
-13. Per-site overlay tab `editor/app/sites/[siteId]/tags/page.tsx` (presentation fields + site-only rules).
+13. Per-site overlay tab `editor/app/sites/[siteId]/tags/page.tsx` (presentation fields only — no rule editor,
+ see decision 3 as amended).
14. Video page `videos/[id]/components/TagsPanel.tsx` beside `OperationPanel` (`page.tsx:251`);
`toggleVideoTagAction` in `videoActions.ts` (shows each tag's provenance).
15. `VideoListPane.tsx`: `tag`/`untag` in `BULK_ACTION_OPTIONS` (`:38-49`) with a tag picker via