commit ac144c6c416b819802881d77fd3673d9a017728a
parent 0bd2cfd44cad3fc0a0c948548cf4baab263c6118
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 12:56:19 -0400
tags S1.1: curatedTags model, record fields, export fixture
The frozen contract the other two tag slices branch from.
common/lib/curatedTags.ts is pure: the types (CuratedTagRule/Def/Assignment,
CuratedTagsConfig, PublishedTag(s)), TAGS_FILENAME, TAG_ID_RE, and five
functions — sanitizeTagsConfig (tolerant like coerceAliasConfig; the one thing
it does NOT drop is a rule whose regex will not compile, which is kept disabled
with a reason so the operator sees their typo), mergeTagDefs (field-wise
presentation overlay + appended site rules; it can never delete a corpus rule),
effectiveTagsFor ((hits u manual) - suppressed, a pin beating a suppression),
compileTagRules/evaluateCompiledRules (compile once per build, evaluate per
video) and the evaluateTagRules convenience for the editor's preview.
chat-author rules read the `<author>: ` prefix of a chat cue, because that IS
the contract — common/lib/liveChat.ts formats every cue as `${author}: ${text}`
and there is no author field on Cue. Channel scope and date range are applied
before any regex runs, and a cue scan drops each rule as soon as it fires.
The record field is curatedTags, never tags: TranscriptSummary.tags is yt-dlp
keywords. It is omitted when empty so an untagged corpus's pages stay
byte-identical and the build's sha1 skip still short-circuits.
The export fixture tags two of its three summaries and gains tagsJson() plus
installTagRoutes(); installRoutes leaves /tags.json a 404, which is the real
default for a site with nothing to publish.
Co-Authored-By: Claude Opus <noreply@anthropic.com>
Diffstat:
5 files changed, 1074 insertions(+), 3 deletions(-)
diff --git a/common/lib/curatedTags.test.ts b/common/lib/curatedTags.test.ts
@@ -0,0 +1,437 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ assignmentKey,
+ chatAuthorOf,
+ compileTagRules,
+ effectiveTagsFor,
+ evaluateCompiledRules,
+ evaluateTagRules,
+ isTagId,
+ mergeTagDefs,
+ sanitizeTagsConfig,
+ TAG_ID_RE,
+ TAGS_FILENAME,
+ type CuratedTagDef,
+} from "./curatedTags";
+
+const evaCollab: CuratedTagDef = {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [
+ {
+ id: "meta",
+ kind: "metadata",
+ pattern: "elfpire|elf ?pire ?eva",
+ enabled: true,
+ },
+ ],
+};
+
+const evaInChat: CuratedTagDef = {
+ id: "eva-in-chat",
+ label: "In chat",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 2,
+ rules: [
+ { id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true },
+ ],
+};
+
+const evaTopic: CuratedTagDef = {
+ id: "eva-topic",
+ label: "Discussed",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 3,
+ rules: [
+ { id: "cap", kind: "caption", pattern: "\\belf ?pire\\b", enabled: true },
+ ],
+};
+
+const DEFS = [evaCollab, evaInChat, evaTopic];
+
+// ─── constants ───
+
+test("TAGS_FILENAME and TAG_ID_RE are the frozen contract", () => {
+ assert.equal(TAGS_FILENAME, "tags.json");
+ for (const ok of ["eva-collab", "a", "a.b_c-d", "x9"]) {
+ assert.equal(TAG_ID_RE.test(ok), true, ok);
+ assert.equal(isTagId(ok), true, ok);
+ }
+ for (const bad of ["", "-lead", ".lead", "Eva", "has space", "a/b", "é"]) {
+ assert.equal(TAG_ID_RE.test(bad), false, bad);
+ assert.equal(isTagId(bad), false, bad);
+ }
+});
+
+test("assignmentKey is channel-first", () => {
+ assert.equal(assignmentKey("legal-mindset", "XZqL6k9IHGA"), "legal-mindset/XZqL6k9IHGA");
+});
+
+// ─── sanitizeTagsConfig ───
+
+test("sanitizeTagsConfig returns an empty config for junk, never throws", () => {
+ for (const junk of [null, undefined, 3, "x", [], { tags: 7 }]) {
+ const cfg = sanitizeTagsConfig(junk);
+ assert.deepEqual(cfg.tags, []);
+ assert.deepEqual(cfg.assignments, {});
+ assert.equal(cfg.version, 1);
+ }
+});
+
+test("sanitizeTagsConfig drops bad ids and keeps the first of a duplicate", () => {
+ const cfg = sanitizeTagsConfig({
+ version: 1,
+ tags: [
+ { id: "Eva-Collab", label: "Collab" }, // lowercased
+ { id: "-nope", label: "bad lead" },
+ { id: "has space" },
+ { id: 4 },
+ "string",
+ { id: "eva-collab", label: "Second wins? no" },
+ ],
+ });
+ assert.deepEqual(
+ cfg.tags.map((t) => t.id),
+ ["eva-collab"],
+ );
+ assert.equal(cfg.tags[0].label, "Collab");
+});
+
+test("sanitizeTagsConfig defaults label to the id and drops empty extras", () => {
+ const cfg = sanitizeTagsConfig({ tags: [{ id: "solo", group: " " }] });
+ assert.deepEqual(cfg.tags[0], { id: "solo", label: "solo" });
+});
+
+test("sanitizeTagsConfig drops unusable rules but keeps a non-compiling one disabled", () => {
+ const cfg = sanitizeTagsConfig({
+ tags: [
+ {
+ id: "t",
+ rules: [
+ { id: "ok", kind: "metadata", pattern: "abc" },
+ { id: "bad-kind", kind: "nope", pattern: "abc" },
+ { id: "no-pattern", kind: "caption", pattern: " " },
+ { id: "broken", kind: "caption", pattern: "a(" },
+ "junk",
+ ],
+ },
+ ],
+ });
+ const rules = cfg.tags[0].rules ?? [];
+ assert.deepEqual(
+ rules.map((r) => r.id),
+ ["ok", "broken"],
+ );
+ assert.equal(rules[0].enabled, true);
+ assert.equal(rules[1].enabled, false);
+ assert.match(String(rules[1].disabledReason), /does not compile/);
+ // The pattern itself is preserved so the operator can fix the typo.
+ assert.equal(rules[1].pattern, "a(");
+});
+
+test("sanitizeTagsConfig gives an id-less rule a positional id and normalizes dates", () => {
+ const cfg = sanitizeTagsConfig({
+ tags: [
+ {
+ id: "t",
+ rules: [
+ {
+ kind: "metadata",
+ pattern: "x",
+ dateFrom: "2026-01-01",
+ dateTo: "20261231",
+ channels: ["a", " ", 3, " b "],
+ },
+ ],
+ },
+ ],
+ });
+ const rule = (cfg.tags[0].rules ?? [])[0];
+ assert.equal(rule.id, "r1");
+ assert.equal(rule.dateFrom, "20260101");
+ assert.equal(rule.dateTo, "20261231");
+ assert.deepEqual(rule.channels, ["a", "b"]);
+});
+
+test("sanitizeTagsConfig keeps well-formed assignments and drops the rest", () => {
+ const cfg = sanitizeTagsConfig({
+ assignments: {
+ "legal-mindset/XZqL6k9IHGA": {
+ manual: ["eva-collab", "eva-collab", "BAD ID", "eva-topic"],
+ suppressed: ["eva-in-chat"],
+ sources: {
+ "eva-collab": { source: "umtool:elfpire-eva", setAt: "2026-09-21T18:04:11Z" },
+ "eva-in-chat": { source: "operator", setAt: "2026-09-21T18:05:02Z" },
+ "not-assigned": { source: "operator", setAt: "2026-09-21T18:05:02Z" },
+ "eva-topic": { source: "operator" },
+ },
+ },
+ "no-slash": { manual: ["x"] },
+ "a/b/c": { manual: ["x"] },
+ "/onlyid": { manual: ["x"] },
+ "empty/one": { manual: [], suppressed: [] },
+ "junk/one": 7,
+ },
+ });
+ assert.deepEqual(Object.keys(cfg.assignments), ["legal-mindset/XZqL6k9IHGA"]);
+ const a = cfg.assignments["legal-mindset/XZqL6k9IHGA"];
+ assert.deepEqual(a.manual, ["eva-collab", "eva-topic"]);
+ assert.deepEqual(a.suppressed, ["eva-in-chat"]);
+ // Provenance survives for a pin AND a suppression; an entry for a tag that is
+ // assigned nowhere, or missing setAt, is dropped.
+ assert.deepEqual(Object.keys(a.sources ?? {}).sort(), ["eva-collab", "eva-in-chat"]);
+ // Video ids are case-sensitive: the key is not lowercased.
+ assert.ok("legal-mindset/XZqL6k9IHGA" in cfg.assignments);
+});
+
+test("sanitizeTagsConfig is idempotent", () => {
+ const once = sanitizeTagsConfig({
+ tags: [{ id: "t", rules: [{ kind: "caption", pattern: "a(" }] }],
+ assignments: { "c/v": { manual: ["t"] } },
+ });
+ assert.deepEqual(sanitizeTagsConfig(JSON.parse(JSON.stringify(once))), once);
+});
+
+// ─── mergeTagDefs ───
+
+test("mergeTagDefs overlays presentation fields and appends site rules", () => {
+ const merged = mergeTagDefs(DEFS, [
+ {
+ id: "eva-collab",
+ label: "On mic",
+ groupLabel: "Elfpire Eva",
+ color: "#b48ead",
+ order: 9,
+ hidden: true,
+ rules: [{ id: "site", kind: "metadata", pattern: "extra", enabled: true }],
+ },
+ ]);
+ const collab = merged.find((t) => t.id === "eva-collab")!;
+ assert.equal(collab.label, "On mic");
+ assert.equal(collab.groupLabel, "Elfpire Eva");
+ assert.equal(collab.color, "#b48ead");
+ assert.equal(collab.order, 9);
+ assert.equal(collab.hidden, true);
+ assert.deepEqual(
+ (collab.rules ?? []).map((r) => r.id),
+ ["meta", "site"],
+ );
+ // Untouched fields survive, and the global input is not mutated.
+ assert.equal(collab.group, "eva");
+ assert.equal(evaCollab.label, "Collab");
+ assert.deepEqual(
+ (evaCollab.rules ?? []).map((r) => r.id),
+ ["meta"],
+ );
+});
+
+test("mergeTagDefs cannot delete a global rule or a global tag", () => {
+ const merged = mergeTagDefs(DEFS, [{ id: "eva-collab", label: "x", rules: [] }]);
+ assert.deepEqual(
+ merged.map((t) => t.id),
+ ["eva-collab", "eva-in-chat", "eva-topic"],
+ );
+ assert.deepEqual(
+ (merged[0].rules ?? []).map((r) => r.id),
+ ["meta"],
+ );
+});
+
+test("mergeTagDefs appends a site-only tag whole", () => {
+ const siteOnly: CuratedTagDef = {
+ id: "site-thing",
+ label: "Site thing",
+ rules: [{ id: "r1", kind: "caption", pattern: "zzz", enabled: true }],
+ };
+ const merged = mergeTagDefs(DEFS, [siteOnly]);
+ assert.equal(merged.length, 4);
+ assert.deepEqual(merged[3], siteOnly);
+});
+
+// ─── effectiveTagsFor ───
+
+test("effectiveTagsFor unions hits with pins and removes suppressions", () => {
+ assert.deepEqual(
+ effectiveTagsFor(["eva-topic"], { manual: ["eva-collab"] }, DEFS),
+ ["eva-collab", "eva-topic"],
+ );
+ assert.deepEqual(
+ effectiveTagsFor(["eva-topic", "eva-in-chat"], { suppressed: ["eva-topic"] }, DEFS),
+ ["eva-in-chat"],
+ );
+ assert.deepEqual(effectiveTagsFor([], undefined, DEFS), []);
+});
+
+test("effectiveTagsFor lets a pin beat a suppression of the same tag", () => {
+ assert.deepEqual(
+ effectiveTagsFor([], { manual: ["eva-topic"], suppressed: ["eva-topic"] }, DEFS),
+ ["eva-topic"],
+ );
+});
+
+test("effectiveTagsFor sorts by def order then id, unknown ids last", () => {
+ const defs: CuratedTagDef[] = [
+ { id: "zed", label: "Zed", order: 1 },
+ { id: "alpha", label: "Alpha", order: 2 },
+ { id: "beta", label: "Beta", order: 2 },
+ ];
+ assert.deepEqual(
+ effectiveTagsFor(["beta", "alpha", "zed", "orphan"], undefined, defs),
+ ["zed", "alpha", "beta", "orphan"],
+ );
+});
+
+// ─── rule evaluation ───
+
+test("chatAuthorOf reads the `<author>: ` prefix contract from liveChat.ts", () => {
+ assert.equal(chatAuthorOf("ElfpireEva: hello there"), "ElfpireEva");
+ assert.equal(chatAuthorOf("@user1: msg: with colon"), "@user1");
+ assert.equal(chatAuthorOf("a message with no author"), null);
+ assert.equal(chatAuthorOf(": leading"), null);
+ assert.equal(chatAuthorOf(""), null);
+});
+
+const base = { channelSlug: "legal-mindset", id: "v1", uploadDate: "20260301" };
+
+test("metadata rules match title, description and keywords, case-insensitively", () => {
+ assert.deepEqual(
+ evaluateTagRules({ ...base, title: "Stream with ELFPIRE Eva" }, DEFS),
+ ["eva-collab"],
+ );
+ assert.deepEqual(
+ evaluateTagRules({ ...base, description: "guest: elf pire eva" }, DEFS),
+ ["eva-collab"],
+ );
+ assert.deepEqual(
+ evaluateTagRules({ ...base, tags: ["vtuber", "elfpire"] }, DEFS),
+ ["eva-collab"],
+ );
+ assert.deepEqual(evaluateTagRules({ ...base, title: "Just a stream" }, DEFS), []);
+});
+
+test("chat-author rules match the author prefix only, never the message body", () => {
+ assert.deepEqual(
+ evaluateTagRules(
+ { ...base, chatCues: [{ text: "someone: elfpire is cool" }] },
+ DEFS,
+ ),
+ [],
+ );
+ assert.deepEqual(
+ evaluateTagRules(
+ { ...base, chatCues: [{ text: "hi" }, { text: "ElfpireEva: hi chat" }] },
+ DEFS,
+ ),
+ ["eva-in-chat"],
+ );
+});
+
+test("caption rules match cue text", () => {
+ assert.deepEqual(
+ evaluateTagRules(
+ { ...base, captionCues: [{ text: "we were talking about Elf pire yesterday" }] },
+ DEFS,
+ ),
+ ["eva-topic"],
+ );
+ assert.deepEqual(
+ evaluateTagRules({ ...base, captionCues: [{ text: "nothing here" }] }, DEFS),
+ [],
+ );
+});
+
+test("a caption scan stops at the first hit", () => {
+ let read = 0;
+ const cues = Array.from({ length: 100 }, (_, i) => ({
+ get text() {
+ read++;
+ return i === 2 ? "elfpire appears" : "filler";
+ },
+ }));
+ assert.deepEqual(evaluateTagRules({ ...base, captionCues: cues }, [evaTopic]), [
+ "eva-topic",
+ ]);
+ assert.equal(read, 3);
+});
+
+test("channel scope and date range are applied before any regex", () => {
+ const scoped: CuratedTagDef = {
+ id: "scoped",
+ label: "Scoped",
+ rules: [
+ {
+ id: "r1",
+ kind: "metadata",
+ pattern: "elfpire",
+ channels: ["legal-mindset"],
+ dateFrom: "20260101",
+ dateTo: "20261231",
+ enabled: true,
+ },
+ ],
+ };
+ const hit = { ...base, title: "elfpire" };
+ assert.deepEqual(evaluateTagRules(hit, [scoped]), ["scoped"]);
+ assert.deepEqual(evaluateTagRules({ ...hit, channelSlug: "other" }, [scoped]), []);
+ assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: "20251231" }, [scoped]), []);
+ assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: "20270101" }, [scoped]), []);
+ // A dated rule cannot fire on a record with no upload date.
+ assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: undefined }, [scoped]), []);
+});
+
+test("a disabled rule never fires, and neither does a broken one", () => {
+ const defs = sanitizeTagsConfig({
+ tags: [
+ {
+ id: "t",
+ rules: [
+ { id: "off", kind: "metadata", pattern: "elfpire", enabled: false },
+ { id: "broken", kind: "metadata", pattern: "a(" },
+ ],
+ },
+ ],
+ }).tags;
+ assert.deepEqual(evaluateTagRules({ ...base, title: "elfpire a(" }, defs), []);
+});
+
+test("compileTagRules partitions by kind and reports emptiness", () => {
+ const compiled = compileTagRules(DEFS);
+ assert.equal(compiled.metadata.length, 1);
+ assert.equal(compiled.chatAuthor.length, 1);
+ assert.equal(compiled.caption.length, 1);
+ assert.equal(compiled.isEmpty, false);
+ const empty = compileTagRules([{ id: "t", label: "T" }]);
+ assert.equal(empty.isEmpty, true);
+ assert.deepEqual(evaluateCompiledRules({ ...base, title: "anything" }, empty), []);
+});
+
+test("a compiled rule set is reusable across videos (no lastIndex carry-over)", () => {
+ const compiled = compileTagRules(DEFS);
+ for (let i = 0; i < 3; i++) {
+ assert.deepEqual(
+ evaluateCompiledRules({ ...base, id: `v${i}`, title: "elfpire" }, compiled),
+ ["eva-collab"],
+ );
+ }
+});
+
+test("all three kinds can fire on one video and the result is sorted", () => {
+ assert.deepEqual(
+ evaluateTagRules(
+ {
+ ...base,
+ title: "elfpire collab",
+ chatCues: [{ text: "ElfpireEva: hi" }],
+ captionCues: [{ text: "elfpire said" }],
+ },
+ DEFS,
+ ),
+ ["eva-collab", "eva-in-chat", "eva-topic"],
+ );
+});
diff --git a/common/lib/curatedTags.ts b/common/lib/curatedTags.ts
@@ -0,0 +1,560 @@
+// Curated per-video tags — an operator-owned vocabulary that cuts ACROSS
+// channels ("every stream where Eva is on mic", on any site, any channel).
+//
+// Three unrelated things in this repo are already called "tags":
+// 1. `TranscriptSummary.tags` — yt-dlp keywords from metadata.info.json
+// (the search scope "tags", relabelled "Keywords" in the UI);
+// 2. AI digest topic tags;
+// 3. THIS — curated tags, carried on a record as `curatedTags`.
+// The record field is therefore NEVER `tags`. See plans/FACTS.md.
+//
+// This module is PURE (no I/O) so it can be unit-tested and shared by the
+// server-side store (common/lib/curatedTagsStore.ts), the index builder
+// (common/controller/curatedTagsIndex.ts), the editor and the export viewer.
+// Disk/merge/defaulting policy lives in the store; this file only knows how to
+// coerce a config, layer a site over the corpus, evaluate rules and fold rule
+// hits together with the operator's pins and suppressions.
+
+// The file name, identical at every layer: `transcripts/tags.json`,
+// `transcripts/sites/<id>/tags.json` and the published `/tags.json`.
+export const TAGS_FILENAME = "tags.json";
+
+// Tag ids are free-form but url/filename-safe and lowercase, so they can travel
+// in a query param, a CLI argument and a JSON key without quoting.
+export const TAG_ID_RE = /^[a-z0-9][a-z0-9._-]*$/;
+
+export const CURATED_TAGS_VERSION = 1;
+
+export type CuratedTagRuleKind = "metadata" | "chat-author" | "caption";
+
+const RULE_KINDS: readonly CuratedTagRuleKind[] = [
+ "metadata",
+ "chat-author",
+ "caption",
+];
+
+export type CuratedTagRule = {
+ // Stable within its tag; used in provenance (`rule:<ruleId>`) and to let a
+ // site layer append rules without colliding with the corpus ones.
+ id: string;
+ kind: CuratedTagRuleKind;
+ // A JS regex source, matched case-insensitively. What it is matched against
+ // depends on `kind` — see evaluateCompiledRules.
+ pattern: string;
+ // Channel slugs this rule may fire on. Absent or empty = every channel.
+ channels?: string[];
+ // Inclusive upload-date bounds, compared digit-wise so both "20260101" and
+ // "2026-01-01" work. null/absent = unbounded.
+ dateFrom?: string | null;
+ dateTo?: string | null;
+ // Defaults true. A rule whose pattern does not compile is KEPT with
+ // `enabled: false` and a `disabledReason` — never dropped, never thrown, so
+ // the operator can see and fix their typo instead of losing the rule.
+ enabled: boolean;
+ disabledReason?: string;
+};
+
+export type CuratedTagDef = {
+ id: string;
+ label: string;
+ // Optional grouping so a chip row can read "Eva: Collab · In chat · Discussed".
+ group?: string;
+ groupLabel?: string;
+ description?: string;
+ color?: string;
+ // Sort key for chips and for the order of a record's `curatedTags`.
+ order?: number;
+ // Presentation only: hidden tags are dropped from a published /tags.json but
+ // still ride on records (a site hides a chip; it does not erase a fact).
+ hidden?: boolean;
+ // Optional — a purely manual tag is legal.
+ rules?: CuratedTagRule[];
+};
+
+// Where a pin or a suppression came from: `operator`, `agent:<label>`,
+// `umtool:<project>` or `rule:<ruleId>`. Free text, recorded verbatim.
+export type CuratedTagProvenance = {
+ source: string;
+ // ISO-8601.
+ setAt: string;
+};
+
+export type CuratedTagAssignment = {
+ // Tags pinned onto this video regardless of what the rules say.
+ manual?: string[];
+ // Tags a rule would have produced but the operator rejected. If a tag is in
+ // BOTH lists, `manual` wins (a later pin beats an older suppression).
+ suppressed?: string[];
+ // Provenance per tag id, recorded for pins AND suppressions.
+ sources?: Record<string, CuratedTagProvenance>;
+};
+
+export type CuratedTagsConfig = {
+ version: number;
+ tags: CuratedTagDef[];
+ // Keyed by `assignmentKey(channelSlug, videoId)`.
+ assignments: Record<string, CuratedTagAssignment>;
+};
+
+// The published, presentation-side document served at `/tags.json`. Counts are
+// per site, computed at index time; compose drops hidden and zero-count tags,
+// and writes no file at all when nothing survives (a 404 is a valid empty
+// state, exactly like /duplicates.json).
+export type PublishedTag = {
+ id: string;
+ label: string;
+ group?: string;
+ groupLabel?: string;
+ color?: string;
+ order?: number;
+ count: number;
+ // slug -> number of videos carrying this tag on that channel.
+ channels: Record<string, number>;
+};
+
+export type PublishedTags = {
+ version: number;
+ tags: PublishedTag[];
+};
+
+// The one key shape for an assignment. Channel slug first so a prefix scan over
+// the assignment keys is a per-channel scan.
+export function assignmentKey(channelSlug: string, id: string): string {
+ return `${channelSlug}/${id}`;
+}
+
+export function isTagId(v: unknown): v is string {
+ return typeof v === "string" && TAG_ID_RE.test(v);
+}
+
+function normalizeTagId(v: unknown): string | null {
+ if (typeof v !== "string") return null;
+ const id = v.trim().toLowerCase();
+ return TAG_ID_RE.test(id) ? id : null;
+}
+
+function trimmedString(v: unknown): string | undefined {
+ if (typeof v !== "string") return undefined;
+ const s = v.trim();
+ return s === "" ? undefined : s;
+}
+
+function dateBound(v: unknown): string | null {
+ const s = trimmedString(v);
+ if (!s) return null;
+ const digits = s.replace(/\D/g, "");
+ return digits === "" ? null : digits;
+}
+
+// Compile once, here, so both sanitize (which disables a broken rule) and
+// compileTagRules (which skips one) agree on what "compiles" means.
+function tryCompile(pattern: string): { re: RegExp } | { error: string } {
+ try {
+ return { re: new RegExp(pattern, "i") };
+ } catch (err) {
+ return { error: err instanceof Error ? err.message : String(err) };
+ }
+}
+
+function coerceRule(raw: unknown, index: number): CuratedTagRule | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ const kind = RULE_KINDS.find((k) => k === r.kind);
+ if (!kind) return null;
+ const pattern = typeof r.pattern === "string" ? r.pattern : "";
+ // A rule with no pattern cannot mean anything; drop it.
+ if (pattern.trim() === "") return null;
+ const id = trimmedString(r.id) ?? `r${index + 1}`;
+ const channels = Array.isArray(r.channels)
+ ? r.channels
+ .filter((c): c is string => typeof c === "string" && c.trim() !== "")
+ .map((c) => c.trim())
+ : [];
+ const dateFrom = dateBound(r.dateFrom);
+ const dateTo = dateBound(r.dateTo);
+ const rule: CuratedTagRule = {
+ id,
+ kind,
+ pattern,
+ enabled: r.enabled !== false, // default true
+ ...(channels.length > 0 ? { channels } : {}),
+ ...(dateFrom ? { dateFrom } : {}),
+ ...(dateTo ? { dateTo } : {}),
+ };
+ const compiled = tryCompile(pattern);
+ if ("error" in compiled) {
+ // Kept, disabled, with the reason — the operator's typo stays visible.
+ rule.enabled = false;
+ rule.disabledReason = `pattern does not compile: ${compiled.error}`;
+ }
+ return rule;
+}
+
+function coerceTagDef(raw: unknown): CuratedTagDef | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ const id = normalizeTagId(r.id);
+ if (!id) return null;
+ const rules = Array.isArray(r.rules)
+ ? r.rules
+ .map((rule, i) => coerceRule(rule, i))
+ .filter((rule): rule is CuratedTagRule => rule !== null)
+ : [];
+ const order =
+ typeof r.order === "number" && Number.isFinite(r.order)
+ ? r.order
+ : undefined;
+ return {
+ id,
+ label: trimmedString(r.label) ?? id,
+ ...(trimmedString(r.group) ? { group: trimmedString(r.group) } : {}),
+ ...(trimmedString(r.groupLabel)
+ ? { groupLabel: trimmedString(r.groupLabel) }
+ : {}),
+ ...(trimmedString(r.description)
+ ? { description: trimmedString(r.description) }
+ : {}),
+ ...(trimmedString(r.color) ? { color: trimmedString(r.color) } : {}),
+ ...(order !== undefined ? { order } : {}),
+ ...(r.hidden === true ? { hidden: true } : {}),
+ ...(rules.length > 0 ? { rules } : {}),
+ };
+}
+
+function coerceAssignment(raw: unknown): CuratedTagAssignment | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ const list = (v: unknown): string[] => {
+ if (!Array.isArray(v)) return [];
+ const out: string[] = [];
+ for (const item of v) {
+ const id = normalizeTagId(item);
+ if (id && !out.includes(id)) out.push(id);
+ }
+ return out;
+ };
+ const manual = list(r.manual);
+ const suppressed = list(r.suppressed);
+ // An assignment that pins and suppresses nothing carries no fact.
+ if (manual.length === 0 && suppressed.length === 0) return null;
+ const sources: Record<string, CuratedTagProvenance> = {};
+ if (r.sources && typeof r.sources === "object") {
+ for (const [key, value] of Object.entries(
+ r.sources as Record<string, unknown>,
+ )) {
+ const tagId = normalizeTagId(key);
+ if (!tagId) continue;
+ if (!manual.includes(tagId) && !suppressed.includes(tagId)) continue;
+ if (!value || typeof value !== "object") continue;
+ const v = value as Record<string, unknown>;
+ const source = trimmedString(v.source);
+ const setAt = trimmedString(v.setAt);
+ if (!source || !setAt) continue;
+ sources[tagId] = { source, setAt };
+ }
+ }
+ return {
+ ...(manual.length > 0 ? { manual } : {}),
+ ...(suppressed.length > 0 ? { suppressed } : {}),
+ ...(Object.keys(sources).length > 0 ? { sources } : {}),
+ };
+}
+
+// Coerce a parsed JSON value into a valid CuratedTagsConfig, dropping malformed
+// entries. Tolerant like coerceAliasConfig (common/lib/searchAliases.ts): it
+// NEVER throws, and missing/invalid input yields an empty config. The one
+// deliberate exception to "drop what is malformed" is a rule whose regex does
+// not compile: that rule is kept, disabled, with a reason.
+export function sanitizeTagsConfig(raw: unknown): CuratedTagsConfig {
+ const empty: CuratedTagsConfig = {
+ version: CURATED_TAGS_VERSION,
+ tags: [],
+ assignments: {},
+ };
+ if (!raw || typeof raw !== "object") return empty;
+ const r = raw as Record<string, unknown>;
+ const version =
+ typeof r.version === "number" && Number.isFinite(r.version)
+ ? r.version
+ : CURATED_TAGS_VERSION;
+
+ const tags: CuratedTagDef[] = [];
+ const seen = new Set<string>();
+ if (Array.isArray(r.tags)) {
+ for (const entry of r.tags) {
+ const def = coerceTagDef(entry);
+ if (!def) continue;
+ if (seen.has(def.id)) continue; // first definition of an id wins
+ seen.add(def.id);
+ tags.push(def);
+ }
+ }
+
+ const assignments: Record<string, CuratedTagAssignment> = {};
+ if (r.assignments && typeof r.assignments === "object") {
+ for (const [key, value] of Object.entries(
+ r.assignments as Record<string, unknown>,
+ )) {
+ // `<channelSlug>/<videoId>` — exactly one slash, both halves non-empty.
+ // Video ids are case-sensitive, so keys are NOT lowercased.
+ const parts = key.split("/");
+ if (parts.length !== 2 || !parts[0].trim() || !parts[1].trim()) continue;
+ const assignment = coerceAssignment(value);
+ if (!assignment) continue;
+ assignments[key] = assignment;
+ }
+ }
+
+ return { version, tags, assignments };
+}
+
+// Layer a site's tags.json over the corpus one.
+//
+// For an id the corpus already defines, the site entry is a FIELD-WISE OVERLAY
+// of label/groupLabel/color/order/hidden and may APPEND rules. It can never
+// delete a corpus rule (and assignments do not live at the site layer at all —
+// an assignment is a fact about a video, not a presentation choice). A site id
+// the corpus does not define becomes a full site-only tag.
+//
+// Deliberately NOT mergeAliases' wholesale replacement: aliases are suggestions,
+// tags are a shared vocabulary that a site may dress up but not gut.
+export function mergeTagDefs(
+ global: CuratedTagDef[],
+ site: CuratedTagDef[],
+): CuratedTagDef[] {
+ const out: CuratedTagDef[] = global.map((def) => ({ ...def }));
+ const byId = new Map<string, CuratedTagDef>();
+ for (const def of out) byId.set(def.id, def);
+ for (const overlay of site) {
+ const base = byId.get(overlay.id);
+ if (!base) {
+ const copy = { ...overlay };
+ out.push(copy);
+ byId.set(copy.id, copy);
+ continue;
+ }
+ if (overlay.label !== undefined) base.label = overlay.label;
+ if (overlay.groupLabel !== undefined) base.groupLabel = overlay.groupLabel;
+ if (overlay.color !== undefined) base.color = overlay.color;
+ if (overlay.order !== undefined) base.order = overlay.order;
+ if (overlay.hidden !== undefined) base.hidden = overlay.hidden;
+ if (overlay.rules && overlay.rules.length > 0) {
+ base.rules = [...(base.rules ?? []), ...overlay.rules];
+ }
+ }
+ return out;
+}
+
+// (rule hits ∪ manual pins) − suppressions, with a pin beating a suppression of
+// the same tag, sorted by def `order` then id so a record's `curatedTags` array
+// is stable across builds.
+//
+// A tag id with no definition (a def deleted after the pin was made) is KEPT —
+// losing an operator's fact silently would be worse — and sorts last by id. It
+// simply never appears in a published /tags.json, which is built from defs.
+export function effectiveTagsFor(
+ ruleHits: string[],
+ assignment: CuratedTagAssignment | undefined,
+ defs: CuratedTagDef[],
+): string[] {
+ const manual = assignment?.manual ?? [];
+ const suppressed = new Set(assignment?.suppressed ?? []);
+ const out = new Set<string>();
+ for (const id of ruleHits) {
+ if (!suppressed.has(id)) out.add(id);
+ }
+ for (const id of manual) out.add(id); // a pin beats a suppression
+ if (out.size === 0) return [];
+
+ const rank = new Map<string, number>();
+ defs.forEach((def, i) => {
+ if (!rank.has(def.id)) rank.set(def.id, def.order ?? i);
+ });
+ const known = (id: string) => rank.has(id);
+ return Array.from(out).sort((a, b) => {
+ if (known(a) !== known(b)) return known(a) ? -1 : 1;
+ if (known(a) && known(b)) {
+ const ra = rank.get(a)!;
+ const rb = rank.get(b)!;
+ if (ra !== rb) return ra - rb;
+ }
+ return a < b ? -1 : a > b ? 1 : 0;
+ });
+}
+
+// ─── Rule evaluation ───
+
+export type TagRuleInput = {
+ channelSlug: string;
+ id: string;
+ title?: string;
+ description?: string;
+ // yt-dlp keywords (TranscriptSummary.tags) — part of the `metadata` haystack.
+ tags?: string[];
+ uploadDate?: string;
+ // Caption cues, for `caption` rules. Only `text` is read.
+ captionCues?: readonly { text: string }[];
+ // Live-chat cues, for `chat-author` rules. Only `text` is read; the author is
+ // the `<author>: ` prefix (see chatAuthorOf).
+ chatCues?: readonly { text: string }[];
+};
+
+export type CompiledTagRule = {
+ tagId: string;
+ ruleId: string;
+ kind: CuratedTagRuleKind;
+ re: RegExp;
+ // null = every channel.
+ channels: Set<string> | null;
+ dateFrom: string | null;
+ dateTo: string | null;
+};
+
+// Rules pre-split by kind so a per-video evaluation never re-partitions them.
+export type CompiledTagRules = {
+ metadata: CompiledTagRule[];
+ chatAuthor: CompiledTagRule[];
+ caption: CompiledTagRule[];
+ // True when there is nothing to evaluate at all — the common case for an
+ // untagged corpus, and the caller's cue to skip cue reads entirely.
+ isEmpty: boolean;
+};
+
+export const EMPTY_COMPILED_TAG_RULES: CompiledTagRules = {
+ metadata: [],
+ chatAuthor: [],
+ caption: [],
+ isEmpty: true,
+};
+
+// Compile every enabled rule ONCE — the index build compiles per build, not per
+// video, and then calls evaluateCompiledRules for each of tens of thousands of
+// records. A rule that does not compile is skipped (sanitizeTagsConfig has
+// already disabled it with a reason).
+export function compileTagRules(defs: CuratedTagDef[]): CompiledTagRules {
+ const metadata: CompiledTagRule[] = [];
+ const chatAuthor: CompiledTagRule[] = [];
+ const caption: CompiledTagRule[] = [];
+ for (const def of defs) {
+ for (const rule of def.rules ?? []) {
+ if (rule.enabled === false) continue;
+ const compiled = tryCompile(rule.pattern);
+ if ("error" in compiled) continue;
+ const entry: CompiledTagRule = {
+ tagId: def.id,
+ ruleId: rule.id,
+ kind: rule.kind,
+ re: compiled.re,
+ channels:
+ rule.channels && rule.channels.length > 0
+ ? new Set(rule.channels)
+ : null,
+ dateFrom: dateBound(rule.dateFrom),
+ dateTo: dateBound(rule.dateTo),
+ };
+ if (rule.kind === "metadata") metadata.push(entry);
+ else if (rule.kind === "chat-author") chatAuthor.push(entry);
+ else caption.push(entry);
+ }
+ }
+ return {
+ metadata,
+ chatAuthor,
+ caption,
+ isEmpty:
+ metadata.length === 0 && chatAuthor.length === 0 && caption.length === 0,
+ };
+}
+
+// Channel scope and date range are cheap string work and are applied BEFORE any
+// regex runs — the whole point of scoping a rule to one channel is not paying
+// for it on the other twenty-nine.
+function ruleApplies(rule: CompiledTagRule, input: TagRuleInput): boolean {
+ if (rule.channels && !rule.channels.has(input.channelSlug)) return false;
+ if (rule.dateFrom || rule.dateTo) {
+ const date = (input.uploadDate ?? "").replace(/\D/g, "");
+ if (!date) return false;
+ if (rule.dateFrom && date < rule.dateFrom) return false;
+ if (rule.dateTo && date > rule.dateTo) return false;
+ }
+ return true;
+}
+
+// The author of a live-chat cue. `common/lib/liveChat.ts` formats every cue as
+// `${author}: ${message}` (and emits the bare message when the renderer carries
+// no author), so the prefix up to the first ": " IS the contract — there is no
+// `author` field on `Cue`. A cue with no such prefix has no author and can
+// never match a chat-author rule.
+export function chatAuthorOf(text: string): string | null {
+ const at = text.indexOf(": ");
+ if (at <= 0) return null;
+ return text.slice(0, at);
+}
+
+// Tag ids whose rules match, sorted and deduped. Does not consult assignments —
+// that is effectiveTagsFor's job.
+export function evaluateCompiledRules(
+ input: TagRuleInput,
+ compiled: CompiledTagRules,
+): string[] {
+ if (compiled.isEmpty) return [];
+ const hits = new Set<string>();
+
+ const metaRules = compiled.metadata.filter((r) => ruleApplies(r, input));
+ if (metaRules.length > 0) {
+ const haystack = [
+ input.title ?? "",
+ input.description ?? "",
+ (input.tags ?? []).join(" "),
+ ].join("\n");
+ for (const rule of metaRules) {
+ if (hits.has(rule.tagId)) continue;
+ if (rule.re.test(haystack)) hits.add(rule.tagId);
+ }
+ }
+
+ // One pass over the cues for all rules of a kind, dropping each rule as soon
+ // as it fires (first hit short-circuits — a tag is on a video or it is not,
+ // and nothing counts hits).
+ const scanCues = (
+ rules: CompiledTagRule[],
+ cues: readonly { text: string }[] | undefined,
+ project: (text: string) => string | null,
+ ) => {
+ if (!cues || cues.length === 0) return;
+ let pending = rules.filter(
+ (r) => ruleApplies(r, input) && !hits.has(r.tagId),
+ );
+ if (pending.length === 0) return;
+ for (const cue of cues) {
+ const subject = project(cue.text ?? "");
+ if (subject === null || subject === "") continue;
+ let fired = false;
+ for (const rule of pending) {
+ if (rule.re.test(subject)) {
+ hits.add(rule.tagId);
+ fired = true;
+ }
+ }
+ if (fired) {
+ pending = pending.filter((r) => !hits.has(r.tagId));
+ if (pending.length === 0) return;
+ }
+ }
+ };
+
+ scanCues(compiled.chatAuthor, input.chatCues, chatAuthorOf);
+ scanCues(compiled.caption, input.captionCues, (text) => text);
+
+ return Array.from(hits).sort();
+}
+
+// Convenience for callers with only a handful of videos (the editor's rule
+// preview, tests). The index build must use compileTagRules once per build and
+// call evaluateCompiledRules per video instead.
+export function evaluateTagRules(
+ input: TagRuleInput,
+ defs: CuratedTagDef[],
+): string[] {
+ return evaluateCompiledRules(input, compileTagRules(defs));
+}
diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts
@@ -26,6 +26,11 @@ export type TranscriptSummary = {
// HLS master-playlist URL for platforms with no iframe embed (Kick VODs).
// Undefined for every other platform. Played via react-player/file + hls.js.
hlsUrl?: string;
+ // Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel
+ // vocabulary, NOT the yt-dlp keywords in `tags` above. OMITTED when empty, so
+ // an untagged corpus's pages stay byte-identical to the ones already on disk
+ // and the build's sha1 skip still short-circuits.
+ curatedTags?: string[];
};
export type DisplaySummary = {
@@ -52,6 +57,11 @@ export type DisplaySummary = {
state?: VideoState;
platform: Platform;
webpageUrl: string;
+ // Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel
+ // vocabulary. A DisplaySummary carries no yt-dlp keywords, but the sibling
+ // TranscriptSummary.tags does and these are NOT those. OMITTED when empty,
+ // for the same byte-identical-pages reason as `state` above.
+ curatedTags?: string[];
};
export type TranscriptDetail = TranscriptSummary & {
diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts
@@ -1,4 +1,5 @@
import { writeFileSync } from "node:fs";
+import type { PublishedTags } from "yt-dlp-transcript-common/lib/curatedTags";
export const CHANNEL = "Test Channel";
export const CHANNEL_SLUG = "test-channel";
@@ -22,7 +23,13 @@ function slug(id: string) {
return `${CHANNEL_SLUG}/${id}`;
}
-function makeSummary(id: string, title: string) {
+// Curated tags (common/lib/curatedTags.ts) carried by the fixture summaries.
+// Two of the three videos are tagged, one is not — the untagged one is the
+// control for "omitted when empty", which is how a real untagged corpus ships.
+export const TAG_COLLAB = "eva-collab";
+export const TAG_TOPIC = "eva-topic";
+
+function makeSummary(id: string, title: string, curatedTags?: string[]) {
return {
slug: slug(id),
id,
@@ -38,6 +45,7 @@ function makeSummary(id: string, title: string) {
isUnlisted: false,
platform: "rumble" as const,
webpageUrl: `https://example.com/${id}`,
+ ...(curatedTags && curatedTags.length > 0 ? { curatedTags } : {}),
};
}
@@ -78,11 +86,44 @@ function largeChatCues(): Cue[] {
export function summaries() {
return [
makeSummary(VIDEO_TRANSCRIPT_ONLY, "Transcript only — no chat"),
- makeSummary(VIDEO_CHAT_SMALL, "Small live chat"),
- makeSummary(VIDEO_CHAT_LARGE, "Large live chat"),
+ makeSummary(VIDEO_CHAT_SMALL, "Small live chat", [TAG_COLLAB]),
+ makeSummary(VIDEO_CHAT_LARGE, "Large live chat", [TAG_COLLAB, TAG_TOPIC]),
];
}
+// The published /tags.json for this fixture site, in the shape compose-site
+// writes: counts are per site and must agree with summaries() above —
+// eva-collab on two videos, eva-topic on one, both on the one channel. A site
+// with no shippable tag writes no file at all, so a 404 is a valid empty state
+// (installRoutes leaves it 404 by default; a spec that wants chips calls
+// installTagRoutes, the same way duplicates.json works).
+export function tagsJson(): PublishedTags {
+ return {
+ version: 1,
+ tags: [
+ {
+ id: TAG_COLLAB,
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ color: "#b48ead",
+ order: 1,
+ count: 2,
+ channels: { [CHANNEL_SLUG]: 2 },
+ },
+ {
+ id: TAG_TOPIC,
+ label: "Discussed",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 2,
+ count: 1,
+ channels: { [CHANNEL_SLUG]: 1 },
+ },
+ ],
+ };
+}
+
export function summariesManifest() {
const list = summaries();
return {
diff --git a/export/e2e/helpers.ts b/export/e2e/helpers.ts
@@ -12,6 +12,7 @@ import {
postsPage,
summaries,
summariesManifest,
+ tagsJson,
transcriptPage,
} from "./fixtures/data";
@@ -94,6 +95,28 @@ export async function installRoutes(page: Page) {
body: "{}",
});
});
+ // Curated tags — absent by default, which is also the real default:
+ // compose-site writes /tags.json only for a site with at least one visible,
+ // non-zero-count tag, so a 404 is a legitimate empty state (exactly like
+ // duplicates.json). A spec that wants the tag chips calls installTagRoutes
+ // afterwards, which takes precedence.
+ await page.route("**/tags.json", async (route) => {
+ await route.fulfill({
+ status: 404,
+ contentType: "application/json",
+ body: "{}",
+ });
+ });
+}
+
+// Serve the populated /tags.json fixture — two curated tags in one group, with
+// counts that agree with the `curatedTags` on summaries(). Call AFTER
+// installRoutes (a later route wins) in any spec that asserts on tag chips or
+// tag filtering.
+export async function installTagRoutes(page: Page) {
+ await page.route("**/tags.json", async (route) => {
+ await fulfillJson(route, tagsJson());
+ });
}
// Charts page fetches: stats dataset + baked templates. Also installs the