Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit ac144c6c416b819802881d77fd3673d9a017728a
parent 0bd2cfd44cad3fc0a0c948548cf4baab263c6118
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 12:56:19 -0400

tags S1.1: curatedTags model, record fields, export fixture

The frozen contract the other two tag slices branch from.

common/lib/curatedTags.ts is pure: the types (CuratedTagRule/Def/Assignment,
CuratedTagsConfig, PublishedTag(s)), TAGS_FILENAME, TAG_ID_RE, and five
functions — sanitizeTagsConfig (tolerant like coerceAliasConfig; the one thing
it does NOT drop is a rule whose regex will not compile, which is kept disabled
with a reason so the operator sees their typo), mergeTagDefs (field-wise
presentation overlay + appended site rules; it can never delete a corpus rule),
effectiveTagsFor ((hits u manual) - suppressed, a pin beating a suppression),
compileTagRules/evaluateCompiledRules (compile once per build, evaluate per
video) and the evaluateTagRules convenience for the editor's preview.

chat-author rules read the `<author>: ` prefix of a chat cue, because that IS
the contract — common/lib/liveChat.ts formats every cue as `${author}: ${text}`
and there is no author field on Cue. Channel scope and date range are applied
before any regex runs, and a cue scan drops each rule as soon as it fires.

The record field is curatedTags, never tags: TranscriptSummary.tags is yt-dlp
keywords. It is omitted when empty so an untagged corpus's pages stay
byte-identical and the build's sha1 skip still short-circuits.

The export fixture tags two of its three summaries and gains tagsJson() plus
installTagRoutes(); installRoutes leaves /tags.json a 404, which is the real
default for a site with nothing to publish.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Acommon/lib/curatedTags.test.ts | 437+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/curatedTags.ts | 560+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/transcripts.ts | 10++++++++++
Mexport/e2e/fixtures/data.ts | 47++++++++++++++++++++++++++++++++++++++++++++---
Mexport/e2e/helpers.ts | 23+++++++++++++++++++++++
5 files changed, 1074 insertions(+), 3 deletions(-)

diff --git a/common/lib/curatedTags.test.ts b/common/lib/curatedTags.test.ts @@ -0,0 +1,437 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + assignmentKey, + chatAuthorOf, + compileTagRules, + effectiveTagsFor, + evaluateCompiledRules, + evaluateTagRules, + isTagId, + mergeTagDefs, + sanitizeTagsConfig, + TAG_ID_RE, + TAGS_FILENAME, + type CuratedTagDef, +} from "./curatedTags"; + +const evaCollab: CuratedTagDef = { + id: "eva-collab", + label: "Collab", + group: "eva", + groupLabel: "Eva", + order: 1, + rules: [ + { + id: "meta", + kind: "metadata", + pattern: "elfpire|elf ?pire ?eva", + enabled: true, + }, + ], +}; + +const evaInChat: CuratedTagDef = { + id: "eva-in-chat", + label: "In chat", + group: "eva", + groupLabel: "Eva", + order: 2, + rules: [ + { id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true }, + ], +}; + +const evaTopic: CuratedTagDef = { + id: "eva-topic", + label: "Discussed", + group: "eva", + groupLabel: "Eva", + order: 3, + rules: [ + { id: "cap", kind: "caption", pattern: "\\belf ?pire\\b", enabled: true }, + ], +}; + +const DEFS = [evaCollab, evaInChat, evaTopic]; + +// ─── constants ─── + +test("TAGS_FILENAME and TAG_ID_RE are the frozen contract", () => { + assert.equal(TAGS_FILENAME, "tags.json"); + for (const ok of ["eva-collab", "a", "a.b_c-d", "x9"]) { + assert.equal(TAG_ID_RE.test(ok), true, ok); + assert.equal(isTagId(ok), true, ok); + } + for (const bad of ["", "-lead", ".lead", "Eva", "has space", "a/b", "é"]) { + assert.equal(TAG_ID_RE.test(bad), false, bad); + assert.equal(isTagId(bad), false, bad); + } +}); + +test("assignmentKey is channel-first", () => { + assert.equal(assignmentKey("legal-mindset", "XZqL6k9IHGA"), "legal-mindset/XZqL6k9IHGA"); +}); + +// ─── sanitizeTagsConfig ─── + +test("sanitizeTagsConfig returns an empty config for junk, never throws", () => { + for (const junk of [null, undefined, 3, "x", [], { tags: 7 }]) { + const cfg = sanitizeTagsConfig(junk); + assert.deepEqual(cfg.tags, []); + assert.deepEqual(cfg.assignments, {}); + assert.equal(cfg.version, 1); + } +}); + +test("sanitizeTagsConfig drops bad ids and keeps the first of a duplicate", () => { + const cfg = sanitizeTagsConfig({ + version: 1, + tags: [ + { id: "Eva-Collab", label: "Collab" }, // lowercased + { id: "-nope", label: "bad lead" }, + { id: "has space" }, + { id: 4 }, + "string", + { id: "eva-collab", label: "Second wins? no" }, + ], + }); + assert.deepEqual( + cfg.tags.map((t) => t.id), + ["eva-collab"], + ); + assert.equal(cfg.tags[0].label, "Collab"); +}); + +test("sanitizeTagsConfig defaults label to the id and drops empty extras", () => { + const cfg = sanitizeTagsConfig({ tags: [{ id: "solo", group: " " }] }); + assert.deepEqual(cfg.tags[0], { id: "solo", label: "solo" }); +}); + +test("sanitizeTagsConfig drops unusable rules but keeps a non-compiling one disabled", () => { + const cfg = sanitizeTagsConfig({ + tags: [ + { + id: "t", + rules: [ + { id: "ok", kind: "metadata", pattern: "abc" }, + { id: "bad-kind", kind: "nope", pattern: "abc" }, + { id: "no-pattern", kind: "caption", pattern: " " }, + { id: "broken", kind: "caption", pattern: "a(" }, + "junk", + ], + }, + ], + }); + const rules = cfg.tags[0].rules ?? []; + assert.deepEqual( + rules.map((r) => r.id), + ["ok", "broken"], + ); + assert.equal(rules[0].enabled, true); + assert.equal(rules[1].enabled, false); + assert.match(String(rules[1].disabledReason), /does not compile/); + // The pattern itself is preserved so the operator can fix the typo. + assert.equal(rules[1].pattern, "a("); +}); + +test("sanitizeTagsConfig gives an id-less rule a positional id and normalizes dates", () => { + const cfg = sanitizeTagsConfig({ + tags: [ + { + id: "t", + rules: [ + { + kind: "metadata", + pattern: "x", + dateFrom: "2026-01-01", + dateTo: "20261231", + channels: ["a", " ", 3, " b "], + }, + ], + }, + ], + }); + const rule = (cfg.tags[0].rules ?? [])[0]; + assert.equal(rule.id, "r1"); + assert.equal(rule.dateFrom, "20260101"); + assert.equal(rule.dateTo, "20261231"); + assert.deepEqual(rule.channels, ["a", "b"]); +}); + +test("sanitizeTagsConfig keeps well-formed assignments and drops the rest", () => { + const cfg = sanitizeTagsConfig({ + assignments: { + "legal-mindset/XZqL6k9IHGA": { + manual: ["eva-collab", "eva-collab", "BAD ID", "eva-topic"], + suppressed: ["eva-in-chat"], + sources: { + "eva-collab": { source: "umtool:elfpire-eva", setAt: "2026-09-21T18:04:11Z" }, + "eva-in-chat": { source: "operator", setAt: "2026-09-21T18:05:02Z" }, + "not-assigned": { source: "operator", setAt: "2026-09-21T18:05:02Z" }, + "eva-topic": { source: "operator" }, + }, + }, + "no-slash": { manual: ["x"] }, + "a/b/c": { manual: ["x"] }, + "/onlyid": { manual: ["x"] }, + "empty/one": { manual: [], suppressed: [] }, + "junk/one": 7, + }, + }); + assert.deepEqual(Object.keys(cfg.assignments), ["legal-mindset/XZqL6k9IHGA"]); + const a = cfg.assignments["legal-mindset/XZqL6k9IHGA"]; + assert.deepEqual(a.manual, ["eva-collab", "eva-topic"]); + assert.deepEqual(a.suppressed, ["eva-in-chat"]); + // Provenance survives for a pin AND a suppression; an entry for a tag that is + // assigned nowhere, or missing setAt, is dropped. + assert.deepEqual(Object.keys(a.sources ?? {}).sort(), ["eva-collab", "eva-in-chat"]); + // Video ids are case-sensitive: the key is not lowercased. + assert.ok("legal-mindset/XZqL6k9IHGA" in cfg.assignments); +}); + +test("sanitizeTagsConfig is idempotent", () => { + const once = sanitizeTagsConfig({ + tags: [{ id: "t", rules: [{ kind: "caption", pattern: "a(" }] }], + assignments: { "c/v": { manual: ["t"] } }, + }); + assert.deepEqual(sanitizeTagsConfig(JSON.parse(JSON.stringify(once))), once); +}); + +// ─── mergeTagDefs ─── + +test("mergeTagDefs overlays presentation fields and appends site rules", () => { + const merged = mergeTagDefs(DEFS, [ + { + id: "eva-collab", + label: "On mic", + groupLabel: "Elfpire Eva", + color: "#b48ead", + order: 9, + hidden: true, + rules: [{ id: "site", kind: "metadata", pattern: "extra", enabled: true }], + }, + ]); + const collab = merged.find((t) => t.id === "eva-collab")!; + assert.equal(collab.label, "On mic"); + assert.equal(collab.groupLabel, "Elfpire Eva"); + assert.equal(collab.color, "#b48ead"); + assert.equal(collab.order, 9); + assert.equal(collab.hidden, true); + assert.deepEqual( + (collab.rules ?? []).map((r) => r.id), + ["meta", "site"], + ); + // Untouched fields survive, and the global input is not mutated. + assert.equal(collab.group, "eva"); + assert.equal(evaCollab.label, "Collab"); + assert.deepEqual( + (evaCollab.rules ?? []).map((r) => r.id), + ["meta"], + ); +}); + +test("mergeTagDefs cannot delete a global rule or a global tag", () => { + const merged = mergeTagDefs(DEFS, [{ id: "eva-collab", label: "x", rules: [] }]); + assert.deepEqual( + merged.map((t) => t.id), + ["eva-collab", "eva-in-chat", "eva-topic"], + ); + assert.deepEqual( + (merged[0].rules ?? []).map((r) => r.id), + ["meta"], + ); +}); + +test("mergeTagDefs appends a site-only tag whole", () => { + const siteOnly: CuratedTagDef = { + id: "site-thing", + label: "Site thing", + rules: [{ id: "r1", kind: "caption", pattern: "zzz", enabled: true }], + }; + const merged = mergeTagDefs(DEFS, [siteOnly]); + assert.equal(merged.length, 4); + assert.deepEqual(merged[3], siteOnly); +}); + +// ─── effectiveTagsFor ─── + +test("effectiveTagsFor unions hits with pins and removes suppressions", () => { + assert.deepEqual( + effectiveTagsFor(["eva-topic"], { manual: ["eva-collab"] }, DEFS), + ["eva-collab", "eva-topic"], + ); + assert.deepEqual( + effectiveTagsFor(["eva-topic", "eva-in-chat"], { suppressed: ["eva-topic"] }, DEFS), + ["eva-in-chat"], + ); + assert.deepEqual(effectiveTagsFor([], undefined, DEFS), []); +}); + +test("effectiveTagsFor lets a pin beat a suppression of the same tag", () => { + assert.deepEqual( + effectiveTagsFor([], { manual: ["eva-topic"], suppressed: ["eva-topic"] }, DEFS), + ["eva-topic"], + ); +}); + +test("effectiveTagsFor sorts by def order then id, unknown ids last", () => { + const defs: CuratedTagDef[] = [ + { id: "zed", label: "Zed", order: 1 }, + { id: "alpha", label: "Alpha", order: 2 }, + { id: "beta", label: "Beta", order: 2 }, + ]; + assert.deepEqual( + effectiveTagsFor(["beta", "alpha", "zed", "orphan"], undefined, defs), + ["zed", "alpha", "beta", "orphan"], + ); +}); + +// ─── rule evaluation ─── + +test("chatAuthorOf reads the `<author>: ` prefix contract from liveChat.ts", () => { + assert.equal(chatAuthorOf("ElfpireEva: hello there"), "ElfpireEva"); + assert.equal(chatAuthorOf("@user1: msg: with colon"), "@user1"); + assert.equal(chatAuthorOf("a message with no author"), null); + assert.equal(chatAuthorOf(": leading"), null); + assert.equal(chatAuthorOf(""), null); +}); + +const base = { channelSlug: "legal-mindset", id: "v1", uploadDate: "20260301" }; + +test("metadata rules match title, description and keywords, case-insensitively", () => { + assert.deepEqual( + evaluateTagRules({ ...base, title: "Stream with ELFPIRE Eva" }, DEFS), + ["eva-collab"], + ); + assert.deepEqual( + evaluateTagRules({ ...base, description: "guest: elf pire eva" }, DEFS), + ["eva-collab"], + ); + assert.deepEqual( + evaluateTagRules({ ...base, tags: ["vtuber", "elfpire"] }, DEFS), + ["eva-collab"], + ); + assert.deepEqual(evaluateTagRules({ ...base, title: "Just a stream" }, DEFS), []); +}); + +test("chat-author rules match the author prefix only, never the message body", () => { + assert.deepEqual( + evaluateTagRules( + { ...base, chatCues: [{ text: "someone: elfpire is cool" }] }, + DEFS, + ), + [], + ); + assert.deepEqual( + evaluateTagRules( + { ...base, chatCues: [{ text: "hi" }, { text: "ElfpireEva: hi chat" }] }, + DEFS, + ), + ["eva-in-chat"], + ); +}); + +test("caption rules match cue text", () => { + assert.deepEqual( + evaluateTagRules( + { ...base, captionCues: [{ text: "we were talking about Elf pire yesterday" }] }, + DEFS, + ), + ["eva-topic"], + ); + assert.deepEqual( + evaluateTagRules({ ...base, captionCues: [{ text: "nothing here" }] }, DEFS), + [], + ); +}); + +test("a caption scan stops at the first hit", () => { + let read = 0; + const cues = Array.from({ length: 100 }, (_, i) => ({ + get text() { + read++; + return i === 2 ? "elfpire appears" : "filler"; + }, + })); + assert.deepEqual(evaluateTagRules({ ...base, captionCues: cues }, [evaTopic]), [ + "eva-topic", + ]); + assert.equal(read, 3); +}); + +test("channel scope and date range are applied before any regex", () => { + const scoped: CuratedTagDef = { + id: "scoped", + label: "Scoped", + rules: [ + { + id: "r1", + kind: "metadata", + pattern: "elfpire", + channels: ["legal-mindset"], + dateFrom: "20260101", + dateTo: "20261231", + enabled: true, + }, + ], + }; + const hit = { ...base, title: "elfpire" }; + assert.deepEqual(evaluateTagRules(hit, [scoped]), ["scoped"]); + assert.deepEqual(evaluateTagRules({ ...hit, channelSlug: "other" }, [scoped]), []); + assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: "20251231" }, [scoped]), []); + assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: "20270101" }, [scoped]), []); + // A dated rule cannot fire on a record with no upload date. + assert.deepEqual(evaluateTagRules({ ...hit, uploadDate: undefined }, [scoped]), []); +}); + +test("a disabled rule never fires, and neither does a broken one", () => { + const defs = sanitizeTagsConfig({ + tags: [ + { + id: "t", + rules: [ + { id: "off", kind: "metadata", pattern: "elfpire", enabled: false }, + { id: "broken", kind: "metadata", pattern: "a(" }, + ], + }, + ], + }).tags; + assert.deepEqual(evaluateTagRules({ ...base, title: "elfpire a(" }, defs), []); +}); + +test("compileTagRules partitions by kind and reports emptiness", () => { + const compiled = compileTagRules(DEFS); + assert.equal(compiled.metadata.length, 1); + assert.equal(compiled.chatAuthor.length, 1); + assert.equal(compiled.caption.length, 1); + assert.equal(compiled.isEmpty, false); + const empty = compileTagRules([{ id: "t", label: "T" }]); + assert.equal(empty.isEmpty, true); + assert.deepEqual(evaluateCompiledRules({ ...base, title: "anything" }, empty), []); +}); + +test("a compiled rule set is reusable across videos (no lastIndex carry-over)", () => { + const compiled = compileTagRules(DEFS); + for (let i = 0; i < 3; i++) { + assert.deepEqual( + evaluateCompiledRules({ ...base, id: `v${i}`, title: "elfpire" }, compiled), + ["eva-collab"], + ); + } +}); + +test("all three kinds can fire on one video and the result is sorted", () => { + assert.deepEqual( + evaluateTagRules( + { + ...base, + title: "elfpire collab", + chatCues: [{ text: "ElfpireEva: hi" }], + captionCues: [{ text: "elfpire said" }], + }, + DEFS, + ), + ["eva-collab", "eva-in-chat", "eva-topic"], + ); +}); diff --git a/common/lib/curatedTags.ts b/common/lib/curatedTags.ts @@ -0,0 +1,560 @@ +// Curated per-video tags — an operator-owned vocabulary that cuts ACROSS +// channels ("every stream where Eva is on mic", on any site, any channel). +// +// Three unrelated things in this repo are already called "tags": +// 1. `TranscriptSummary.tags` — yt-dlp keywords from metadata.info.json +// (the search scope "tags", relabelled "Keywords" in the UI); +// 2. AI digest topic tags; +// 3. THIS — curated tags, carried on a record as `curatedTags`. +// The record field is therefore NEVER `tags`. See plans/FACTS.md. +// +// This module is PURE (no I/O) so it can be unit-tested and shared by the +// server-side store (common/lib/curatedTagsStore.ts), the index builder +// (common/controller/curatedTagsIndex.ts), the editor and the export viewer. +// Disk/merge/defaulting policy lives in the store; this file only knows how to +// coerce a config, layer a site over the corpus, evaluate rules and fold rule +// hits together with the operator's pins and suppressions. + +// The file name, identical at every layer: `transcripts/tags.json`, +// `transcripts/sites/<id>/tags.json` and the published `/tags.json`. +export const TAGS_FILENAME = "tags.json"; + +// Tag ids are free-form but url/filename-safe and lowercase, so they can travel +// in a query param, a CLI argument and a JSON key without quoting. +export const TAG_ID_RE = /^[a-z0-9][a-z0-9._-]*$/; + +export const CURATED_TAGS_VERSION = 1; + +export type CuratedTagRuleKind = "metadata" | "chat-author" | "caption"; + +const RULE_KINDS: readonly CuratedTagRuleKind[] = [ + "metadata", + "chat-author", + "caption", +]; + +export type CuratedTagRule = { + // Stable within its tag; used in provenance (`rule:<ruleId>`) and to let a + // site layer append rules without colliding with the corpus ones. + id: string; + kind: CuratedTagRuleKind; + // A JS regex source, matched case-insensitively. What it is matched against + // depends on `kind` — see evaluateCompiledRules. + pattern: string; + // Channel slugs this rule may fire on. Absent or empty = every channel. + channels?: string[]; + // Inclusive upload-date bounds, compared digit-wise so both "20260101" and + // "2026-01-01" work. null/absent = unbounded. + dateFrom?: string | null; + dateTo?: string | null; + // Defaults true. A rule whose pattern does not compile is KEPT with + // `enabled: false` and a `disabledReason` — never dropped, never thrown, so + // the operator can see and fix their typo instead of losing the rule. + enabled: boolean; + disabledReason?: string; +}; + +export type CuratedTagDef = { + id: string; + label: string; + // Optional grouping so a chip row can read "Eva: Collab · In chat · Discussed". + group?: string; + groupLabel?: string; + description?: string; + color?: string; + // Sort key for chips and for the order of a record's `curatedTags`. + order?: number; + // Presentation only: hidden tags are dropped from a published /tags.json but + // still ride on records (a site hides a chip; it does not erase a fact). + hidden?: boolean; + // Optional — a purely manual tag is legal. + rules?: CuratedTagRule[]; +}; + +// Where a pin or a suppression came from: `operator`, `agent:<label>`, +// `umtool:<project>` or `rule:<ruleId>`. Free text, recorded verbatim. +export type CuratedTagProvenance = { + source: string; + // ISO-8601. + setAt: string; +}; + +export type CuratedTagAssignment = { + // Tags pinned onto this video regardless of what the rules say. + manual?: string[]; + // Tags a rule would have produced but the operator rejected. If a tag is in + // BOTH lists, `manual` wins (a later pin beats an older suppression). + suppressed?: string[]; + // Provenance per tag id, recorded for pins AND suppressions. + sources?: Record<string, CuratedTagProvenance>; +}; + +export type CuratedTagsConfig = { + version: number; + tags: CuratedTagDef[]; + // Keyed by `assignmentKey(channelSlug, videoId)`. + assignments: Record<string, CuratedTagAssignment>; +}; + +// The published, presentation-side document served at `/tags.json`. Counts are +// per site, computed at index time; compose drops hidden and zero-count tags, +// and writes no file at all when nothing survives (a 404 is a valid empty +// state, exactly like /duplicates.json). +export type PublishedTag = { + id: string; + label: string; + group?: string; + groupLabel?: string; + color?: string; + order?: number; + count: number; + // slug -> number of videos carrying this tag on that channel. + channels: Record<string, number>; +}; + +export type PublishedTags = { + version: number; + tags: PublishedTag[]; +}; + +// The one key shape for an assignment. Channel slug first so a prefix scan over +// the assignment keys is a per-channel scan. +export function assignmentKey(channelSlug: string, id: string): string { + return `${channelSlug}/${id}`; +} + +export function isTagId(v: unknown): v is string { + return typeof v === "string" && TAG_ID_RE.test(v); +} + +function normalizeTagId(v: unknown): string | null { + if (typeof v !== "string") return null; + const id = v.trim().toLowerCase(); + return TAG_ID_RE.test(id) ? id : null; +} + +function trimmedString(v: unknown): string | undefined { + if (typeof v !== "string") return undefined; + const s = v.trim(); + return s === "" ? undefined : s; +} + +function dateBound(v: unknown): string | null { + const s = trimmedString(v); + if (!s) return null; + const digits = s.replace(/\D/g, ""); + return digits === "" ? null : digits; +} + +// Compile once, here, so both sanitize (which disables a broken rule) and +// compileTagRules (which skips one) agree on what "compiles" means. +function tryCompile(pattern: string): { re: RegExp } | { error: string } { + try { + return { re: new RegExp(pattern, "i") }; + } catch (err) { + return { error: err instanceof Error ? err.message : String(err) }; + } +} + +function coerceRule(raw: unknown, index: number): CuratedTagRule | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + const kind = RULE_KINDS.find((k) => k === r.kind); + if (!kind) return null; + const pattern = typeof r.pattern === "string" ? r.pattern : ""; + // A rule with no pattern cannot mean anything; drop it. + if (pattern.trim() === "") return null; + const id = trimmedString(r.id) ?? `r${index + 1}`; + const channels = Array.isArray(r.channels) + ? r.channels + .filter((c): c is string => typeof c === "string" && c.trim() !== "") + .map((c) => c.trim()) + : []; + const dateFrom = dateBound(r.dateFrom); + const dateTo = dateBound(r.dateTo); + const rule: CuratedTagRule = { + id, + kind, + pattern, + enabled: r.enabled !== false, // default true + ...(channels.length > 0 ? { channels } : {}), + ...(dateFrom ? { dateFrom } : {}), + ...(dateTo ? { dateTo } : {}), + }; + const compiled = tryCompile(pattern); + if ("error" in compiled) { + // Kept, disabled, with the reason — the operator's typo stays visible. + rule.enabled = false; + rule.disabledReason = `pattern does not compile: ${compiled.error}`; + } + return rule; +} + +function coerceTagDef(raw: unknown): CuratedTagDef | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + const id = normalizeTagId(r.id); + if (!id) return null; + const rules = Array.isArray(r.rules) + ? r.rules + .map((rule, i) => coerceRule(rule, i)) + .filter((rule): rule is CuratedTagRule => rule !== null) + : []; + const order = + typeof r.order === "number" && Number.isFinite(r.order) + ? r.order + : undefined; + return { + id, + label: trimmedString(r.label) ?? id, + ...(trimmedString(r.group) ? { group: trimmedString(r.group) } : {}), + ...(trimmedString(r.groupLabel) + ? { groupLabel: trimmedString(r.groupLabel) } + : {}), + ...(trimmedString(r.description) + ? { description: trimmedString(r.description) } + : {}), + ...(trimmedString(r.color) ? { color: trimmedString(r.color) } : {}), + ...(order !== undefined ? { order } : {}), + ...(r.hidden === true ? { hidden: true } : {}), + ...(rules.length > 0 ? { rules } : {}), + }; +} + +function coerceAssignment(raw: unknown): CuratedTagAssignment | null { + if (!raw || typeof raw !== "object") return null; + const r = raw as Record<string, unknown>; + const list = (v: unknown): string[] => { + if (!Array.isArray(v)) return []; + const out: string[] = []; + for (const item of v) { + const id = normalizeTagId(item); + if (id && !out.includes(id)) out.push(id); + } + return out; + }; + const manual = list(r.manual); + const suppressed = list(r.suppressed); + // An assignment that pins and suppresses nothing carries no fact. + if (manual.length === 0 && suppressed.length === 0) return null; + const sources: Record<string, CuratedTagProvenance> = {}; + if (r.sources && typeof r.sources === "object") { + for (const [key, value] of Object.entries( + r.sources as Record<string, unknown>, + )) { + const tagId = normalizeTagId(key); + if (!tagId) continue; + if (!manual.includes(tagId) && !suppressed.includes(tagId)) continue; + if (!value || typeof value !== "object") continue; + const v = value as Record<string, unknown>; + const source = trimmedString(v.source); + const setAt = trimmedString(v.setAt); + if (!source || !setAt) continue; + sources[tagId] = { source, setAt }; + } + } + return { + ...(manual.length > 0 ? { manual } : {}), + ...(suppressed.length > 0 ? { suppressed } : {}), + ...(Object.keys(sources).length > 0 ? { sources } : {}), + }; +} + +// Coerce a parsed JSON value into a valid CuratedTagsConfig, dropping malformed +// entries. Tolerant like coerceAliasConfig (common/lib/searchAliases.ts): it +// NEVER throws, and missing/invalid input yields an empty config. The one +// deliberate exception to "drop what is malformed" is a rule whose regex does +// not compile: that rule is kept, disabled, with a reason. +export function sanitizeTagsConfig(raw: unknown): CuratedTagsConfig { + const empty: CuratedTagsConfig = { + version: CURATED_TAGS_VERSION, + tags: [], + assignments: {}, + }; + if (!raw || typeof raw !== "object") return empty; + const r = raw as Record<string, unknown>; + const version = + typeof r.version === "number" && Number.isFinite(r.version) + ? r.version + : CURATED_TAGS_VERSION; + + const tags: CuratedTagDef[] = []; + const seen = new Set<string>(); + if (Array.isArray(r.tags)) { + for (const entry of r.tags) { + const def = coerceTagDef(entry); + if (!def) continue; + if (seen.has(def.id)) continue; // first definition of an id wins + seen.add(def.id); + tags.push(def); + } + } + + const assignments: Record<string, CuratedTagAssignment> = {}; + if (r.assignments && typeof r.assignments === "object") { + for (const [key, value] of Object.entries( + r.assignments as Record<string, unknown>, + )) { + // `<channelSlug>/<videoId>` — exactly one slash, both halves non-empty. + // Video ids are case-sensitive, so keys are NOT lowercased. + const parts = key.split("/"); + if (parts.length !== 2 || !parts[0].trim() || !parts[1].trim()) continue; + const assignment = coerceAssignment(value); + if (!assignment) continue; + assignments[key] = assignment; + } + } + + return { version, tags, assignments }; +} + +// Layer a site's tags.json over the corpus one. +// +// For an id the corpus already defines, the site entry is a FIELD-WISE OVERLAY +// of label/groupLabel/color/order/hidden and may APPEND rules. It can never +// delete a corpus rule (and assignments do not live at the site layer at all — +// an assignment is a fact about a video, not a presentation choice). A site id +// the corpus does not define becomes a full site-only tag. +// +// Deliberately NOT mergeAliases' wholesale replacement: aliases are suggestions, +// tags are a shared vocabulary that a site may dress up but not gut. +export function mergeTagDefs( + global: CuratedTagDef[], + site: CuratedTagDef[], +): CuratedTagDef[] { + const out: CuratedTagDef[] = global.map((def) => ({ ...def })); + const byId = new Map<string, CuratedTagDef>(); + for (const def of out) byId.set(def.id, def); + for (const overlay of site) { + const base = byId.get(overlay.id); + if (!base) { + const copy = { ...overlay }; + out.push(copy); + byId.set(copy.id, copy); + continue; + } + if (overlay.label !== undefined) base.label = overlay.label; + if (overlay.groupLabel !== undefined) base.groupLabel = overlay.groupLabel; + if (overlay.color !== undefined) base.color = overlay.color; + if (overlay.order !== undefined) base.order = overlay.order; + if (overlay.hidden !== undefined) base.hidden = overlay.hidden; + if (overlay.rules && overlay.rules.length > 0) { + base.rules = [...(base.rules ?? []), ...overlay.rules]; + } + } + return out; +} + +// (rule hits ∪ manual pins) − suppressions, with a pin beating a suppression of +// the same tag, sorted by def `order` then id so a record's `curatedTags` array +// is stable across builds. +// +// A tag id with no definition (a def deleted after the pin was made) is KEPT — +// losing an operator's fact silently would be worse — and sorts last by id. It +// simply never appears in a published /tags.json, which is built from defs. +export function effectiveTagsFor( + ruleHits: string[], + assignment: CuratedTagAssignment | undefined, + defs: CuratedTagDef[], +): string[] { + const manual = assignment?.manual ?? []; + const suppressed = new Set(assignment?.suppressed ?? []); + const out = new Set<string>(); + for (const id of ruleHits) { + if (!suppressed.has(id)) out.add(id); + } + for (const id of manual) out.add(id); // a pin beats a suppression + if (out.size === 0) return []; + + const rank = new Map<string, number>(); + defs.forEach((def, i) => { + if (!rank.has(def.id)) rank.set(def.id, def.order ?? i); + }); + const known = (id: string) => rank.has(id); + return Array.from(out).sort((a, b) => { + if (known(a) !== known(b)) return known(a) ? -1 : 1; + if (known(a) && known(b)) { + const ra = rank.get(a)!; + const rb = rank.get(b)!; + if (ra !== rb) return ra - rb; + } + return a < b ? -1 : a > b ? 1 : 0; + }); +} + +// ─── Rule evaluation ─── + +export type TagRuleInput = { + channelSlug: string; + id: string; + title?: string; + description?: string; + // yt-dlp keywords (TranscriptSummary.tags) — part of the `metadata` haystack. + tags?: string[]; + uploadDate?: string; + // Caption cues, for `caption` rules. Only `text` is read. + captionCues?: readonly { text: string }[]; + // Live-chat cues, for `chat-author` rules. Only `text` is read; the author is + // the `<author>: ` prefix (see chatAuthorOf). + chatCues?: readonly { text: string }[]; +}; + +export type CompiledTagRule = { + tagId: string; + ruleId: string; + kind: CuratedTagRuleKind; + re: RegExp; + // null = every channel. + channels: Set<string> | null; + dateFrom: string | null; + dateTo: string | null; +}; + +// Rules pre-split by kind so a per-video evaluation never re-partitions them. +export type CompiledTagRules = { + metadata: CompiledTagRule[]; + chatAuthor: CompiledTagRule[]; + caption: CompiledTagRule[]; + // True when there is nothing to evaluate at all — the common case for an + // untagged corpus, and the caller's cue to skip cue reads entirely. + isEmpty: boolean; +}; + +export const EMPTY_COMPILED_TAG_RULES: CompiledTagRules = { + metadata: [], + chatAuthor: [], + caption: [], + isEmpty: true, +}; + +// Compile every enabled rule ONCE — the index build compiles per build, not per +// video, and then calls evaluateCompiledRules for each of tens of thousands of +// records. A rule that does not compile is skipped (sanitizeTagsConfig has +// already disabled it with a reason). +export function compileTagRules(defs: CuratedTagDef[]): CompiledTagRules { + const metadata: CompiledTagRule[] = []; + const chatAuthor: CompiledTagRule[] = []; + const caption: CompiledTagRule[] = []; + for (const def of defs) { + for (const rule of def.rules ?? []) { + if (rule.enabled === false) continue; + const compiled = tryCompile(rule.pattern); + if ("error" in compiled) continue; + const entry: CompiledTagRule = { + tagId: def.id, + ruleId: rule.id, + kind: rule.kind, + re: compiled.re, + channels: + rule.channels && rule.channels.length > 0 + ? new Set(rule.channels) + : null, + dateFrom: dateBound(rule.dateFrom), + dateTo: dateBound(rule.dateTo), + }; + if (rule.kind === "metadata") metadata.push(entry); + else if (rule.kind === "chat-author") chatAuthor.push(entry); + else caption.push(entry); + } + } + return { + metadata, + chatAuthor, + caption, + isEmpty: + metadata.length === 0 && chatAuthor.length === 0 && caption.length === 0, + }; +} + +// Channel scope and date range are cheap string work and are applied BEFORE any +// regex runs — the whole point of scoping a rule to one channel is not paying +// for it on the other twenty-nine. +function ruleApplies(rule: CompiledTagRule, input: TagRuleInput): boolean { + if (rule.channels && !rule.channels.has(input.channelSlug)) return false; + if (rule.dateFrom || rule.dateTo) { + const date = (input.uploadDate ?? "").replace(/\D/g, ""); + if (!date) return false; + if (rule.dateFrom && date < rule.dateFrom) return false; + if (rule.dateTo && date > rule.dateTo) return false; + } + return true; +} + +// The author of a live-chat cue. `common/lib/liveChat.ts` formats every cue as +// `${author}: ${message}` (and emits the bare message when the renderer carries +// no author), so the prefix up to the first ": " IS the contract — there is no +// `author` field on `Cue`. A cue with no such prefix has no author and can +// never match a chat-author rule. +export function chatAuthorOf(text: string): string | null { + const at = text.indexOf(": "); + if (at <= 0) return null; + return text.slice(0, at); +} + +// Tag ids whose rules match, sorted and deduped. Does not consult assignments — +// that is effectiveTagsFor's job. +export function evaluateCompiledRules( + input: TagRuleInput, + compiled: CompiledTagRules, +): string[] { + if (compiled.isEmpty) return []; + const hits = new Set<string>(); + + const metaRules = compiled.metadata.filter((r) => ruleApplies(r, input)); + if (metaRules.length > 0) { + const haystack = [ + input.title ?? "", + input.description ?? "", + (input.tags ?? []).join(" "), + ].join("\n"); + for (const rule of metaRules) { + if (hits.has(rule.tagId)) continue; + if (rule.re.test(haystack)) hits.add(rule.tagId); + } + } + + // One pass over the cues for all rules of a kind, dropping each rule as soon + // as it fires (first hit short-circuits — a tag is on a video or it is not, + // and nothing counts hits). + const scanCues = ( + rules: CompiledTagRule[], + cues: readonly { text: string }[] | undefined, + project: (text: string) => string | null, + ) => { + if (!cues || cues.length === 0) return; + let pending = rules.filter( + (r) => ruleApplies(r, input) && !hits.has(r.tagId), + ); + if (pending.length === 0) return; + for (const cue of cues) { + const subject = project(cue.text ?? ""); + if (subject === null || subject === "") continue; + let fired = false; + for (const rule of pending) { + if (rule.re.test(subject)) { + hits.add(rule.tagId); + fired = true; + } + } + if (fired) { + pending = pending.filter((r) => !hits.has(r.tagId)); + if (pending.length === 0) return; + } + } + }; + + scanCues(compiled.chatAuthor, input.chatCues, chatAuthorOf); + scanCues(compiled.caption, input.captionCues, (text) => text); + + return Array.from(hits).sort(); +} + +// Convenience for callers with only a handful of videos (the editor's rule +// preview, tests). The index build must use compileTagRules once per build and +// call evaluateCompiledRules per video instead. +export function evaluateTagRules( + input: TagRuleInput, + defs: CuratedTagDef[], +): string[] { + return evaluateCompiledRules(input, compileTagRules(defs)); +} diff --git a/common/lib/transcripts.ts b/common/lib/transcripts.ts @@ -26,6 +26,11 @@ export type TranscriptSummary = { // HLS master-playlist URL for platforms with no iframe embed (Kick VODs). // Undefined for every other platform. Played via react-player/file + hls.js. hlsUrl?: string; + // Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel + // vocabulary, NOT the yt-dlp keywords in `tags` above. OMITTED when empty, so + // an untagged corpus's pages stay byte-identical to the ones already on disk + // and the build's sha1 skip still short-circuits. + curatedTags?: string[]; }; export type DisplaySummary = { @@ -52,6 +57,11 @@ export type DisplaySummary = { state?: VideoState; platform: Platform; webpageUrl: string; + // Curated tag ids (common/lib/curatedTags.ts) — the operator's cross-channel + // vocabulary. A DisplaySummary carries no yt-dlp keywords, but the sibling + // TranscriptSummary.tags does and these are NOT those. OMITTED when empty, + // for the same byte-identical-pages reason as `state` above. + curatedTags?: string[]; }; export type TranscriptDetail = TranscriptSummary & { diff --git a/export/e2e/fixtures/data.ts b/export/e2e/fixtures/data.ts @@ -1,4 +1,5 @@ import { writeFileSync } from "node:fs"; +import type { PublishedTags } from "yt-dlp-transcript-common/lib/curatedTags"; export const CHANNEL = "Test Channel"; export const CHANNEL_SLUG = "test-channel"; @@ -22,7 +23,13 @@ function slug(id: string) { return `${CHANNEL_SLUG}/${id}`; } -function makeSummary(id: string, title: string) { +// Curated tags (common/lib/curatedTags.ts) carried by the fixture summaries. +// Two of the three videos are tagged, one is not — the untagged one is the +// control for "omitted when empty", which is how a real untagged corpus ships. +export const TAG_COLLAB = "eva-collab"; +export const TAG_TOPIC = "eva-topic"; + +function makeSummary(id: string, title: string, curatedTags?: string[]) { return { slug: slug(id), id, @@ -38,6 +45,7 @@ function makeSummary(id: string, title: string) { isUnlisted: false, platform: "rumble" as const, webpageUrl: `https://example.com/${id}`, + ...(curatedTags && curatedTags.length > 0 ? { curatedTags } : {}), }; } @@ -78,11 +86,44 @@ function largeChatCues(): Cue[] { export function summaries() { return [ makeSummary(VIDEO_TRANSCRIPT_ONLY, "Transcript only — no chat"), - makeSummary(VIDEO_CHAT_SMALL, "Small live chat"), - makeSummary(VIDEO_CHAT_LARGE, "Large live chat"), + makeSummary(VIDEO_CHAT_SMALL, "Small live chat", [TAG_COLLAB]), + makeSummary(VIDEO_CHAT_LARGE, "Large live chat", [TAG_COLLAB, TAG_TOPIC]), ]; } +// The published /tags.json for this fixture site, in the shape compose-site +// writes: counts are per site and must agree with summaries() above — +// eva-collab on two videos, eva-topic on one, both on the one channel. A site +// with no shippable tag writes no file at all, so a 404 is a valid empty state +// (installRoutes leaves it 404 by default; a spec that wants chips calls +// installTagRoutes, the same way duplicates.json works). +export function tagsJson(): PublishedTags { + return { + version: 1, + tags: [ + { + id: TAG_COLLAB, + label: "Collab", + group: "eva", + groupLabel: "Eva", + color: "#b48ead", + order: 1, + count: 2, + channels: { [CHANNEL_SLUG]: 2 }, + }, + { + id: TAG_TOPIC, + label: "Discussed", + group: "eva", + groupLabel: "Eva", + order: 2, + count: 1, + channels: { [CHANNEL_SLUG]: 1 }, + }, + ], + }; +} + export function summariesManifest() { const list = summaries(); return { diff --git a/export/e2e/helpers.ts b/export/e2e/helpers.ts @@ -12,6 +12,7 @@ import { postsPage, summaries, summariesManifest, + tagsJson, transcriptPage, } from "./fixtures/data"; @@ -94,6 +95,28 @@ export async function installRoutes(page: Page) { body: "{}", }); }); + // Curated tags — absent by default, which is also the real default: + // compose-site writes /tags.json only for a site with at least one visible, + // non-zero-count tag, so a 404 is a legitimate empty state (exactly like + // duplicates.json). A spec that wants the tag chips calls installTagRoutes + // afterwards, which takes precedence. + await page.route("**/tags.json", async (route) => { + await route.fulfill({ + status: 404, + contentType: "application/json", + body: "{}", + }); + }); +} + +// Serve the populated /tags.json fixture — two curated tags in one group, with +// counts that agree with the `curatedTags` on summaries(). Call AFTER +// installRoutes (a later route wins) in any spec that asserts on tag chips or +// tag filtering. +export async function installTagRoutes(page: Page) { + await page.route("**/tags.json", async (route) => { + await fulfillJson(route, tagsJson()); + }); } // Charts page fetches: stats dataset + baked templates. Also installs the