commit 3096e352ba2c3424065986aa1a15cc3e1b8df4c3
parent de35c0490c045f21d0a1e1dad6dc60eaa708631d
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 13:51:09 -0400
Merge branch 'main' into tags/editor
Diffstat:
45 files changed, 3361 insertions(+), 52 deletions(-)
diff --git a/.gitignore b/.gitignore
@@ -63,6 +63,7 @@ yarn-error.log*
/export/public/chart-templates.json
/export/public/search-aliases.json
/export/public/duplicates.json
+/export/public/tags.json
/export/public/site.json
/export/public/_headers
/export/public/sw.js
diff --git a/AGENTS.md b/AGENTS.md
@@ -198,6 +198,19 @@ apps*, not [DEPLOY_DOCKER.md](DEPLOY_DOCKER.md), which is about *building sites*
| `transcripts/index.mdb` | The LMDB transcript index. Key-only range scans over its `byChannel` sub-DB are cheap; see `common/controller/recencyIndex.ts`. |
| `transcripts/saved-videos/` | Persisted source-video store. |
| `transcripts/search-aliases.json`, `duplicates*.json` | Corpus-wide curated data. |
+| `transcripts/tags.json` | **Curated per-video tags** — the cross-channel vocabulary AND every assignment. Curated data: never hand-edited, never a scratch file. |
+
+**Curated tags are written through ONE path, and it is not your text editor.**
+`transcripts/tags.json` holds the operator's vocabulary (`eva-collab`, rules that
+re-evaluate at index build) and every pin/suppression with its provenance; a site's
+`sites/<id>/tags.json` is presentation only. Every writer — editor UI, `pnpm ops`, umtool —
+goes through `applyTagAssignments` in `common/lib/curatedTagsStore.ts`, which validates,
+records who claimed what, and writes the file once, atomically. Hand-editing it loses
+provenance and races whatever is running. Tests use temp dirs.
+
+Three unrelated things in this repo are called "tags": the yt-dlp keywords on a record
+(`TranscriptSummary.tags`), AI digest topic tags, and these. The record field for these is
+**`curatedTags`**, never `tags` — see `plans/FACTS.md`, "Naming hazards".
**The roots a channel's media may be moved to are named entities**, `settings.storage.locations`
(id, label, root, `autoRepoint`, and the volume UUID learned at the last probe) — managed on
diff --git a/common/bin/compose-site.ts b/common/bin/compose-site.ts
@@ -35,6 +35,13 @@ import { buildSiteDescriptor, type PublicSiteDescriptor } from "../lib/siteDescr
import { shipsPwa } from "../lib/archive/contract";
import { renderHeadersFile } from "../lib/archive/headers";
import { effectiveSiteAliases } from "../lib/aliasesStore";
+import { effectiveSiteTags } from "../lib/curatedTagsStore";
+import { TAGS_FILENAME } from "../lib/curatedTags";
+import {
+ TAG_COUNTS_FILENAME,
+ publishedTagsFrom,
+ type TagCountsFile,
+} from "../controller/curatedTagsIndex";
import {
buildSiteCorpus,
renderSiteLlmsTxt,
@@ -137,10 +144,16 @@ async function emitAiFiles(paths: ReturnType<typeof getPaths>): Promise<void> {
/* no digests manifest for this site */
}
+ // Whether this site published a curated vocabulary. Presence of the file is
+ // the fact — it was written (or removed) just above, by the same rule that
+ // decides whether any tag means anything on this site.
+ const hasTags = await exists(path.join(paths.exportPublicDir, TAGS_FILENAME));
+
const corpus = buildSiteCorpus(descriptor, {
hasArchives,
postCounts,
digestCounts,
+ hasTags,
});
await writeFile(
path.join(paths.exportPublicDir, "corpus.json"),
@@ -751,6 +764,37 @@ async function main(): Promise<void> {
JSON.stringify({ aliases }),
);
+ // --- curated tags (corpus vocabulary + this site's overlay + its counts) ---
+ // The vocabulary is read from source at compose time, like the aliases; the
+ // counts come from the per-site tag-counts.json build:index wrote while it
+ // was already streaming this site's summaries.
+ //
+ // What ships is only what means something HERE: hidden tags are dropped, and
+ // so is any tag with no video on this site — a corpus-wide tag that matched
+ // nothing here gets no chip here. Nothing left → no file at all, and the 404
+ // is the empty state (the same contract duplicates.json has). The stale copy
+ // is removed in that case, so a site that loses its last tagged video stops
+ // advertising tags on the very next build.
+ const tagDefs = effectiveSiteTags(paths, siteId);
+ let tagCounts: TagCountsFile | null = null;
+ try {
+ tagCounts = JSON.parse(
+ await readFile(
+ path.join(paths.exportSitesIndexDir, siteId, TAG_COUNTS_FILENAME),
+ "utf8",
+ ),
+ ) as TagCountsFile;
+ } catch {
+ /* no counts staged for this site yet — treated as "nothing tagged" */
+ }
+ const publishedTags = publishedTagsFrom(tagDefs, tagCounts);
+ const tagsDest = path.join(paths.exportPublicDir, TAGS_FILENAME);
+ if (publishedTags.tags.length > 0) {
+ await writeFile(tagsDest, JSON.stringify(publishedTags));
+ } else {
+ await rm(tagsDest, { force: true });
+ }
+
// --- duplicate-shorts report (global → site-filtered) ---
// The detector writes one global duplicates.json over the whole channel pool;
// each site only serves its own channels, so filter clusters to the site's
diff --git a/common/components/FiltersPanel.tsx b/common/components/FiltersPanel.tsx
@@ -13,7 +13,7 @@
// to be the outer one. It also doubles as the inline collapse, driven by the
// session's persisted `filtersCollapsed`.
-import { Fragment, useCallback } from "react";
+import { Fragment, useCallback, useMemo } from "react";
import { ChevronRightIcon } from "lucide-react";
import {
MISSING_STATES,
@@ -32,6 +32,10 @@ import {
DEFAULT_FLUSH_INTERVAL_MS,
} from "./SearchSessionContext";
import { useSearchData } from "./SearchDataContext";
+import {
+ groupPublishedTags,
+ selectableTags,
+} from "../lib/publishedTags";
export default function FiltersPanel({
// Inside the sheet the sheet itself IS the disclosure, so the wrapper stays
@@ -401,6 +405,7 @@ export default function FiltersPanel({
</div>
</div>
)}
+ <TagChipRow />
<div className="flex flex-wrap items-center gap-x-3 gap-y-1">
<span className="text-xs uppercase tracking-wide text-muted-foreground">
Type
@@ -735,6 +740,150 @@ function ProfilesRow({
);
}
+// The curated-tag chip row (lib/curatedTags.ts) — "Eva: Collab · In chat ·
+// Discussed" beside the channel-group chips.
+//
+// Built ONLY from this site's published /tags.json, which is the operator's
+// decision and the reason there is no static list anywhere in this file: a tag
+// defined corpus-wide but matching nothing on THIS site does not appear here,
+// and neither does a hidden one, because compose-site never wrote it into this
+// site's document. The whole row is absent when the site publishes nothing —
+// which is every site until someone curates a tag, and every site built before
+// corpus spec 4.
+//
+// OR across the chips: selecting Collab and In chat is every stream with
+// either, never only the ones with both. A selected chip clears on click, so
+// the row is its own reset and needs no second control.
+function TagChipRow() {
+ const { curatedTags } = useSearchData();
+ const { draftTags, toggleDraftTag } = useSearchSession();
+
+ // selectableTags is the belt to compose-site's braces: a stale or
+ // hand-written document must not be able to offer a chip that finds nothing.
+ const publishable = useMemo(() => selectableTags(curatedTags), [curatedTags]);
+ const groups = useMemo(() => groupPublishedTags(publishable), [publishable]);
+
+ // Tags the SELECTION carries that this site does not publish. A `?tg=` link
+ // written against another site in the family, or against this one before a
+ // rebuild, arrives with a perfectly valid id that has no chip — and without
+ // this the Filters badge and the "N selected" line would say a filter is on
+ // while nothing in the row is lit, and (if it is the only tag) the results
+ // would go to zero with nothing to click off. So the selection ALWAYS has a
+ // chip, even when the vocabulary does not have the tag.
+ const unpublished = useMemo(() => {
+ const known = new Set(publishable.map((t) => t.id));
+ return Array.from(draftTags)
+ .filter((id) => !known.has(id))
+ .sort();
+ }, [publishable, draftTags]);
+
+ if (groups.length === 0 && unpublished.length === 0) return null;
+
+ return (
+ <div
+ className="flex flex-col gap-2 w-full"
+ data-testid="tag-chip-row"
+ >
+ <div className="flex flex-wrap items-center gap-x-3 gap-y-1">
+ <span className="text-xs uppercase tracking-wide text-muted-foreground">
+ Tags
+ </span>
+ <span className="text-xs text-muted-foreground">
+ {draftTags.size === 0
+ ? "any"
+ : `${draftTags.size} selected — a video with ANY of them`}
+ </span>
+ </div>
+ {/* One wrapping flow, phone-first: a group is a label plus its chips and
+ wraps as a unit, so a narrow viewport gets one group per line rather
+ than a torn row. */}
+ <div className="flex flex-wrap items-center gap-x-4 gap-y-2">
+ {groups.map((group) => (
+ <div
+ key={group.id || "ungrouped"}
+ data-testid="tag-chip-group"
+ data-group-id={group.id}
+ className="flex flex-wrap items-center gap-1.5 min-w-0"
+ >
+ {group.label && (
+ <span className="text-xs text-muted-foreground shrink-0">
+ {group.label}:
+ </span>
+ )}
+ {group.tags.map((tag) => {
+ const selected = draftTags.has(tag.id);
+ return (
+ <button
+ key={tag.id}
+ type="button"
+ aria-pressed={selected}
+ data-testid="tag-chip"
+ data-tag-id={tag.id}
+ title={
+ selected
+ ? `Clear "${tag.label}"`
+ : `Keep only videos tagged "${tag.label}"`
+ }
+ onClick={() => toggleDraftTag(tag.id)}
+ className={cn(
+ "flex items-center gap-1.5 rounded border px-2.5 py-1.5 text-sm select-none min-w-0 transition-colors",
+ selected
+ ? "border-primary bg-primary/10 text-foreground"
+ : "border-border bg-card/60 hover:bg-accent hover:text-accent-foreground",
+ )}
+ >
+ {tag.color && (
+ <span
+ aria-hidden="true"
+ className="inline-block size-2 shrink-0 rounded-full"
+ style={{ background: tag.color }}
+ />
+ )}
+ <span className="truncate">{tag.label}</span>
+ {/* The count is this site's, computed at index time — the
+ same number the chip's filter will produce. */}
+ <span className="text-xs text-muted-foreground shrink-0">
+ {tag.count}
+ </span>
+ </button>
+ );
+ })}
+ </div>
+ ))}
+ {unpublished.length > 0 && (
+ <div
+ data-testid="tag-chip-group"
+ data-group-id="__unpublished"
+ className="flex flex-wrap items-center gap-1.5 min-w-0"
+ >
+ {unpublished.map((id) => (
+ <button
+ key={id}
+ type="button"
+ aria-pressed={true}
+ data-testid="tag-chip"
+ data-tag-id={id}
+ data-unpublished="true"
+ title={`"${id}" is not published by this site — click to clear it`}
+ onClick={() => toggleDraftTag(id)}
+ className="flex items-center gap-1.5 rounded border border-dashed border-primary px-2.5 py-1.5 text-sm select-none min-w-0 bg-primary/5 transition-colors hover:bg-accent hover:text-accent-foreground"
+ >
+ <span className="truncate">{id}</span>
+ <span className="text-xs text-muted-foreground shrink-0">
+ (not on this site)
+ </span>
+ <span aria-hidden="true" className="text-xs shrink-0">
+ ×
+ </span>
+ </button>
+ ))}
+ </div>
+ )}
+ </div>
+ </div>
+ );
+}
+
function ChannelAllToggle({
channelOptions,
draftExcludedChannels,
diff --git a/common/components/QueryLeafView.tsx b/common/components/QueryLeafView.tsx
@@ -37,13 +37,19 @@ type Props = {
onWrapInGroup?: () => void;
};
+// LABELS ONLY. The `tags` scope searches yt-dlp KEYWORDS from
+// metadata.info.json (TranscriptSummary.tags) and is shown as "Keywords",
+// because "Tags" now means the operator's curated per-video vocabulary
+// (lib/curatedTags.ts) — the chip row in the filter panel. The code token, the
+// URL and the MCP `scopes` enum stay `"tags"`: renaming those would break every
+// saved query and share link to fix a word on a screen.
const SCOPE_LABELS: Record<LayerScope, string> = {
transcripts: "Transcripts",
chat: "Live chat",
posts: "Posts",
metadata: "Title / channel",
description: "Description",
- tags: "Tags",
+ tags: "Keywords",
};
export default function QueryLeafView({
@@ -155,8 +161,8 @@ export default function QueryLeafView({
: "Search video descriptions..."
: leaf.scope === "tags"
? leaf.useRegex
- ? "Tags regex..."
- : "Search video tags..."
+ ? "Keywords regex..."
+ : "Search video keywords..."
: leaf.scope === "chat"
? leaf.useRegex
? "Live chat regex..."
diff --git a/common/components/SearchBar.tsx b/common/components/SearchBar.tsx
@@ -45,6 +45,7 @@ export default function SearchBar({ nav }: { nav?: ReactNode }) {
draftStates,
draftDateFrom,
draftDateTo,
+ draftTags,
} = useSearchSession();
// From xl the filters are an inline block under the bar with their own
@@ -63,6 +64,7 @@ export default function SearchBar({ nav }: { nav?: ReactNode }) {
if (draftNaa || draftNar) n += 1;
if (draftStates.size < MISSING_STATES.length + 1) n += 1;
if (draftDateFrom || draftDateTo) n += 1;
+ if (draftTags.size > 0) n += 1;
return n;
}, [
draftExcludedChannels,
@@ -74,6 +76,7 @@ export default function SearchBar({ nav }: { nav?: ReactNode }) {
draftStates,
draftDateFrom,
draftDateTo,
+ draftTags,
]);
const submit = (
diff --git a/common/components/SearchDataContext.tsx b/common/components/SearchDataContext.tsx
@@ -19,6 +19,8 @@ import { useSummaries, type SummariesState } from "./summariesCache";
import { useSubsManifest } from "./subsCache";
import { usePostsManifest } from "./postsCache";
import { useSearchAliases, fetchAliases } from "./aliasesCache";
+import { useCuratedTags } from "./tagsCache";
+import type { PublishedTag } from "../lib/curatedTags";
import { mergeAliases, type SearchAlias } from "../lib/searchAliases";
import { makeId } from "./originId";
import type { DisplaySummary } from "../lib/transcripts";
@@ -68,6 +70,13 @@ export type SearchDataValue = {
// Known search-alias dictionary for the leaf suggestion chip. Single-site: the
// site's own list. Hub: merged across federated origins. Empty when unauthored.
aliases: SearchAlias[];
+ // The curated tags THIS SITE publishes, with its own per-site counts
+ // (/tags.json). Empty when the site publishes none — which is both "nothing
+ // curated yet" and "built before corpus spec 4", and in either case means the
+ // filter panel offers no tag chips at all. Read from the page's own origin in
+ // both modes: a hub's counts are the hub's, and summing member documents here
+ // would put a number in front of someone that no site can reproduce.
+ curatedTags: PublishedTag[];
};
// Single-site mode has no provenance accents; a module constant keeps the
@@ -92,6 +101,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) {
const subsManifest = useSubsManifest("").data ?? null;
const postsManifest = usePostsManifest("").data ?? null;
const aliases = useSearchAliases("");
+ const curatedTags = useCuratedTags("");
const manifest = summariesState.manifest;
const groups = useMemo<ChannelGroup[]>(() => {
@@ -139,6 +149,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) {
channelKeyOf: (t) => t.channel,
accentOf: NO_ACCENT,
aliases,
+ curatedTags,
}),
[
summariesState,
@@ -148,6 +159,7 @@ export function SingleSiteDataProvider({ children }: { children: ReactNode }) {
groups,
defaultGroupId,
aliases,
+ curatedTags,
],
);
@@ -382,6 +394,11 @@ export function MultiSiteDataProvider({
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [aliasesSettled]);
+ // The hub's OWN /tags.json, not a merge of its members'. See the field note
+ // on SearchDataValue: a federated count nobody can reproduce is worse than no
+ // chip, so until a hub build writes one, hub mode offers no tag chips.
+ const curatedTags = useCuratedTags("");
+
// Synthetic merged summaries manifest. TranscriptSearch reads channels/groups
// from the context (above), not from here, but the field is part of the
// SummariesState contract, so provide a coherent merged view.
@@ -443,6 +460,7 @@ export function MultiSiteDataProvider({
channelKeyOf: (t) => t.channelSlug,
accentOf,
aliases,
+ curatedTags,
}),
[
summariesState,
@@ -453,6 +471,7 @@ export function MultiSiteDataProvider({
defaultGroupId,
accentOf,
aliases,
+ curatedTags,
],
);
diff --git a/common/components/SearchResults.tsx b/common/components/SearchResults.tsx
@@ -40,6 +40,8 @@ import {
type LeafInfo,
type ResultGroup,
} from "./SearchSessionContext";
+import { useSearchData } from "./SearchDataContext";
+import type { PublishedTag } from "../lib/curatedTags";
// Rough first-paint guess for one full result card (header + a couple of
// leaf sections + a handful of hits). After mount, ResizeObserver measures
@@ -496,6 +498,17 @@ const ResultCard = memo(function ResultCard({
expanded: boolean;
onToggleExpand: (slug: string) => void;
}) {
+ // The site's published tag vocabulary, for labelling this record's
+ // `curatedTags`. A tag the site does not publish (hidden here, zero-count
+ // here, or defined after this page was built) is still shown — by its id.
+ // The fact is on the record either way, and a raw id is more honest than
+ // silently dropping a tag someone put there.
+ const { curatedTags: publishedTags } = useSearchData();
+ const tagById = useMemo<ReadonlyMap<string, PublishedTag>>(
+ () => new Map(publishedTags.map((t) => [t.id, t])),
+ [publishedTags],
+ );
+
// Other copies of this video elsewhere in the archive.
//
// FEDERATION GUARD: in hub mode a group's slug may be an origin-qualified id,
@@ -624,6 +637,34 @@ const ResultCard = memo(function ResultCard({
{group.hits.length > 0 &&
` · ${group.hits.length} hit${group.hits.length === 1 ? "" : "s"}`}
</span>
+ {/* Curated tags this record carries. Non-interactive spans: the card
+ header is already one big <Button>, so a clickable chip here would
+ be a control inside a control. */}
+ {group.curatedTags && group.curatedTags.length > 0 && (
+ <span className="basis-full sm:basis-auto flex flex-wrap items-center gap-1 shrink-0">
+ {group.curatedTags.map((id) => {
+ const def = tagById.get(id);
+ return (
+ <span
+ key={id}
+ data-testid="card-tag"
+ data-tag-id={id}
+ title={def?.groupLabel ? `${def.groupLabel}: ${def.label}` : undefined}
+ className="flex items-center gap-1 rounded border border-border bg-muted/60 px-1.5 py-0.5 text-[10px] text-muted-foreground"
+ >
+ {def?.color && (
+ <span
+ aria-hidden="true"
+ className="inline-block size-1.5 shrink-0 rounded-full"
+ style={{ background: def.color }}
+ />
+ )}
+ {def?.label ?? id}
+ </span>
+ );
+ })}
+ </span>
+ )}
</Button>
</div>
<button
diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx
@@ -78,6 +78,7 @@ import {
type ChartShape,
} from "../lib/chartShare";
import type { DisplaySummary, Platform } from "../lib/transcripts";
+import { isTagId } from "../lib/curatedTags";
import type { Post } from "../lib/posts";
import { makeId, splitId } from "./originId";
import { sortGroups, type ChannelGroup } from "../lib/channelGroups";
@@ -149,6 +150,11 @@ export type ResultGroup = {
platform: Platform;
uploadDate: string;
hits: LayerHit[];
+ // The operator's curated tag ids on this record (lib/curatedTags.ts), when it
+ // carries any — so a card can show what it was tagged as. Absent on posts, on
+ // untagged videos, and on every record from a site built before corpus spec
+ // 4. Labels come from the site's /tags.json; the id is the fallback.
+ curatedTags?: string[];
// Provenance accent of the source origin (hub mode only); undefined
// single-site, so no marker renders.
accent?: string;
@@ -365,6 +371,12 @@ function useSearchSessionState() {
// Inclusive upload-date bounds ("YYYYMMDD"); "" => unbounded on that end.
const [committedDateFrom, setCommittedDateFrom] = useState("");
const [committedDateTo, setCommittedDateTo] = useState("");
+ // Curated tag ids selected in the chip row, ORed: a video carrying ANY of
+ // them passes. An EMPTY set is no filter at all — the row starts empty, and
+ // on a site that publishes no /tags.json it is empty forever.
+ const [committedTags, setCommittedTags] = useState<ReadonlySet<string>>(
+ () => new Set(),
+ );
const committedChannelsKey = useMemo(
() => Array.from(committedExcludedChannels).sort().join(" "),
[committedExcludedChannels],
@@ -396,6 +408,9 @@ function useSearchSessionState() {
);
const [draftDateFrom, setDraftDateFrom] = useState("");
const [draftDateTo, setDraftDateTo] = useState("");
+ const [draftTags, setDraftTags] = useState<ReadonlySet<string>>(
+ () => new Set(),
+ );
const [hydrated, setHydrated] = useState(false);
const [profiles, setProfiles] = useState<Record<string, FilterSnapshot>>({});
@@ -703,6 +718,14 @@ function useSearchSessionState() {
if (t.isLivestream ? committedNol : committedNov) return false;
if (t.ageRestricted ? committedNar : committedNaa) return false;
if (!committedStates.has(summaryState(t))) return false;
+ // tg — curated tags, ORed across the selection. Mirrors
+ // lib/search/evalTree.ts:passesFilters, which is the same rule applied by
+ // the MCP and the record-level evaluator; when one moves the other must.
+ if (committedTags.size > 0) {
+ const on = t.curatedTags;
+ if (!on || on.length === 0) return false;
+ if (!on.some((id) => committedTags.has(id))) return false;
+ }
return true;
};
}, [
@@ -715,6 +738,7 @@ function useSearchSessionState() {
committedNaa,
committedNar,
committedStates,
+ committedTags,
]);
const committedStatesKey = useMemo(
@@ -722,7 +746,14 @@ function useSearchSessionState() {
[committedStates],
);
- const filterKey = `${committedChannelsKey}|${committedNov ? 1 : 0}|${committedNop ? 1 : 0}|${committedNol ? 1 : 0}|${committedNaa ? 1 : 0}|${committedNar ? 1 : 0}|${committedStatesKey}|${committedDateFrom}|${committedDateTo}`;
+ // Sorted, so a selection built in a different click order is the same key and
+ // does not re-run the pipeline.
+ const committedTagsKey = useMemo(
+ () => Array.from(committedTags).sort().join(" "),
+ [committedTags],
+ );
+
+ const filterKey = `${committedChannelsKey}|${committedNov ? 1 : 0}|${committedNop ? 1 : 0}|${committedNol ? 1 : 0}|${committedNaa ? 1 : 0}|${committedNar ? 1 : 0}|${committedStatesKey}|${committedDateFrom}|${committedDateTo}|${committedTagsKey}`;
// The scope universe is videos AND posts. The two namespaces are disjoint;
// searchEval partitions them per-leaf (see EvalCtx.postScopeSlugs) so a posts
@@ -732,11 +763,32 @@ function useSearchSessionState() {
if (!transcripts) return [];
const out: string[] = [];
for (const t of transcripts) if (passesFilter(t)) out.push(t.slug);
- if (!committedNop) for (const slug of postScopeSlugs) out.push(slug);
+ // A post carries no curated tags, so a tag filter drops the whole posts
+ // corpus rather than letting it through un-filtered — the same answer
+ // passesFilter gives an untagged video. See tagFilteredPostScope below:
+ // the DEFAULT scope is only half of it, because a `posts`-scope leaf
+ // carries its own slug set past this.
+ if (!committedNop && committedTags.size === 0) {
+ for (const slug of postScopeSlugs) out.push(slug);
+ }
return out;
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [transcripts, filterKey, postScopeSlugs, committedNop]);
+ // The posts slug set as the PIPELINE sees it. A `posts`-scope leaf is fed
+ // from here and not from the global scope above, so gating only the global
+ // build left an explicit scopes:['posts'] leaf searching the whole posts
+ // corpus while a tag chip was on: the rows were then dropped by the display
+ // fold, but `totalHits` sums the pipeline's hits BEFORE that fold, so the
+ // header read "3 videos, 8 hits" with hits nobody could see.
+ //
+ // `null` (not `[]`) is the "no posts in scope" signal the pipeline already
+ // understands — the same value it gets when no leaf needs posts at all.
+ const tagFilteredPostScope = useMemo<ReadonlySet<string> | null>(
+ () => (committedTags.size > 0 ? null : postScopeSlugs),
+ [committedTags, postScopeSlugs],
+ );
+
const hasActiveQuery = useMemo(
() => isNodeActive(committedRoot),
[committedRoot],
@@ -765,7 +817,7 @@ function useSearchSessionState() {
globalScope: globalScopeSlugs,
summaries: transcripts,
chatScopeSlugs: needsChatManifests ? chatScopeSlugs : null,
- postScopeSlugs: needsPostsManifests ? postScopeSlugs : null,
+ postScopeSlugs: needsPostsManifests ? tagFilteredPostScope : null,
initialHitLimit: hitLimit,
concurrency: fetchConcurrency,
flushIntervalMs,
@@ -822,6 +874,9 @@ function useSearchSessionState() {
platform: t.platform,
uploadDate: t.uploadDate,
hits: [],
+ ...(t.curatedTags && t.curatedTags.length > 0
+ ? { curatedTags: t.curatedTags }
+ : {}),
accent: accentOf(splitId(t.slug).origin),
});
}
@@ -846,6 +901,9 @@ function useSearchSessionState() {
platform: t.platform,
uploadDate: t.uploadDate,
hits: hitsBySlug.get(t.slug) ?? [],
+ ...(t.curatedTags && t.curatedTags.length > 0
+ ? { curatedTags: t.curatedTags }
+ : {}),
accent: accentOf(splitId(t.slug).origin),
});
}
@@ -860,6 +918,8 @@ function useSearchSessionState() {
if (!post) continue;
if (committedDateFrom && post.uploadDate < committedDateFrom) continue;
if (committedDateTo && post.uploadDate > committedDateTo) continue;
+ // See globalScopeSlugs: a tag filter excludes posts, which carry none.
+ if (committedTags.size > 0) continue;
// A post is either still up or deleted at the source — the two states of
// the six that can apply to it — so it participates in the same
// availability filter videos use.
@@ -897,6 +957,7 @@ function useSearchSessionState() {
committedDateFrom,
committedDateTo,
committedStates,
+ committedTags,
]);
const leafStates = treeProgress?.leafStates ?? new Map<string, LeafState>();
@@ -943,7 +1004,8 @@ function useSearchSessionState() {
draftNar !== committedNar ||
!sameSet(draftStates, committedStates) ||
draftDateFrom !== committedDateFrom ||
- draftDateTo !== committedDateTo;
+ draftDateTo !== committedDateTo ||
+ !sameSet(draftTags, committedTags);
const committedSnapshot = useMemo<FilterSnapshot>(() => {
const deltas = buildSnapshotDeltas(
@@ -959,6 +1021,9 @@ function useSearchSessionState() {
writeSnapshotStates(snap, committedStates);
if (committedDateFrom) snap.dateFrom = committedDateFrom;
if (committedDateTo) snap.dateTo = committedDateTo;
+ // Omitted when empty, so a profile saved on an untagged site stays
+ // byte-identical to one saved before tags existed.
+ if (committedTags.size > 0) snap.tg = Array.from(committedTags).sort();
snap.query = stringifyRoot(committedRoot);
return snap;
}, [
@@ -972,6 +1037,7 @@ function useSearchSessionState() {
committedStates,
committedDateFrom,
committedDateTo,
+ committedTags,
committedRoot,
]);
@@ -991,6 +1057,7 @@ function useSearchSessionState() {
writeSnapshotStates(snap, draftStates);
if (draftDateFrom) snap.dateFrom = draftDateFrom;
if (draftDateTo) snap.dateTo = draftDateTo;
+ if (draftTags.size > 0) snap.tg = Array.from(draftTags).sort();
snap.query = stringifyRoot(draftRoot);
return snap;
}, [
@@ -1005,6 +1072,7 @@ function useSearchSessionState() {
draftStates,
draftDateFrom,
draftDateTo,
+ draftTags,
draftRoot,
]);
@@ -1018,6 +1086,9 @@ function useSearchSessionState() {
setCommittedStates(new Set(draftStates));
setCommittedDateFrom(draftDateFrom);
setCommittedDateTo(draftDateTo);
+ setCommittedTags(new Set(draftTags));
+ // `tg` survives this on purpose: it is not a legacy or share-v1 key, and
+ // the effect below rewrites it from the committed selection.
stripAllFilterParamsFromUrl();
}, [
draftExcludedChannels,
@@ -1029,6 +1100,7 @@ function useSearchSessionState() {
draftStates,
draftDateFrom,
draftDateTo,
+ draftTags,
]);
// ─── Hydration ────────────────────────────────────────────────────────────
@@ -1085,6 +1157,16 @@ function useSearchSessionState() {
let initialStates: ReadonlySet<VideoState> = new Set(VIDEO_STATES);
let initialDateFrom = "";
let initialDateTo = "";
+ // `tg` is read from the URL FIRST and independently of the share-v1 /
+ // legacy branch below, because it is not part of either schema: it is the
+ // live representation of the committed tag selection, so its presence on
+ // the URL always wins and its absence falls through to the snapshot.
+ // isTagId, not a bare non-empty check: the URL is an entry point like the
+ // MCP's parseSearchArgs and must agree with it about what an id is. A
+ // malformed token can only ever produce a silent empty result.
+ const tagsFromUrl = params.getAll("tg").filter((t) => isTagId(t));
+ let initialTags: ReadonlySet<string> | null =
+ tagsFromUrl.length > 0 ? new Set(tagsFromUrl) : null;
if (hasShareV1(search)) {
const sel = parseShareV1(search, channelOptions);
@@ -1138,6 +1220,7 @@ function useSearchSessionState() {
initialStates = snapshotStates(snapshot);
initialDateFrom = snapshot?.dateFrom ?? "";
initialDateTo = snapshot?.dateTo ?? "";
+ initialTags ??= new Set(snapshot?.tg ?? []);
if (!resolvedRoot && snapshot?.query) {
resolvedRoot = parseRoot(snapshot.query);
}
@@ -1158,6 +1241,7 @@ function useSearchSessionState() {
setDraftStates(new Set(initialStates));
setDraftDateFrom(initialDateFrom);
setDraftDateTo(initialDateTo);
+ setDraftTags(new Set(initialTags ?? []));
setCommittedExcludedChannels(initialExcluded);
setCommittedNov(initialNov);
setCommittedNol(initialNol);
@@ -1166,6 +1250,7 @@ function useSearchSessionState() {
setCommittedStates(new Set(initialStates));
setCommittedDateFrom(initialDateFrom);
setCommittedDateTo(initialDateTo);
+ setCommittedTags(new Set(initialTags ?? []));
// Chart view + shape (after query/filters so the default data source can
// key off whether a query was restored).
@@ -1195,6 +1280,17 @@ function useSearchSessionState() {
writeViewChartParams(view, view === "chart" ? chartShape : null);
}, [hydrated, view, chartShape]);
+ // `tg` likewise: the URL carries the COMMITTED tag selection at all times, so
+ // a chip survives a reload, a share of the address bar, and a profile load —
+ // without teaching the share-v1 schema or the strip-on-commit helper about a
+ // key that is not theirs. Sorted, so the same selection is always the same
+ // URL. Deliberately post-hydration only: writing before the URL has been
+ // read would erase the selection the link arrived with.
+ useEffect(() => {
+ if (!hydrated) return;
+ writeUrlParams({ tags: committedTagsKey ? committedTagsKey.split(" ") : [] });
+ }, [hydrated, committedTagsKey]);
+
// Persist UI-only collapse state immediately on toggle. Loads + merges +
// saves so concurrent writes to other top-level fields aren't clobbered.
const persistUiCollapse = useCallback(
@@ -1296,6 +1392,7 @@ function useSearchSessionState() {
setDraftStates(snapshotStates(snapshot));
setDraftDateFrom(snapshot?.dateFrom ?? "");
setDraftDateTo(snapshot?.dateTo ?? "");
+ setDraftTags(new Set(snapshot?.tg ?? []));
if (snapshot?.query) {
const parsed = parseRoot(snapshot.query);
if (parsed) setDraftRoot(parsed);
@@ -1320,6 +1417,7 @@ function useSearchSessionState() {
setCommittedStates(snapshotStates(snapshot));
setCommittedDateFrom(snapshot?.dateFrom ?? "");
setCommittedDateTo(snapshot?.dateTo ?? "");
+ setCommittedTags(new Set(snapshot?.tg ?? []));
if (snapshot?.query) {
const parsed = parseRoot(snapshot.query);
if (parsed) setCommittedRoot(parsed);
@@ -1461,6 +1559,18 @@ function useSearchSessionState() {
setDraftExcludedChannels(excluded);
}, [channelOptions, defaultSelectedChannels]);
+ // One click on a chip: select it, or (when it is already selected) clear it.
+ // The chip row is the whole control — there is no separate "clear tags"
+ // affordance, because a selected chip IS the clear affordance.
+ const toggleDraftTag = useCallback((id: string) => {
+ setDraftTags((prev) => {
+ const next = new Set(prev);
+ if (next.has(id)) next.delete(id);
+ else next.add(id);
+ return next;
+ });
+ }, []);
+
const handleResetAllFilters = useCallback(() => {
setActiveProfileName(null);
writeStorage((s) => {
@@ -1514,6 +1624,11 @@ function useSearchSessionState() {
if (isNodeActive(committedRoot)) {
params.set("qt", stringifyRoot(committedRoot));
}
+ // `tg` rides along under its own name rather than joining the share-v1
+ // schema: buildShareSearchParams starts from an empty URLSearchParams, so
+ // without this a shared link would silently drop the tag selection it was
+ // copied with.
+ for (const id of Array.from(committedTags).sort()) params.append("tg", id);
// Carry the chart view + its shape so a shared link opens straight to the
// same chart (query/filters stay in qt= and the share params above).
if (view === "chart") {
@@ -1857,6 +1972,10 @@ function useSearchSessionState() {
setDraftDateFrom,
draftDateTo,
setDraftDateTo,
+ // ── Curated tags (draft selection + what is actually being filtered on) ──
+ draftTags,
+ toggleDraftTag,
+ committedTags,
// ── Advanced options ──
hitBatchValue,
setHitBatchValue,
diff --git a/common/components/exportFilterStorage.ts b/common/components/exportFilterStorage.ts
@@ -45,6 +45,10 @@ export type FilterSnapshot = {
dateTo?: string;
mode?: SearchMode;
tracks?: string[];
+ // Curated tag ids selected (lib/curatedTags.ts), ORed. Absent => no tag
+ // filter, which is what every profile saved before tags existed means. NOT
+ // `tracks` above and not the yt-dlp keyword scope — see urlState's `tg`.
+ tg?: string[];
query?: string;
};
@@ -151,6 +155,7 @@ function parseSnapshot(raw: unknown): FilterSnapshot | null {
}
if (r.mode === "transcripts" || r.mode === "subs") snap.mode = r.mode;
if (isStringArray(r.tracks)) snap.tracks = r.tracks.slice();
+ if (isStringArray(r.tg)) snap.tg = r.tg.slice();
if (typeof r.query === "string") snap.query = r.query;
return snap;
}
@@ -266,6 +271,7 @@ export function snapshotsEqual(a: FilterSnapshot, b: FilterSnapshot): boolean {
if ((a.dateTo ?? "") !== (b.dateTo ?? "")) return false;
if ((a.mode ?? "transcripts") !== (b.mode ?? "transcripts")) return false;
if (sortedJson(a.tracks ?? []) !== sortedJson(b.tracks ?? [])) return false;
+ if (sortedJson(a.tg ?? []) !== sortedJson(b.tg ?? [])) return false;
if ((a.query ?? "") !== (b.query ?? "")) return false;
return true;
}
diff --git a/common/components/tagsCache.ts b/common/components/tagsCache.ts
@@ -0,0 +1,45 @@
+"use client";
+
+// Client fetch for a site's shipped /tags.json — the curated per-video tag
+// vocabulary (lib/curatedTags.ts) with the per-site counts compose-site wrote.
+// Mirrors aliasesCache / summariesCache exactly, including its answer to
+// absence, because the two documents have the same shape of optionality: a
+// site publishes one only when it has something to publish.
+//
+// A 404 is the NORMAL state here in two different ways, and neither is an
+// error: compose-site writes no file at all for a site with no visible,
+// non-zero-count tag, and a site built before corpus spec 4 has never heard of
+// tags. Both resolve to an empty list, which the chip row reads as "offer no
+// chips" — the operator's rule that a site's row is built only from what that
+// site actually publishes, with no static list anywhere.
+
+import { useQuery } from "@tanstack/react-query";
+import { ArchiveHttpError } from "../lib/archive/reader";
+import { readerFor } from "../lib/archive/readers";
+import { coercePublishedTags } from "../lib/publishedTags";
+import type { PublishedTag } from "../lib/curatedTags";
+
+const EMPTY: PublishedTag[] = [];
+
+// ANY status the archive itself returned resolves empty — a 404 because the
+// site ships no tags, but a 500 or a 403 the same way. A TRANSPORT failure is
+// different and is rethrown: it is not an answer at all, and with
+// staleTime: Infinity below a swallowed blip would mean "this site has no
+// tags" for the rest of the session. Same reasoning as fetchAliases.
+export async function fetchTags(origin = ""): Promise<PublishedTag[]> {
+ try {
+ return coercePublishedTags(await readerFor(origin).readTags());
+ } catch (err) {
+ if (err instanceof ArchiveHttpError) return [];
+ throw err;
+ }
+}
+
+export function useCuratedTags(origin = ""): PublishedTag[] {
+ const { data } = useQuery<PublishedTag[]>({
+ queryKey: ["curated-tags", origin],
+ queryFn: () => fetchTags(origin),
+ staleTime: Infinity, // static per export build
+ });
+ return data ?? EMPTY;
+}
diff --git a/common/components/urlState.ts b/common/components/urlState.ts
@@ -7,6 +7,7 @@ import { useMemo, useSyncExternalStore } from "react";
// unchanged.
export type { SearchMode } from "../lib/searchQuery";
import type { SearchMode } from "../lib/searchQuery";
+import { isTagId } from "../lib/curatedTags";
// Per-video modal content mode. Independent from the search page's `mode` so
// the modal can be toggled without disturbing search state. Absence on the
@@ -39,6 +40,17 @@ export type UrlParams = {
// URL as repeated `tk=` params (e.g. ?m=subs&tk=live_chat to hide live
// chat results).
tracks: string[];
+ // Curated tag ids selected in the filter panel, as repeated `tg=` params
+ // (e.g. ?tg=eva-collab&tg=eva-in-chat). ORed: a video carrying ANY of them
+ // passes. `tg` and not `tag`, following `tk`'s two-letter convention, and
+ // deliberately nowhere near the `tags` SEARCH SCOPE, which is yt-dlp
+ // keywords — a different thing wearing the same English word (see
+ // lib/curatedTags.ts).
+ //
+ // NOT a share-v1 key: unlike ch/nov/…, `tg` IS the live representation of
+ // the committed selection, so stripAllFilterParamsFromUrl leaves it alone
+ // and it round-trips through a reload on its own.
+ tags: string[];
};
const listeners = new Set<() => void>();
@@ -92,6 +104,14 @@ function parse(search: string): UrlParams {
nu: p.get("nu") === "1",
mode,
tracks: p.getAll("tk"),
+ // Validated against TAG_ID_RE, so this entry point agrees with the MCP's
+ // parseSearchArgs: a lowercase [a-z0-9._-] id or nothing. A malformed
+ // token is DROPPED rather than carried, because the only thing a
+ // never-matching id can do downstream is turn a filter into a silent
+ // empty result. (A well-formed id this site does not publish is a
+ // different case and is kept — the chip row renders it so it can be
+ // cleared.)
+ tags: p.getAll("tg").filter((t) => isTagId(t)),
};
}
@@ -116,6 +136,7 @@ type Patch = Partial<{
nu: boolean;
mode: SearchMode;
tracks: string[];
+ tags: string[];
}>;
export function writeUrlParams(patch: Patch) {
@@ -185,6 +206,10 @@ export function writeUrlParams(patch: Patch) {
params.delete("tk");
for (const v of patch.tracks) params.append("tk", v);
}
+ if (patch.tags !== undefined) {
+ params.delete("tg");
+ for (const v of patch.tags) if (isTagId(v)) params.append("tg", v);
+ }
const qs = params.toString();
const next = `${window.location.pathname}${qs ? `?${qs}` : ""}`;
if (next === window.location.pathname + window.location.search) return;
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -66,6 +66,7 @@ import { getSettings } from "../lib/settings";
import {
listSites,
siteDigestsDir,
+ siteIndexDir,
sitePostsDir,
siteSummariesDir,
siteSubsDir,
@@ -127,6 +128,15 @@ import {
} from "../lib/digests";
import { DIGEST_FILENAME, DIGEST_OVERRIDES_FILENAME, effectiveDigest } from "../lib/digest";
import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
+import {
+ TAG_COUNTS_FILENAME,
+ applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
+ createTagCounter,
+ deriveCuratedTags,
+ loadCuratedTagsRuntime,
+ reapplyCuratedTags,
+} from "./curatedTagsIndex";
// v10: multi-site build. Shared per-channel transcript/subs pages are written
// once; per-site summaries + subs manifests are filtered selections. Bumped to
@@ -140,6 +150,18 @@ import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
// (effectiveDigest of the machine sidecar + the human override file) plus a
// shared /digests/<slug>/ page tree. Bumped so existing indexes populate the
// new sub-DB — nothing re-derives it lazily.
+// NOT a v14 — curated tags (`curatedTags` on every summary, derived from
+// transcripts/tags.json) deliberately did NOT bump this, and the reason is
+// worth keeping: every prior bump existed because something would otherwise
+// never be derived at all (v13 populates a new sub-DB; nothing re-derives it
+// lazily). Curated tags DO re-derive lazily, by design. `curatedRulesHash` is
+// absent on an index built before them, so it can never equal the current hash
+// and `reapplyCuratedTags` re-derives every record on the first build after
+// this ships — measured at ~80 s across 30k videos, straight out of LMDB, and
+// zero page writes on an untagged corpus because an empty derivation leaves
+// each summary byte-identical. A bump would instead wipe the cache and re-read
+// 76k video directories to reach exactly the same state. No sub-DB was added,
+// so the clearAsync() enumeration above is unchanged too.
const SCHEMA_VERSION = 13;
// Per-channel post stats, persisted so per-site aggregates survive a no-op
@@ -611,9 +633,18 @@ export async function buildIndex({
}
return true;
};
- const sharedNeedsBuild =
+ let sharedNeedsBuild =
anyMutations || schemaBumped || !(await sharedManifestsPresent());
+ // The curated-tag vocabulary + assignments, loaded and compiled ONCE for the
+ // whole build. Every per-video derivation below uses this; reapplyCuratedTags
+ // (after the worker loop) decides whether a tag-only edit has to re-derive
+ // records the mtime diff never looked at.
+ const curated = loadCuratedTagsRuntime(paths);
+ // indexKeyIds the worker derived fresh this build, so the re-apply pass does
+ // not do them a second time.
+ const curatedFresh = new Set<string>();
+
log(
`Diff: +${added.length} added, ~${changed.length} changed, -${removed.length} removed, ${live.length} total.`,
);
@@ -694,7 +725,6 @@ export async function buildIndex({
byChannel.remove(indexToChannelKey(prev.indexKey));
}
- sums.put(indexKey, summary);
if (cueList) cues.put(indexKey, cueList);
else cues.remove(indexKey);
@@ -730,6 +760,29 @@ export async function buildIndex({
if (parsedSubs.length > 0) subs.put(indexKey, parsedSubs);
else subs.remove(indexKey);
+ // Curated tags. Derived here, with the caption cues and the live-chat
+ // track already in hand, and stored ON the summary — which is why the
+ // sums.put waited for the subs parse. Omitted when empty, so an
+ // untagged corpus's pages stay byte-identical. Costs nothing at all
+ // when the corpus has no tags (curated.isNoop short-circuits).
+ if (!curated.isNoop) {
+ applyCuratedTagsToSummary(
+ summary,
+ deriveCuratedTags(curated, {
+ channelSlug: s.channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ captionCues: cueList,
+ chatCues: parsedSubs.find((t) => t.track === "live_chat")?.cues,
+ }),
+ );
+ }
+ curatedFresh.add(indexKeyId(indexKey));
+ sums.put(indexKey, summary);
+
// The derived layer. Only opened when the scan saw a sidecar, so the
// ~76k videos without one cost zero extra reads. What is stored is
// effectiveDigest(machine, overrides) — human corrections applied,
@@ -824,6 +877,34 @@ export async function buildIndex({
mtimes.remove(pathKey);
}
+ // Curated tags can change with no video on disk moving at all — a rule
+ // edited, a video pinned — and nothing in the mtime diff above can see that.
+ // The two meta hashes are therefore checked on EVERY build, and the videos
+ // whose tags moved are re-derived straight out of LMDB (no disk re-read).
+ // Anything that changed has to be re-paged, which is what the
+ // sharedNeedsBuild flip below buys: the page writer's sha1 skip then keeps
+ // the untouched pages untouched, so compose still copies only real changes.
+ const curatedReapply = reapplyCuratedTags({
+ runtime: curated,
+ sums,
+ cues,
+ subs,
+ byChannel,
+ meta,
+ alreadyFresh: curatedFresh,
+ allFresh: schemaBumped,
+ log,
+ });
+ // `pagesPending` also covers a PREVIOUS build that re-derived records and was
+ // then interrupted before writing the pages: its hashes are already stored,
+ // so without the flag nothing would ever ask for those shards again and they
+ // would ship stale curatedTags for ever.
+ if (curatedReapply.pagesPending) sharedNeedsBuild = true;
+ // Commit the flag (and the re-derived summaries) BEFORE the page build, so an
+ // interrupt during it finds the debt recorded rather than lost in a pending
+ // write batch.
+ await meta.flushed;
+
await sums.flushed;
await cues.flushed;
await subs.flushed;
@@ -1285,6 +1366,13 @@ export async function buildIndex({
log(`Shared sub pages up to date; ${channelStats.size} channels with subs.`);
}
+ // Both shared trees are now written (or were already current), so the
+ // curated-tag page debt is settled. Deliberately AFTER the page build and not
+ // beside the hashes: an interrupt anywhere above must leave the flag standing
+ // so the next build rewrites the shards.
+ clearCuratedPagesPending(meta);
+ await meta.flushed;
+
// ---------------------------------------------------------------------------
// The social-post corpus: a parallel per-channel page tree beside transcripts
// and subs. Social channels have no `data/` dir and so never appear in the
@@ -1680,6 +1768,12 @@ export async function buildIndex({
groups: site.groups,
defaultGroupId: site.defaultGroupId,
channels: site.channels,
+ // A tag-only edit moves no site config and no video mtime, but it does
+ // change this site's tag-counts.json (and the curatedTags on its summary
+ // pages). Without these the site would report "up to date" and ship
+ // yesterday's counts.
+ curatedRules: curated.rulesHash,
+ curatedAssign: curated.assignHash,
});
const fpKey = `siteFp:${site.siteId}`;
if (
@@ -1739,12 +1833,18 @@ export async function buildIndex({
pageIndex++;
};
+ // Per-tag totals for THIS site, accumulated in the pass that is already
+ // streaming every member record — no second walk of the corpus. compose
+ // joins these to the site's effective vocabulary to write /tags.json.
+ const tagCounter = createTagCounter();
+
for (const { key, value } of sums.getRange({ reverse: true })) {
const ik = key as IndexKey;
if (!slugSet.has(ik[1])) continue;
const s = value as TranscriptSummary;
const acc = chan.get(ik[1]);
if (acc) acc.count++;
+ tagCounter.add(ik[1], s.curatedTags);
buffer.push(
toDisplaySummary(s, { state: stateByIndexKey.get(indexKeyId(ik)) }),
);
@@ -1753,6 +1853,14 @@ export async function buildIndex({
}
await flushPage();
+ // Written even when empty (a tiny `{tags:{}}`), so a site that loses its
+ // last tagged video does not keep serving a stale count file. compose
+ // decides from this whether to publish /tags.json at all.
+ await writeJsonAtomic(
+ path.join(siteIndexDir(paths, site.siteId), TAG_COUNTS_FILENAME),
+ tagCounter.file(new Date().toISOString()),
+ );
+
const expectedPages = new Set<string>();
for (let i = 0; i < pageIndex; i++) expectedPages.add(pageFileName(i));
for (const name of await readdir(summariesOut).catch(() => [] as string[])) {
diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts
@@ -0,0 +1,287 @@
+// Integration: a curated-tag edit reaching the published files, through the
+// REAL buildIndex and the real compose script, over a temp corpus.
+//
+// The unit tests around reapplyCuratedTags prove the derivation. This proves
+// the thing that actually breaks in production: a tag edit moves no video and
+// no mtime, so every incremental short-circuit in the build has a chance to
+// decide nothing happened and ship yesterday's shards.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/curatedTagsBuild.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { execFile } from "node:child_process";
+import { mkdtempSync, writeFileSync, mkdirSync, readFileSync, existsSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { promisify } from "node:util";
+import { open } from "lmdb";
+
+// getPaths() is lazy and cached, and nothing above calls it at import time, so
+// pointing the whole path graph at a temp root here is enough to isolate this
+// file's process from the real corpus. Every path the build touches is derived
+// from these three.
+const ROOT = mkdtempSync(path.join(tmpdir(), "curated-tags-build-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public");
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { writeGlobalTags } = await import("../lib/curatedTagsStore");
+const {
+ curatedTagsRuntime,
+ reapplyCuratedTags,
+ META_PAGES_PENDING,
+} = await import("./curatedTagsIndex");
+
+const paths = getPaths();
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+const exec = promisify(execFile);
+
+// compose-site.ts is a SCRIPT (it runs main() on import), so it is exercised
+// the way the build runs it: as a child process with SITE_ID, inheriting this
+// file's temp-root env. `.bin/tsx` is a shell wrapper, so the CLI entry is
+// invoked directly.
+const composeSite = () =>
+ exec(
+ process.execPath,
+ [
+ path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"),
+ path.join(REPO, "common", "bin", "compose-site.ts"),
+ ],
+ { env: { ...process.env, SITE_ID: SITE }, cwd: REPO },
+ );
+
+const CHANNEL = "test-channel";
+const SITE = "testsite";
+const HIT = "vid-hit";
+const MISS = "vid-miss";
+
+function writeJson(file: string, value: unknown): void {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+}
+
+function videoDir(id: string): string {
+ return path.join(paths.channelsDir, CHANNEL, "data", id);
+}
+
+function seedCorpus(): void {
+ writeFileSync(paths.settingsFile, JSON.stringify({}));
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Test Channel",
+ url: "https://www.youtube.com/@example/videos",
+ });
+ const video = (id: string, title: string, uploadDate: string) => {
+ writeJson(path.join(videoDir(id), "metadata.info.json"), {
+ id,
+ title,
+ channel: "Test Channel",
+ upload_date: uploadDate,
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(
+ path.join(videoDir(id), "transcript.en.vtt"),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nA line of transcript.\n",
+ );
+ };
+ video(HIT, "Stream with Elfpire Eva", "20260102");
+ video(MISS, "An ordinary stream", "20260101");
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+}
+
+const build = () => buildIndex({ paths, onLog: () => {} });
+
+function transcriptPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSharedTranscriptsDir, CHANNEL, "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function summariesPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "summaries", "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function tagCounts(): { tags: Record<string, { count: number; channels: Record<string, number> }> } {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "tag-counts.json"),
+ "utf8",
+ ),
+ );
+}
+
+const recordIn = (page: { id: string; curatedTags?: string[] }[], id: string) =>
+ page.find((r) => r.id === id)!;
+
+const EVA = {
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [
+ {
+ id: "meta",
+ kind: "metadata" as const,
+ pattern: "elfpire",
+ enabled: true,
+ },
+ ],
+ },
+ ],
+ assignments: {},
+};
+
+// The whole slice, in one sequence, because each step's fixture is the previous
+// step's output. node:test runs top-level tests in order within a file.
+
+test("build 1: an untagged corpus ships no curatedTags key at all", async () => {
+ seedCorpus();
+ const res = await build();
+ assert.equal(res.totalCount, 2);
+ for (const page of [transcriptPage(), summariesPage()]) {
+ for (const record of page) {
+ assert.equal(
+ "curatedTags" in record,
+ false,
+ "omitted-when-empty, or every untagged corpus re-pages on upgrade",
+ );
+ }
+ }
+ assert.deepEqual(tagCounts().tags, {});
+});
+
+test("build 2: a rule edit with NO mtime change reaches the shards AND the site pages", async () => {
+ // The only thing that changes between build 1 and build 2. No video
+ // directory is touched, so the mtime diff sees +0 ~0 -0.
+ writeGlobalTags(paths, EVA);
+
+ const res = await build();
+ assert.equal(res.added, 0);
+ assert.equal(res.changed, 0);
+ assert.equal(res.removed, 0);
+
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(transcriptPage(), MISS), false);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(summariesPage(), MISS), false);
+ assert.deepEqual(tagCounts().tags["eva-collab"], {
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ });
+});
+
+test("compose 1: the site publishes /tags.json and corpus.json points at it", async () => {
+ await composeSite();
+ const tagsFile = path.join(paths.exportPublicDir, "tags.json");
+ const published = JSON.parse(readFileSync(tagsFile, "utf8"));
+ assert.deepEqual(published.tags, [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ },
+ ]);
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.spec, 4);
+ assert.equal(corpus.tags.videoField, "curatedTags");
+ assert.match(corpus.tags.url, /\/tags\.json$/);
+ assert.match(
+ readFileSync(path.join(paths.exportPublicDir, "llms.txt"), "utf8"),
+ /tags\.json/,
+ );
+});
+
+test("build 3 + compose 2: dropping the vocabulary removes the file and the pointer", async () => {
+ writeGlobalTags(paths, { version: 1, tags: [], assignments: {} });
+ await build();
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+ assert.deepEqual(tagCounts().tags, {});
+
+ await composeSite();
+ assert.equal(
+ existsSync(path.join(paths.exportPublicDir, "tags.json")),
+ false,
+ "a site with nothing to say must stop serving the file, not serve an empty one",
+ );
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.tags, undefined);
+});
+
+test("an interrupted build's page debt is paid on the next build", async () => {
+ // Reconstruct exactly what a Ctrl-C between the re-derivation and the page
+ // build leaves behind: records re-derived in LMDB, hashes recorded,
+ // curatedPagesPending set — and pages still holding the OLD content.
+ writeGlobalTags(paths, EVA);
+ const runtime = curatedTagsRuntime(EVA);
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const sums = root.openDB<unknown, [string, string, string]>({ name: "sums", encoding: "msgpack" });
+ const cues = root.openDB<unknown, [string, string, string]>({ name: "cues", encoding: "msgpack" });
+ const subs = root.openDB<unknown, [string, string, string]>({ name: "subs", encoding: "msgpack" });
+ const byChannel = root.openDB<number, [string, string, string]>({ name: "byChannel", encoding: "msgpack" });
+ const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ const res = reapplyCuratedTags({
+ runtime,
+ // The sub-DB handles are structurally what the pass needs.
+ sums: sums as never,
+ cues: cues as never,
+ subs: subs as never,
+ byChannel: byChannel as never,
+ meta: meta as never,
+ });
+ assert.equal(res.changedCount, 1, "the interrupted build did re-derive a record");
+ await sums.flushed;
+ await meta.flushed;
+ assert.equal(meta.get(META_PAGES_PENDING), true);
+ await root.close();
+
+ // The shards are stale: the interrupted build never wrote them.
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+
+ // Next build: no mtime moved AND the hashes now match, so nothing but the
+ // pending flag can save these pages.
+ await build();
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+
+ const after = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const meta2 = after.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ assert.equal(meta2.get(META_PAGES_PENDING), false, "the debt is settled");
+ await after.close();
+});
diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts
@@ -0,0 +1,462 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import type { Cue } from "../lib/vtt";
+import type { TranscriptSummary } from "../lib/transcripts";
+import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags";
+import {
+ applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
+ createTagCounter,
+ curatedTagsRuntime,
+ deriveCuratedTags,
+ hashCuratedAssignments,
+ hashCuratedRules,
+ publishedTagsFrom,
+ reapplyCuratedTags,
+ META_ASSIGN_HASH,
+ META_ASSIGN_SIGS,
+ META_PAGES_PENDING,
+ META_RULES_HASH,
+} from "./curatedTagsIndex";
+
+type IndexKey = [string, string, string];
+
+// Minimal stand-ins for the LMDB sub-DBs buildIndex hands the re-apply pass.
+// Keys are the same tuples; range scans only need to be correct, not fast.
+function makeDb<V>() {
+ const map = new Map<string, { key: unknown; value: V }>();
+ const k = (key: unknown) => JSON.stringify(key);
+ return {
+ map,
+ get(key: unknown): V | undefined {
+ return map.get(k(key))?.value;
+ },
+ put(key: unknown, value: V): void {
+ map.set(k(key), { key, value });
+ },
+ getRange(options?: { start?: unknown[] }): Iterable<{ key: unknown; value: V }> {
+ const start = options?.start as string[] | undefined;
+ const all = Array.from(map.values()).sort((a, b) =>
+ k(a.key) < k(b.key) ? -1 : 1,
+ );
+ if (!start) return all;
+ return all.filter((e) => (e.key as string[])[0] === start[0]);
+ },
+ };
+}
+
+function makeMeta() {
+ const map = new Map<string, unknown>();
+ return {
+ map,
+ get: (key: string) => map.get(key),
+ put: (key: string, value: unknown) => void map.set(key, value),
+ };
+}
+
+function summary(
+ channelSlug: string,
+ id: string,
+ uploadDate: string,
+ over: Partial<TranscriptSummary> = {},
+): TranscriptSummary {
+ return {
+ slug: `${channelSlug}/${id}`,
+ id,
+ channelSlug,
+ title: `Video ${id}`,
+ uploadDate,
+ duration: 100,
+ channel: channelSlug,
+ description: "",
+ tags: [],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: `https://example.com/${id}`,
+ ...over,
+ };
+}
+
+const COLLAB_RULE: CuratedTagDef = {
+ id: "eva-collab",
+ label: "Collab",
+ order: 1,
+ rules: [{ id: "meta", kind: "metadata", pattern: "elfpire", enabled: true }],
+};
+
+function fixture(config: CuratedTagsConfig) {
+ const sums = makeDb<TranscriptSummary>();
+ const cues = makeDb<Cue[]>();
+ const subs = makeDb<{ track: string; cues: Cue[] }[]>();
+ const byChannel = makeDb<number>();
+ const meta = makeMeta();
+ const put = (s: TranscriptSummary) => {
+ const ik: IndexKey = [s.uploadDate, s.channelSlug, s.id];
+ sums.put(ik, s);
+ byChannel.put([s.channelSlug, s.uploadDate, s.id], 1);
+ return ik;
+ };
+ return { sums, cues, subs, byChannel, meta, put, runtime: curatedTagsRuntime(config) };
+}
+
+function cfg(
+ tags: CuratedTagDef[],
+ assignments: CuratedTagsConfig["assignments"] = {},
+): CuratedTagsConfig {
+ return { version: 1, tags, assignments };
+}
+
+// ─── hashes ───
+
+test("hashCuratedRules ignores presentation, reacts to derivation", () => {
+ const base = hashCuratedRules([COLLAB_RULE]);
+ assert.equal(
+ hashCuratedRules([{ ...COLLAB_RULE, label: "On mic", color: "#fff", hidden: true }]),
+ base,
+ "relabelling must not re-page the corpus",
+ );
+ assert.notEqual(
+ hashCuratedRules([
+ { ...COLLAB_RULE, rules: [{ id: "meta", kind: "metadata", pattern: "eva", enabled: true }] },
+ ]),
+ base,
+ );
+ assert.notEqual(hashCuratedRules([{ ...COLLAB_RULE, order: 9 }]), base);
+ assert.notEqual(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [{ ...COLLAB_RULE.rules![0], channels: ["legal-mindset"] }],
+ },
+ ]),
+ base,
+ );
+ // A disabled rule is not part of the derivation at all.
+ assert.equal(
+ hashCuratedRules([
+ {
+ ...COLLAB_RULE,
+ rules: [COLLAB_RULE.rules![0], { id: "off", kind: "caption", pattern: "x", enabled: false }],
+ },
+ ]),
+ base,
+ );
+});
+
+test("hashCuratedAssignments is order-insensitive", () => {
+ const a = hashCuratedAssignments({
+ "c/v1": { manual: ["b", "a"] },
+ "c/v2": { suppressed: ["z"] },
+ });
+ const b = hashCuratedAssignments({
+ "c/v2": { suppressed: ["z"] },
+ "c/v1": { manual: ["a", "b"] },
+ });
+ assert.equal(a, b);
+ assert.notEqual(a, hashCuratedAssignments({ "c/v1": { manual: ["a"] } }));
+});
+
+// ─── the summary field ───
+
+test("applyCuratedTagsToSummary omits the key when empty", () => {
+ const s = summary("c", "v", "20260101");
+ assert.equal(applyCuratedTagsToSummary(s, []), false);
+ assert.equal("curatedTags" in s, false);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), true);
+ assert.deepEqual(s.curatedTags, ["a"]);
+ assert.equal(applyCuratedTagsToSummary(s, ["a"]), false, "idempotent");
+ assert.equal(applyCuratedTagsToSummary(s, []), true);
+ assert.equal("curatedTags" in s, false, "the key is deleted, not emptied");
+});
+
+// ─── derivation ───
+
+test("deriveCuratedTags folds rule hits with pins and suppressions", () => {
+ const runtime = curatedTagsRuntime(
+ cfg([COLLAB_RULE, { id: "manual-only", label: "Manual", order: 2 }], {
+ "lm/v2": { manual: ["manual-only"] },
+ "lm/v3": { suppressed: ["eva-collab"] },
+ }),
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v1", title: "with Elfpire" }),
+ ["eva-collab"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v2", title: "nothing" }),
+ ["manual-only"],
+ );
+ assert.deepEqual(
+ deriveCuratedTags(runtime, { channelSlug: "lm", id: "v3", title: "with Elfpire" }),
+ [],
+ "a suppression kills a rule hit",
+ );
+});
+
+// ─── reapplyCuratedTags ───
+
+test("unchanged hashes do no work at all", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.meta.put(META_RULES_HASH, f.runtime.rulesHash);
+ f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash);
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.deepEqual(res, {
+ rulesChanged: false,
+ assignmentsChanged: false,
+ changedCount: 0,
+ changed: [],
+ examined: 0,
+ pagesPending: false,
+ });
+});
+
+// THE COLD START, and the reason curated tags need no SCHEMA_VERSION bump: an
+// index built before they existed carries no hash at all, which can never equal
+// the current one, so the first build re-derives every record by itself.
+test("an index with no stored hashes re-derives every record", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.put(summary("lm", "v2", "20260102", { title: "ordinary" }));
+ assert.equal(f.meta.get(META_RULES_HASH), undefined);
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.equal(res.rulesChanged, true);
+ assert.equal(res.examined, 2, "every record is a candidate on a cold index");
+ assert.equal(res.changedCount, 1);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v1"]);
+});
+
+test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () => {
+ // The state of every existing install the day this ships: no vocabulary, no
+ // assignments. The pass still runs (the hashes have to be recorded), but
+ // every record derives [] and comes out byte-identical, so nothing is written
+ // and the caller never flips sharedNeedsBuild. This is what makes the absent
+ // SCHEMA_VERSION bump free rather than merely cheap.
+ const f = fixture(cfg([]));
+ for (let i = 0; i < 10; i++) f.put(summary("lm", `v${i}`, `2026010${i}`));
+ const before = JSON.stringify(Array.from(f.sums.map.values()));
+ const res = reapplyCuratedTags({ ...f });
+ assert.equal(res.changedCount, 0);
+ assert.equal(res.pagesPending, false, "nothing moved, so nothing to re-page");
+ assert.equal(res.examined, 10);
+ // The keys are not collected unless asked for — a 30k-record first build must
+ // not hold 30k tuples to answer a boolean.
+ assert.equal(res.changed, undefined);
+ assert.equal(JSON.stringify(Array.from(f.sums.map.values())), before);
+ for (const { value } of f.sums.map.values()) {
+ assert.equal("curatedTags" in value, false);
+ }
+ // And the hashes are now recorded, so every later build is a true no-op.
+ assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+});
+
+test("a rule change re-derives the whole corpus from LMDB and reports the movers", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" }));
+ f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" }));
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ assert.equal(res.rulesChanged, true);
+ assert.equal(res.examined, 3);
+ assert.equal(res.changedCount, 2);
+ assert.deepEqual(
+ res.changed?.map((k) => k[2]).sort(),
+ ["v1", "v3"],
+ );
+ assert.deepEqual(f.sums.get(["20260101", "lm", "v1"])!.curatedTags, ["eva-collab"]);
+ assert.equal(f.sums.get(["20260102", "lm", "v2"])!.curatedTags, undefined);
+ // The hashes are recorded, so a second pass is a no-op.
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+});
+
+test("removing the last rule strips the tag back off every record", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ reapplyCuratedTags({ ...f });
+ assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]);
+ const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) };
+ const res = reapplyCuratedTags(gone);
+ assert.equal(res.changedCount, 1);
+ assert.equal(f.sums.get(ik)!.curatedTags, undefined);
+});
+
+test("an assignment-only change touches ONLY the videos whose assignment moved", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ for (let i = 0; i < 20; i++) {
+ f.put(summary("lm", `v${i}`, `202601${String(i + 10).padStart(2, "0")}`));
+ }
+ f.put(summary("other", "w1", "20260201"));
+ reapplyCuratedTags({ ...f }); // establish the baseline hashes
+
+ const pinned = {
+ ...f,
+ runtime: curatedTagsRuntime(
+ cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }),
+ ),
+ };
+ const res = reapplyCuratedTags({ ...pinned, collectChanged: true });
+ assert.equal(res.rulesChanged, false);
+ assert.equal(res.assignmentsChanged, true);
+ // 21 videos in the corpus; exactly one was even looked at.
+ assert.equal(res.examined, 1);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v7"]);
+ assert.deepEqual(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, ["eva-collab"]);
+
+ // Clearing it again is also a one-video pass, and the key comes back off.
+ const cleared = { ...f, runtime: curatedTagsRuntime(cfg([COLLAB_RULE])) };
+ const res2 = reapplyCuratedTags(cleared);
+ assert.equal(res2.examined, 1);
+ assert.equal(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, undefined);
+});
+
+test("videos the worker already derived this build are not done twice", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({
+ ...f,
+ alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]),
+ collectChanged: true,
+ });
+ assert.equal(res.examined, 0);
+ assert.deepEqual(res.changed, []);
+});
+
+test("a schema bump records the hashes and re-applies nothing", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const res = reapplyCuratedTags({ ...f, allFresh: true });
+ assert.equal(res.changedCount, 0);
+ assert.equal(res.examined, 0);
+ // The pages ARE rebuilt after a wipe, so the debt is recorded all the same —
+ // an interrupt mid-rebuild must not be forgotten.
+ assert.equal(res.pagesPending, true);
+ assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
+ assert.deepEqual(f.meta.get(META_ASSIGN_SIGS), {});
+});
+
+// ─── the interrupted-build flag ───
+
+test("a change sets curatedPagesPending, and only clearing it settles the debt", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const first = reapplyCuratedTags({ ...f });
+ assert.equal(first.changedCount, 1);
+ assert.equal(first.pagesPending, true);
+ assert.equal(f.meta.get(META_PAGES_PENDING), true);
+
+ // THE INTERRUPT: the hashes are stored but the page build never ran, so the
+ // flag was never cleared. The next build finds the hashes equal — it
+ // re-derives nothing — and must STILL report that the pages owe a rewrite.
+ const afterCrash = reapplyCuratedTags({ ...f });
+ assert.equal(afterCrash.rulesChanged, false);
+ assert.equal(afterCrash.changedCount, 0);
+ assert.equal(
+ afterCrash.pagesPending,
+ true,
+ "hashes match but the shards are still stale",
+ );
+
+ // The page build completed this time.
+ clearCuratedPagesPending(f.meta);
+ assert.equal(f.meta.get(META_PAGES_PENDING), false);
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+});
+
+test("no change means no debt — an untagged corpus never sets the flag", () => {
+ const f = fixture(cfg([]));
+ f.put(summary("lm", "v1", "20260101"));
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+ assert.notEqual(f.meta.get(META_PAGES_PENDING), true);
+});
+
+test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => {
+ const capRule: CuratedTagDef = {
+ id: "eva-topic",
+ label: "Discussed",
+ rules: [
+ {
+ id: "cap",
+ kind: "caption",
+ pattern: "elfpire",
+ channels: ["lm"],
+ enabled: true,
+ },
+ ],
+ };
+ const chatRule: CuratedTagDef = {
+ id: "eva-in-chat",
+ label: "In chat",
+ rules: [{ id: "chat", kind: "chat-author", pattern: "elfpire", enabled: true }],
+ };
+ const f = fixture(cfg([capRule, chatRule]));
+ const a = f.put(summary("lm", "v1", "20260101"));
+ const b = f.put(summary("other", "v2", "20260102"));
+ f.cues.put(a, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.cues.put(b, [{ start: 0, end: 1, text: "we talked about Elfpire" }]);
+ f.subs.put(a, [
+ { track: "en", cues: [{ start: 0, end: 1, text: "ElfpireEva: not chat" }] },
+ { track: "live_chat", cues: [{ start: 0, end: 1, text: "ElfpireEva: hi" }] },
+ ]);
+
+ reapplyCuratedTags({ ...f });
+ // Order follows the vocabulary (neither def sets `order`, so it is the
+ // config's own order), not the alphabet.
+ assert.deepEqual(f.sums.get(a)!.curatedTags, ["eva-topic", "eva-in-chat"]);
+ // The caption rule is scoped to `lm`, so the other channel gets nothing even
+ // though its cues would match.
+ assert.equal(f.sums.get(b)!.curatedTags, undefined);
+});
+
+// ─── counts + publication ───
+
+test("createTagCounter accumulates per tag and per channel", () => {
+ const c = createTagCounter();
+ c.add("lm", ["eva-collab", "eva-topic"]);
+ c.add("lm", ["eva-collab"]);
+ c.add("nux", ["eva-collab"]);
+ c.add("nux", undefined);
+ c.add("nux", []);
+ const file = c.file("2026-09-21T00:00:00Z");
+ assert.equal(file.version, 1);
+ assert.deepEqual(file.tags["eva-collab"], {
+ count: 3,
+ channels: { lm: 2, nux: 1 },
+ });
+ assert.deepEqual(file.tags["eva-topic"], { count: 1, channels: { lm: 1 } });
+});
+
+test("publishedTagsFrom drops hidden and zero-count tags and sorts by order", () => {
+ const defs: CuratedTagDef[] = [
+ { id: "b-tag", label: "B", order: 2, group: "eva", groupLabel: "Eva" },
+ { id: "a-tag", label: "A", order: 1, color: "#fff" },
+ { id: "hidden-tag", label: "Hidden", order: 0, hidden: true },
+ { id: "unused", label: "Unused", order: 3 },
+ ];
+ const counts = {
+ version: 1,
+ generatedAt: "",
+ tags: {
+ "a-tag": { count: 2, channels: { lm: 2 } },
+ "b-tag": { count: 1, channels: { nux: 1 } },
+ "hidden-tag": { count: 5, channels: { lm: 5 } },
+ },
+ };
+ const published = publishedTagsFrom(defs, counts);
+ assert.deepEqual(
+ published.tags.map((t) => t.id),
+ ["a-tag", "b-tag"],
+ );
+ assert.deepEqual(published.tags[0], {
+ id: "a-tag",
+ label: "A",
+ color: "#fff",
+ order: 1,
+ count: 2,
+ channels: { lm: 2 },
+ });
+ assert.equal(published.tags[1].groupLabel, "Eva");
+ // No counts at all → nothing to publish (compose then writes no file).
+ assert.deepEqual(publishedTagsFrom(defs, null).tags, []);
+});
diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts
@@ -0,0 +1,510 @@
+// Curated tags, as the index build sees them.
+//
+// A video's `curatedTags` is derived, never stored on disk beside the video:
+// it is (rule hits ∪ operator pins) − suppressions, folded fresh at every
+// build. Rule hits in particular are NOT persisted anywhere — edit a rule,
+// rebuild, done — which is why this module exists: nothing in the per-video
+// mtime diff can notice that a rule changed, so the build has to ask.
+//
+// Two hashes in the existing `meta` sub-DB answer that question:
+// curatedRulesHash the derivation-relevant shape of the vocabulary — every
+// def's id and order, and every ENABLED rule's kind,
+// pattern, channel scope and date range
+// curatedAssignHash every pin and suppression
+// Both are checked unconditionally on every build (a skipped check would make
+// a rule edit silently inert until some unrelated mtime moved), and alongside
+// them `curatedAssignSigs` keeps a per-video signature so an assignment-only
+// change can re-derive just the videos whose assignment actually moved instead
+// of the whole corpus.
+//
+// **Only the CORPUS vocabulary is evaluated here.** A site-only rule
+// (sites/<id>/tags.json) contributes its definition to that site's published
+// /tags.json, but it does not tag records: records are shared by every site
+// that carries the channel, so a per-site rule hit would have to live in a
+// per-site record. Promote a rule to transcripts/tags.json to make it bind.
+
+import { createHash } from "node:crypto";
+import type { Cue } from "../lib/vtt";
+import type { Paths } from "../lib/paths";
+import type { TranscriptSummary } from "../lib/transcripts";
+import {
+ compileTagRules,
+ effectiveTagsFor,
+ evaluateCompiledRules,
+ ruleApplies,
+ type CompiledTagRules,
+ type CuratedTagAssignment,
+ type CuratedTagDef,
+ type CuratedTagsConfig,
+ type PublishedTags,
+ type TagRuleInput,
+} from "../lib/curatedTags";
+import { readGlobalTags } from "../lib/curatedTagsStore";
+
+export const META_RULES_HASH = "curatedRulesHash";
+export const META_ASSIGN_HASH = "curatedAssignHash";
+export const META_ASSIGN_SIGS = "curatedAssignSigs";
+// "Records were re-derived; the shared pages do not reflect them yet." Set when
+// the hashes are recorded with changes, cleared only once the shared page build
+// has finished. Without it, a Ctrl-C between the two would leave the hashes
+// stored and the shards stale for ever: the next build would compare equal,
+// re-derive nothing, and never dirty the pages.
+export const META_PAGES_PENDING = "curatedPagesPending";
+
+export const TAG_COUNTS_FILENAME = "tag-counts.json";
+export const TAG_COUNTS_VERSION = 1;
+
+type IndexKey = [string, string, string];
+type ChannelKey = [string, string, string];
+
+function sha1(s: string): string {
+ return createHash("sha1").update(s).digest("hex");
+}
+
+// The shape of the vocabulary that can change a record. Presentation fields
+// (label, colour, group, hidden) are deliberately absent: relabelling a tag
+// must not re-page 30,000 videos.
+export function hashCuratedRules(defs: CuratedTagDef[]): string {
+ const shape = defs.map((def, i) => [
+ def.id,
+ def.order ?? i,
+ (def.rules ?? [])
+ .filter((r) => r.enabled !== false)
+ .map((r) => [
+ r.id,
+ r.kind,
+ r.pattern,
+ [...(r.channels ?? [])].sort(),
+ r.dateFrom ?? null,
+ r.dateTo ?? null,
+ ]),
+ ]);
+ return sha1(JSON.stringify(shape));
+}
+
+// One video's assignment, canonically. Also the per-key signature stored in
+// meta, so the diff is a string compare.
+export function assignmentSignature(a: CuratedTagAssignment): string {
+ return `${[...(a.manual ?? [])].sort().join(",")}|${[...(a.suppressed ?? [])].sort().join(",")}`;
+}
+
+export function assignmentSignatures(
+ assignments: Record<string, CuratedTagAssignment>,
+): Record<string, string> {
+ const out: Record<string, string> = {};
+ for (const key of Object.keys(assignments).sort()) {
+ out[key] = assignmentSignature(assignments[key]);
+ }
+ return out;
+}
+
+export function hashCuratedAssignments(
+ assignments: Record<string, CuratedTagAssignment>,
+): string {
+ return sha1(JSON.stringify(assignmentSignatures(assignments)));
+}
+
+// Everything the build needs, loaded and compiled ONCE per build.
+export type CuratedTagsRuntime = {
+ config: CuratedTagsConfig;
+ defs: CuratedTagDef[];
+ compiled: CompiledTagRules;
+ rulesHash: string;
+ assignHash: string;
+ sigs: Record<string, string>;
+ // True when the corpus has no rules and no assignments at all — the state of
+ // every untagged install, and the cue to do no work whatsoever.
+ isNoop: boolean;
+};
+
+export function loadCuratedTagsRuntime(paths: Paths): CuratedTagsRuntime {
+ return curatedTagsRuntime(readGlobalTags(paths));
+}
+
+export function curatedTagsRuntime(config: CuratedTagsConfig): CuratedTagsRuntime {
+ const compiled = compileTagRules(config.tags);
+ return {
+ config,
+ defs: config.tags,
+ compiled,
+ rulesHash: hashCuratedRules(config.tags),
+ assignHash: hashCuratedAssignments(config.assignments),
+ sigs: assignmentSignatures(config.assignments),
+ isNoop:
+ compiled.isEmpty && Object.keys(config.assignments).length === 0,
+ };
+}
+
+// Cheap pre-checks, so a video on a channel no rule is scoped to never pays for
+// a cue read. They ask `ruleApplies` — the SAME predicate
+// evaluateCompiledRules uses — so this can never answer "no cues needed" for a
+// rule that would then have fired.
+export function needsCaptionCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.caption.some((r) =>
+ ruleApplies(r, { channelSlug, uploadDate }),
+ );
+}
+
+export function needsChatCues(
+ runtime: CuratedTagsRuntime,
+ channelSlug: string,
+ uploadDate?: string,
+): boolean {
+ return runtime.compiled.chatAuthor.some((r) =>
+ ruleApplies(r, { channelSlug, uploadDate }),
+ );
+}
+
+// The whole fold for one video: rule hits, then the operator's pins and
+// suppressions over the top.
+export function deriveCuratedTags(
+ runtime: CuratedTagsRuntime,
+ input: TagRuleInput,
+): string[] {
+ const hits = evaluateCompiledRules(input, runtime.compiled);
+ const assignment = runtime.config.assignments[`${input.channelSlug}/${input.id}`];
+ if (hits.length === 0 && !assignment) return [];
+ return effectiveTagsFor(hits, assignment, runtime.defs);
+}
+
+// Set/clear `curatedTags` on a summary in place. Returns true when the stored
+// value actually changed — omitted-when-empty is the contract, so an untagged
+// record must come out byte-identical to the one already on disk.
+export function applyCuratedTagsToSummary(
+ summary: TranscriptSummary,
+ tags: string[],
+): boolean {
+ const before = summary.curatedTags;
+ if (tags.length === 0) {
+ if (before === undefined) return false;
+ delete summary.curatedTags;
+ return true;
+ }
+ if (
+ before &&
+ before.length === tags.length &&
+ before.every((t, i) => t === tags[i])
+ ) {
+ return false;
+ }
+ summary.curatedTags = tags;
+ return true;
+}
+
+// ─── re-derivation ───
+
+type SumsDb = {
+ get(key: IndexKey): TranscriptSummary | undefined;
+ put(key: IndexKey, value: TranscriptSummary): unknown;
+ getRange(options?: unknown): Iterable<{ key: unknown; value: unknown }>;
+};
+type CuesDb = { get(key: IndexKey): Cue[] | undefined };
+type SubsDb = { get(key: IndexKey): { track: string; cues: Cue[] }[] | undefined };
+type ByChannelDb = {
+ getRange(options: unknown): Iterable<{ key: unknown }>;
+};
+type MetaDb = {
+ get(key: string): unknown;
+ put(key: string, value: unknown): unknown;
+};
+
+export type ReapplyOptions = {
+ runtime: CuratedTagsRuntime;
+ sums: SumsDb;
+ cues: CuesDb;
+ subs: SubsDb;
+ byChannel: ByChannelDb;
+ meta: MetaDb;
+ // indexKeyIds the per-video worker already derived this build. Skipped here
+ // so a rule edit landing in the same build as new videos doesn't do them
+ // twice.
+ alreadyFresh?: Set<string>;
+ // A schema bump cleared the cache and the worker re-derived everything, so
+ // there is nothing to re-apply — just record the hashes.
+ allFresh?: boolean;
+ // Return the changed index keys as well as the count. Off by default: the
+ // build only needs the boolean, and a first build over 30k records would
+ // otherwise hold 30k three-element tuples for no reason. Tests and any future
+ // per-channel dirtying ask for them.
+ collectChanged?: boolean;
+ log?: (msg: string) => void;
+};
+
+export type ReapplyResult = {
+ rulesChanged: boolean;
+ assignmentsChanged: boolean;
+ // How many videos' stored curatedTags moved. Non-zero means the shared pages
+ // must be rewritten.
+ changedCount: number;
+ // Which ones — only when `collectChanged` was asked for.
+ changed?: IndexKey[];
+ // Videos examined (the cost).
+ examined: number;
+ // True when the shared pages still owe a rewrite for curated tags: either
+ // this call changed records, or an earlier build did and was interrupted
+ // before its pages were written. The caller MUST OR this into whatever
+ // decides to rebuild the shared trees, and clear it (clearCuratedPagesPending)
+ // only once that build has finished.
+ pagesPending: boolean;
+};
+
+function indexKeyId(k: IndexKey): string {
+ return `${k[0]}\x00${k[1]}\x00${k[2]}`;
+}
+
+function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[] | undefined {
+ if (!stored) return undefined;
+ for (const t of stored) if (t.track === "live_chat") return t.cues;
+ return undefined;
+}
+
+// Re-derive `curatedTags` for the videos a tag change can reach, WITHOUT
+// re-reading a single video directory: cues and live chat come back out of the
+// LMDB sub-DBs the build already populated.
+//
+// rules changed every video is a candidate (a new pattern can match
+// anything), but cues are read only for the videos a
+// caption/chat rule is actually scoped to
+// assignments changed only the videos whose assignment signature moved —
+// found by a key-only range scan of `byChannel` per
+// affected channel, which is the cheap idiom
+//
+// Returns how many videos changed (and which, on request) so the caller can
+// force those pages to be rewritten: nothing in the mtime diff knows this
+// happened. It also sets `curatedPagesPending` — see the flag's own note.
+export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
+ const { runtime, sums, cues, subs, byChannel, meta } = opts;
+ const log = opts.log ?? (() => {});
+ const prevRules = meta.get(META_RULES_HASH) as string | undefined;
+ const prevAssign = meta.get(META_ASSIGN_HASH) as string | undefined;
+ const prevSigs = (meta.get(META_ASSIGN_SIGS) as Record<string, string> | undefined) ?? {};
+ const rulesChanged = prevRules !== runtime.rulesHash;
+ const assignmentsChanged = prevAssign !== runtime.assignHash;
+ // A previous build re-derived records and was then interrupted before it
+ // finished writing the shared pages. The hashes are already stored, so this
+ // build would otherwise see "nothing changed" and ship stale shards forever.
+ const pendingBefore = meta.get(META_PAGES_PENDING) === true;
+
+ const record = (pages: boolean) => {
+ meta.put(META_RULES_HASH, runtime.rulesHash);
+ meta.put(META_ASSIGN_HASH, runtime.assignHash);
+ meta.put(META_ASSIGN_SIGS, runtime.sigs);
+ // Set BEFORE the pages are written and cleared only after they are; never
+ // un-set here, or an interrupted build's debt would be forgotten.
+ if (pages) meta.put(META_PAGES_PENDING, true);
+ };
+ const result = (over: Partial<ReapplyResult> & { changedCount: number }): ReapplyResult => ({
+ rulesChanged,
+ assignmentsChanged,
+ examined: 0,
+ pagesPending: pendingBefore || over.changedCount > 0,
+ ...over,
+ });
+
+ if (!rulesChanged && !assignmentsChanged) {
+ return result({ changedCount: 0, ...(opts.collectChanged ? { changed: [] } : {}) });
+ }
+ if (opts.allFresh) {
+ // A schema bump: the worker re-derived every record, and the pages are
+ // being rebuilt anyway — but the debt is recorded all the same, so an
+ // interrupt mid-rebuild is not forgotten either.
+ record(true);
+ log(
+ `curated tags: rules ${runtime.rulesHash.slice(0, 8)}, assignments ${runtime.assignHash.slice(0, 8)} (full rebuild, nothing to re-apply)`,
+ );
+ return {
+ rulesChanged,
+ assignmentsChanged,
+ changedCount: 0,
+ ...(opts.collectChanged ? { changed: [] } : {}),
+ examined: 0,
+ pagesPending: true,
+ };
+ }
+
+ const changed: IndexKey[] = [];
+ let changedCount = 0;
+ let examined = 0;
+ const skip = opts.alreadyFresh;
+
+ // `known` is the value the cursor already decoded, when there is one — a
+ // second sums.get() per video would double the msgpack decodes on the
+ // whole-corpus path.
+ const rederive = (indexKey: IndexKey, known?: TranscriptSummary): void => {
+ if (skip?.has(indexKeyId(indexKey))) return;
+ const summary = known ?? sums.get(indexKey);
+ if (!summary) return;
+ examined++;
+ const channelSlug = indexKey[1];
+ const tags = deriveCuratedTags(runtime, {
+ channelSlug,
+ id: summary.id,
+ title: summary.title,
+ description: summary.description,
+ tags: summary.tags,
+ uploadDate: summary.uploadDate,
+ // Only read what a scoped rule can actually use. On an untagged corpus,
+ // and on every channel no rule names, this reads nothing.
+ captionCues: needsCaptionCues(runtime, channelSlug, summary.uploadDate)
+ ? cues.get(indexKey)
+ : undefined,
+ chatCues: needsChatCues(runtime, channelSlug, summary.uploadDate)
+ ? chatCuesOf(subs.get(indexKey))
+ : undefined,
+ });
+ if (applyCuratedTagsToSummary(summary, tags)) {
+ sums.put(indexKey, summary);
+ changedCount++;
+ if (opts.collectChanged) changed.push(indexKey);
+ }
+ };
+
+ if (rulesChanged) {
+ // Every video is a candidate. This walks `sums` keys only; the value comes
+ // from the same cursor, and cue reads are gated per video above.
+ for (const { key, value } of sums.getRange()) {
+ rederive(key as IndexKey, value as TranscriptSummary);
+ }
+ } else {
+ // Assignment-only change: the symmetric difference of the signatures.
+ const movedKeys = new Set<string>();
+ for (const [key, sig] of Object.entries(runtime.sigs)) {
+ if (prevSigs[key] !== sig) movedKeys.add(key);
+ }
+ for (const key of Object.keys(prevSigs)) {
+ if (!(key in runtime.sigs)) movedKeys.add(key);
+ }
+ // Group by channel so each channel costs one key-only range scan.
+ const byChannelSlug = new Map<string, Set<string>>();
+ for (const key of movedKeys) {
+ const at = key.indexOf("/");
+ if (at <= 0) continue;
+ const slug = key.slice(0, at);
+ const id = key.slice(at + 1);
+ const set = byChannelSlug.get(slug) ?? new Set<string>();
+ set.add(id);
+ byChannelSlug.set(slug, set);
+ }
+ for (const [slug, ids] of byChannelSlug) {
+ for (const { key } of byChannel.getRange({
+ start: [slug],
+ end: [slug, ""],
+ values: false,
+ })) {
+ const ck = key as ChannelKey;
+ if (ck[0] !== slug) continue;
+ if (!ids.has(ck[2])) continue;
+ rederive([ck[1], ck[0], ck[2]]);
+ }
+ }
+ }
+
+ record(changedCount > 0);
+ const what = [
+ rulesChanged ? `rules ${runtime.rulesHash.slice(0, 8)} (changed)` : null,
+ assignmentsChanged
+ ? `assignments ${runtime.assignHash.slice(0, 8)} (changed)`
+ : null,
+ ]
+ .filter(Boolean)
+ .join(", ");
+ log(
+ `curated tags: ${what}, examined ${examined}, re-derived ${changedCount}`,
+ );
+ return result({
+ changedCount,
+ ...(opts.collectChanged ? { changed } : {}),
+ examined,
+ });
+}
+
+// Clear the "the shared pages do not yet reflect the current tags" debt. Call
+// ONLY after the shared transcript/subs page build has completed: until then an
+// interrupt must leave the flag standing, which is the whole point of it.
+export function clearCuratedPagesPending(meta: MetaDb): void {
+ meta.put(META_PAGES_PENDING, false);
+}
+
+// ─── per-site counts ───
+
+export type TagCountsFile = {
+ version: number;
+ generatedAt: string;
+ // tag id -> total on this site + the per-channel breakdown.
+ tags: Record<string, { count: number; channels: Record<string, number> }>;
+};
+
+export function emptyTagCounts(): TagCountsFile {
+ return { version: TAG_COUNTS_VERSION, generatedAt: "", tags: {} };
+}
+
+// Accumulator for the per-site summaries stream: one add() per record, no
+// second pass over the corpus.
+export function createTagCounter() {
+ const tags = new Map<string, { count: number; channels: Map<string, number> }>();
+ return {
+ add(channelSlug: string, curatedTags: string[] | undefined): void {
+ if (!curatedTags || curatedTags.length === 0) return;
+ for (const id of curatedTags) {
+ let entry = tags.get(id);
+ if (!entry) {
+ entry = { count: 0, channels: new Map() };
+ tags.set(id, entry);
+ }
+ entry.count++;
+ entry.channels.set(channelSlug, (entry.channels.get(channelSlug) ?? 0) + 1);
+ }
+ },
+ get size(): number {
+ return tags.size;
+ },
+ file(generatedAt: string): TagCountsFile {
+ const out: TagCountsFile["tags"] = {};
+ for (const id of Array.from(tags.keys()).sort()) {
+ const entry = tags.get(id)!;
+ const channels: Record<string, number> = {};
+ for (const slug of Array.from(entry.channels.keys()).sort()) {
+ channels[slug] = entry.channels.get(slug)!;
+ }
+ out[id] = { count: entry.count, channels };
+ }
+ return { version: TAG_COUNTS_VERSION, generatedAt, tags: out };
+ },
+ };
+}
+
+// The published /tags.json for one site: its effective vocabulary joined to its
+// counts. Hidden tags and tags with no video ON THIS SITE are dropped — a
+// global tag that matched nothing here does not get a chip here — and a site
+// with nothing left ships no file at all (compose skips the write; a 404 is a
+// legitimate empty state).
+export function publishedTagsFrom(
+ defs: CuratedTagDef[],
+ counts: TagCountsFile | null,
+): PublishedTags {
+ const tags = defs
+ .filter((def) => def.hidden !== true)
+ .map((def, i) => ({ def, i, entry: counts?.tags[def.id] }))
+ .filter((x) => (x.entry?.count ?? 0) > 0)
+ .sort((a, b) => {
+ const oa = a.def.order ?? a.i;
+ const ob = b.def.order ?? b.i;
+ if (oa !== ob) return oa - ob;
+ return a.def.id < b.def.id ? -1 : a.def.id > b.def.id ? 1 : 0;
+ })
+ .map(({ def, entry }) => ({
+ id: def.id,
+ label: def.label,
+ ...(def.group ? { group: def.group } : {}),
+ ...(def.groupLabel ? { groupLabel: def.groupLabel } : {}),
+ ...(def.color ? { color: def.color } : {}),
+ ...(def.order !== undefined ? { order: def.order } : {}),
+ count: entry!.count,
+ channels: entry!.channels,
+ }));
+ return { version: 1, tags };
+}
diff --git a/common/lib/archive/contract.test.ts b/common/lib/archive/contract.test.ts
@@ -77,12 +77,15 @@ test("per-channel trees take a slug, flat trees do not", () => {
assert.throws(() => pageUrl("posts", undefined, 1), /needs a channel slug/);
});
-test("root files: the four JSON documents a reader fetches by name", () => {
+test("root files: the five JSON documents a reader fetches by name", () => {
assert.deepEqual([...ROOT_FILES], [
"corpus.json",
"site.json",
"search-aliases.json",
"duplicates.json",
+ // Curated tags. Legitimately absent (like duplicates.json) — but unlike it,
+ // DECLARED in corpus.json when present, which is what corpusSpec 4 says.
+ "tags.json",
]);
assert.equal(corpusUrl(), "/corpus.json");
assert.equal(corpusUrl("https://x.example"), "https://x.example/corpus.json");
@@ -161,7 +164,7 @@ test("shipsPwa: the site flag, or hub mode", () => {
test("CONTRACT is frozen where it is published", () => {
// These are on the wire. A change here is a change to every deployed archive's
// machine contract, so it belongs in a slice that says so, not in a refactor.
- assert.equal(CONTRACT.corpusSpec, 3);
+ assert.equal(CONTRACT.corpusSpec, 4);
assert.equal(CONTRACT.siteDescriptor, 1);
assert.equal(CONTRACT.manifest, 3);
assert.equal(CONTRACT.transcriptsManifest, 1);
diff --git a/common/lib/archive/contract.ts b/common/lib/archive/contract.ts
@@ -24,6 +24,7 @@
// dependency graph is a DAG.
import { DUPLICATES_FILENAME } from "../duplicates";
+import { TAGS_FILENAME } from "../curatedTags";
// The machine-readable versions and constants of the published contract. Every
// value here appears on the wire, so a change is a wire change — see the
@@ -31,7 +32,7 @@ import { DUPLICATES_FILENAME } from "../duplicates";
export const CONTRACT = {
// /corpus.json's own `spec`. `generator` is deliberately unversioned; see the
// note in lib/corpus.ts for why a credit line did not bump the spec.
- corpusSpec: 3,
+ corpusSpec: 4,
// /site.json's `contract` (siteDescriptor.ts).
siteDescriptor: 1,
// The four manifest versions (manifest.ts). Each is the version field of one
@@ -118,11 +119,20 @@ export function archiveUrl(base: string | undefined, p: string): string {
// duplicates.json is the one that is legitimately absent: compose-site only
// writes it when there is at least one publishable cluster, and corpus.json
// does not declare it, so a 404 here is data, not an error.
+//
+// tags.json is absent the same way — the curated vocabulary, with per-site
+// counts, written only when this site has at least one visible tag with a
+// non-zero count. Unlike duplicates.json it IS declared in corpus.json (as
+// `tags`) when present, which is what the spec-4 bump announces: a reader that
+// does not know about it misses a document it could have fetched. An archive
+// built before spec 4 simply 404s here, and a tag filter over it must say so
+// rather than silently return nothing.
export const ROOT_FILES = [
"corpus.json",
"site.json",
"search-aliases.json",
DUPLICATES_FILENAME,
+ TAGS_FILENAME,
] as const;
export type RootFile = (typeof ROOT_FILES)[number];
diff --git a/common/lib/archive/headers.test.ts b/common/lib/archive/headers.test.ts
@@ -18,6 +18,8 @@ const SITE_HEADERS = `# Generated by compose-site.ts — do not edit by hand.
Access-Control-Allow-Origin: *
/duplicates.json
Access-Control-Allow-Origin: *
+/tags.json
+ Access-Control-Allow-Origin: *
/summaries/*
Access-Control-Allow-Origin: *
/subs/*
diff --git a/common/lib/archive/headers.ts b/common/lib/archive/headers.ts
@@ -37,6 +37,7 @@ export const SITE_CORS_PATHS: readonly string[] = [
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
"/summaries/*",
"/subs/*",
"/transcripts/*",
diff --git a/common/lib/archive/offlineUrls.test.ts b/common/lib/archive/offlineUrls.test.ts
@@ -95,6 +95,7 @@ test("the site list is the FLAT trees plus the root files, and nothing per-chann
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
const flat = ARCHIVE_TREES.filter(isFlatTree);
@@ -146,6 +147,7 @@ test("a tree the site does not ship costs one probe and contributes nothing", as
"/site.json",
"/search-aliases.json",
"/duplicates.json",
+ "/tags.json",
]);
});
@@ -174,5 +176,6 @@ test("a federated origin prefixes both lists and nothing else changes", async ()
`${O}/site.json`,
`${O}/search-aliases.json`,
`${O}/duplicates.json`,
+ `${O}/tags.json`,
]);
});
diff --git a/common/lib/archive/reader-fs.ts b/common/lib/archive/reader-fs.ts
@@ -23,6 +23,8 @@ import type { TranscriptDetail, DisplaySummary } from "../transcripts";
import type { SubsDetail } from "../subs";
import type { ChannelPostsManifest, Post } from "../posts";
import { coerceAliasConfig, type SearchAlias } from "../searchAliases";
+import { TAGS_FILENAME, type PublishedTag } from "../curatedTags";
+import { coercePublishedTags } from "../publishedTags";
import { DUPLICATES_FILENAME, type DuplicateReport } from "../duplicates";
import type { ChannelDigestsManifest, VideoDigest } from "../digests";
import type { StatsManifest, VideoStat } from "../stats";
@@ -74,6 +76,7 @@ const PREFER_PLATFORM_LINKS = process.env.TRANSCRIPT_PLATFORM_LINKS === "1";
export class LocalSource implements ArchiveReader {
readonly label: string;
private aliases?: SearchAlias[];
+ private tags?: PublishedTag[];
private groups?: ChannelGroups;
private index?: Promise<Map<string, IndexedVideo>>;
private duplicates?: Promise<DuplicateIndex>;
@@ -179,6 +182,20 @@ export class LocalSource implements ArchiveReader {
return this.aliases;
}
+ // A composed public dir either carries /tags.json or it does not — the same
+ // 404-is-data rule the remote reader follows, just as ENOENT.
+ async loadTags(): Promise<PublishedTag[]> {
+ if (this.tags) return this.tags;
+ try {
+ this.tags = coercePublishedTags(
+ await readLocalJson(path.join(this.dir, TAGS_FILENAME), "tags"),
+ );
+ } catch {
+ this.tags = []; // no/invalid file — this build publishes no tags
+ }
+ return this.tags;
+ }
+
async loadGroups(): Promise<ChannelGroups> {
if (this.groups) return this.groups;
try {
diff --git a/common/lib/archive/reader-hub.ts b/common/lib/archive/reader-hub.ts
@@ -9,6 +9,8 @@ import type { TranscriptDetail } from "../transcripts";
import type { SubsDetail } from "../subs";
import type { ChannelPostsManifest, Post } from "../posts";
import { coerceAliasConfig, type SearchAlias } from "../searchAliases";
+import { TAGS_FILENAME, type PublishedTag } from "../curatedTags";
+import { coercePublishedTags } from "../publishedTags";
import type { ChannelDigestsManifest, VideoDigest } from "../digests";
import type { VideoStat } from "../stats";
import { corpusUrl, rootFileUrl, type HubSite } from "./contract";
@@ -31,6 +33,7 @@ export class HubSource implements ArchiveReader {
readonly hubBase: string;
private members = new Map<string, RemoteSource>(); // siteId -> source
private aliases?: SearchAlias[];
+ private tags?: PublishedTag[];
private index?: Promise<Map<string, IndexedVideo>>;
private duplicates?: Promise<DuplicateIndex>;
private stats?: Promise<ReadonlyMap<string, VideoStat>>;
@@ -104,6 +107,22 @@ export class HubSource implements ArchiveReader {
return this.aliases;
}
+ // A hub can ship its own /tags.json — the federation-wide vocabulary with
+ // counts summed across members. If it doesn't, tags are simply off for
+ // hub-wide search: merging member documents here would produce counts that
+ // no single site can reproduce, and a chip whose number nobody can check is
+ // worse than no chip.
+ async loadTags(): Promise<PublishedTag[]> {
+ if (this.tags) return this.tags;
+ try {
+ const res = await fetch(rootFileUrl(TAGS_FILENAME, this.hubBase));
+ this.tags = res.ok ? coercePublishedTags(await res.json()) : [];
+ } catch {
+ this.tags = [];
+ }
+ return this.tags;
+ }
+
// In hub mode each group is itself a federated member site (a different model
// — accent-per-origin), so hub-wide channel-group tokens are deferred: return
// the empty fallback. Multi-channel scoping still works (member groupIds carry
diff --git a/common/lib/archive/reader.ts b/common/lib/archive/reader.ts
@@ -27,6 +27,8 @@ import { summaryState, type VideoState } from "../availability";
import type { SubsDetail } from "../subs";
import type { ChannelPostsManifest, Post, PostsManifest } from "../posts";
import { coerceAliasConfig, type SearchAlias } from "../searchAliases";
+import { TAGS_FILENAME, type PublishedTag } from "../curatedTags";
+import { coercePublishedTags } from "../publishedTags";
import {
resolveCanonicalSlug,
DUPLICATES_FILENAME,
@@ -105,6 +107,13 @@ export type IndexedVideo = {
uploadDate: string;
isLivestream: boolean;
ageRestricted: boolean;
+ // Curated tag ids (lib/curatedTags.ts), when the record carries any. Present
+ // here for the same reason `state` is: it is what the filter-first page
+ // planner prunes on, and FilterableRecord reads it. OMITTED when empty, and
+ // absent entirely on a site built before corpus spec 4 — which the planner
+ // already handles, since a video the index cannot vouch for gets its page
+ // read anyway.
+ curatedTags?: string[];
};
export type VideoIndex = ReadonlyMap<string, IndexedVideo>;
@@ -273,6 +282,9 @@ export async function buildVideoIndex(
uploadDate: r.uploadDate ?? "",
isLivestream: r.isLivestream === true,
ageRestricted: r.ageRestricted === true,
+ ...(Array.isArray(r.curatedTags) && r.curatedTags.length > 0
+ ? { curatedTags: r.curatedTags.filter((t) => typeof t === "string") }
+ : {}),
});
}
}
@@ -314,6 +326,16 @@ export interface ArchiveReader {
// (e.g. "k cups" → "(k|cake)[ -]?cup") also matches the mis-transcribed
// spellings. Result is cached per source.
loadAliases(): Promise<SearchAlias[]>;
+ // The site's published curated tags (/tags.json) with their per-site counts.
+ // Returns [] when the file is absent or malformed — which is the normal
+ // state for a site with nothing shippable to publish AND for any site built
+ // before corpus spec 4, so a caller must read [] as "this source publishes no
+ // curated tags" and say so, never as an error. Cached per source.
+ //
+ // OPTIONAL on the interface for the same reason duplicateIndex and statsIndex
+ // are: the in-memory test stubs have no such concept, and a tool must degrade
+ // rather than assume.
+ loadTags?(): Promise<PublishedTag[]>;
// The site's channel-group definitions (summaries/manifest.json). Returns the
// empty fallback when absent/malformed, or in hub mode (federated per-site
// groups are a different model — deferred). Cached per source.
@@ -567,6 +589,7 @@ export class RemoteSource implements ArchiveReader {
readonly label: string;
private base: string;
private aliases?: SearchAlias[];
+ private tags?: PublishedTag[];
private groups?: ChannelGroups;
private index?: Promise<Map<string, IndexedVideo>>;
private duplicates?: Promise<DuplicateIndex>;
@@ -701,6 +724,16 @@ export class RemoteSource implements ArchiveReader {
return this.aliases;
}
+ async loadTags(): Promise<PublishedTag[]> {
+ if (this.tags) return this.tags;
+ try {
+ this.tags = coercePublishedTags(await this.readTags());
+ } catch {
+ this.tags = []; // 404 (no shippable tag, or a pre-spec-4 site) or junk
+ }
+ return this.tags;
+ }
+
async loadGroups(): Promise<ChannelGroups> {
if (this.groups) return this.groups;
try {
@@ -762,6 +795,10 @@ export class RemoteSource implements ArchiveReader {
return this.getJson<unknown>(rootFileUrl("search-aliases.json"));
}
+ readTags(): Promise<unknown> {
+ return this.getJson<unknown>(rootFileUrl(TAGS_FILENAME));
+ }
+
readDuplicates(): Promise<DuplicateReport> {
return this.getJson<DuplicateReport>(rootFileUrl(DUPLICATES_FILENAME));
}
diff --git a/common/lib/corpus.test.ts b/common/lib/corpus.test.ts
@@ -159,16 +159,41 @@ test("both builders stamp the project generator, and llms.txt trails it", () =>
}
});
-test("adding `generator` did NOT bump the corpus spec", () => {
- // Guard on the reasoning, not just the number: prior bumps announced a new
- // fetchable layer. An informational credit string breaks no reader, so a
- // client pinned to spec 3 must keep validating. If this assertion is ever
- // updated, the version-history comment in corpus.ts must justify why.
- assert.equal(CORPUS_SPEC_VERSION, 3);
- assert.equal(
- buildSiteCorpus(descriptor(), { hasArchives: false }).spec,
- 3,
+test("the spec is 4, and `generator` is still not why", () => {
+ // Guard on the reasoning, not just the number: a bump announces a new
+ // FETCHABLE document. spec 4 is /tags.json. The informational credit string
+ // added at spec 3's time breaks no reader and did not bump anything — if
+ // either assertion is ever updated, the version-history comment in corpus.ts
+ // must justify why.
+ assert.equal(CORPUS_SPEC_VERSION, 4);
+ assert.equal(buildSiteCorpus(descriptor(), { hasArchives: false }).spec, 4);
+});
+
+test("the tags pointer is present only when the site published one", () => {
+ // No /tags.json → corpus.json is shaped exactly as before the bump.
+ const none = buildSiteCorpus(descriptor(), { hasArchives: false });
+ assert.equal(none.tags, undefined);
+ assert.doesNotMatch(renderSiteLlmsTxt(none), /tags\.json/);
+
+ const tagged = buildSiteCorpus(
+ descriptor({ siteUrl: "https://demo.example" }),
+ { hasArchives: false, hasTags: true },
);
+ assert.equal(tagged.tags?.url, "https://demo.example/tags.json");
+ // The field name is the whole point: NOT `tags`, which is yt-dlp keywords.
+ assert.equal(tagged.tags?.videoField, "curatedTags");
+ assert.match(tagged.tags?.description ?? "", /curatedTags/);
+ const txt = renderSiteLlmsTxt(tagged);
+ assert.match(txt, /https:\/\/demo\.example\/tags\.json/);
+ assert.match(txt, /curatedTags/);
+});
+
+test("a root-relative build still names tags.json correctly", () => {
+ const corpus = buildSiteCorpus(descriptor(), {
+ hasArchives: false,
+ hasTags: true,
+ });
+ assert.equal(corpus.tags?.url, "/tags.json");
});
test("renderRobotsTxt: sitemap line only with an absolute siteUrl", () => {
diff --git a/common/lib/corpus.ts b/common/lib/corpus.ts
@@ -1,5 +1,6 @@
import type { PublicSiteDescriptor } from "./siteDescriptor";
import { PROJECT_GENERATOR } from "./project";
+import { TAGS_FILENAME } from "./curatedTags";
import {
CONTRACT,
archiveUrl,
@@ -26,6 +27,12 @@ import {
// (postScheme + per-channel posts manifest pointers).
// v3: …and the DERIVED layer — AI digests (chapters + topic tags) served under
// the same shard scheme (digestScheme + per-channel digests manifest pointers).
+// v4: curated tags — an operator-authored, cross-channel vocabulary published
+// at /tags.json, with the per-record key `curatedTags` on summaries and
+// transcript shards. A new FETCHABLE DOCUMENT, which is exactly what a bump
+// announces: a reader that skips it cannot narrow a corpus to "every stream
+// where X collabs". The per-record key itself is additive and bumps no manifest
+// version — an older reader ignores an unknown field, as it always has.
//
// NOT a v4: the `generator` field added below is deliberately unversioned. Every
// prior bump announced a new FETCHABLE LAYER — a reader that ignored it would
@@ -176,6 +183,11 @@ export type SiteCorpus = {
postScheme?: typeof POST_SCHEME;
// Present when this site ships at least one digested video.
digestScheme?: typeof DIGEST_SCHEME;
+ // Present when this site publishes at least one curated tag. `url` is
+ // /tags.json (the vocabulary + this site's per-tag counts) and `videoField`
+ // names the per-record key those ids appear in — deliberately NOT `tags`,
+ // which on a transcript record is the platform's own keywords.
+ tags?: { url: string; videoField: "curatedTags"; description: string };
// Present when this build ships bulk-download archives (whole-channel zips).
bulkArchives?: { manifest: string; note: string };
// Pointer to the human page and BYO-key chat.
@@ -211,6 +223,10 @@ export function buildSiteCorpus(
// slug -> digested video count. Absent / empty means the site ships no
// derived layer and digestScheme is omitted.
digestCounts?: Record<string, number>;
+ // Whether this site published a /tags.json (compose writes one only when a
+ // visible tag has a non-zero count here). Absent/false leaves corpus.json
+ // shaped as before apart from the spec bump.
+ hasTags?: boolean;
},
): SiteCorpus {
const base = descriptor.siteUrl;
@@ -266,6 +282,19 @@ export function buildSiteCorpus(
if (Object.values(digestCounts).some((n) => n > 0)) {
corpus.digestScheme = DIGEST_SCHEME;
}
+ // The curated vocabulary, when this site ships one.
+ if (opts.hasTags) {
+ corpus.tags = {
+ url: rootFileUrl(TAGS_FILENAME, base),
+ videoField: "curatedTags",
+ description:
+ "Curated cross-channel tags, applied per video by the archive's " +
+ "operator. Fetch tags.json for the vocabulary (id, label, group and " +
+ "the number of videos carrying it on this site), then filter records " +
+ "by their `curatedTags` array — which is NOT the same field as `tags`, " +
+ "the platform's own keywords. Absent on archives built before spec 4.",
+ };
+ }
if (opts.hasArchives) {
corpus.bulkArchives = {
manifest: join(base, "/archives/manifest.json"),
@@ -353,6 +382,14 @@ export function renderSiteLlmsTxt(corpus: SiteCorpus): string {
`see corpus.json's digestScheme. Coverage is partial and growing.`,
);
}
+ if (corpus.tags) {
+ out.push(
+ `- [tags.json](${corpus.tags.url}): curated cross-channel tags applied ` +
+ `per video by this archive's operator — the vocabulary plus how many ` +
+ `videos carry each one here. Records name them in \`curatedTags\` ` +
+ `(not \`tags\`, which is the platform's own keywords).`,
+ );
+ }
if (corpus.bulkArchives) {
out.push(
`- [Bulk archives](${join(base, "/downloads")}): whole-channel transcript ` +
diff --git a/common/lib/curatedTags.ts b/common/lib/curatedTags.ts
@@ -469,7 +469,14 @@ export function compileTagRules(defs: CuratedTagDef[]): CompiledTagRules {
// Channel scope and date range are cheap string work and are applied BEFORE any
// regex runs — the whole point of scoping a rule to one channel is not paying
// for it on the other twenty-nine.
-function ruleApplies(rule: CompiledTagRule, input: TagRuleInput): boolean {
+// Exported because the index build asks the same question one step earlier — to
+// decide whether a video's cues are worth decoding AT ALL
+// (curatedTagsIndex.ts). A second copy of this predicate could disagree with
+// the evaluation it is supposed to be predicting.
+export function ruleApplies(
+ rule: CompiledTagRule,
+ input: { channelSlug: string; uploadDate?: string },
+): boolean {
if (rule.channels && !rule.channels.has(input.channelSlug)) return false;
if (rule.dateFrom || rule.dateTo) {
const date = (input.uploadDate ?? "").replace(/\D/g, "");
diff --git a/common/lib/curatedTagsStore.test.ts b/common/lib/curatedTagsStore.test.ts
@@ -57,6 +57,17 @@ const COLLAB: CuratedTagsConfig = {
assignments: {},
};
+// applyTagAssignments refuses to pin or suppress a tag the vocabulary does not
+// define, so most of these tests need one. Kept explicit per test rather than
+// seeded globally — the refusal itself is a test below.
+function seedVocab(paths: Paths, ...ids: string[]): void {
+ writeGlobalTags(paths, {
+ version: 1,
+ tags: ids.map((id) => ({ id, label: id })),
+ assignments: {},
+ });
+}
+
const V1 = { channelSlug: "legal-mindset", id: "XZqL6k9IHGA" };
const V2 = { channelSlug: "legal-mindset", id: "AAAAAAAAAAA" };
@@ -214,8 +225,44 @@ test("applyTagAssignments rejects a bad op, tag id or video ref", () => {
});
});
+test("add and suppress refuse a tag the vocabulary does not define", () => {
+ // The typo guard. Without it `eva-collabs` would ride on records for ever:
+ // no /tags.json can show a tag that has no definition, so nothing would
+ // surface the mistake.
+ withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
+ for (const op of ["add", "suppress"] as const) {
+ assert.throws(
+ () => applyTagAssignments(paths, { op, tag: "eva-collabs", videos: [V1] }),
+ /unknown tag "eva-collabs" — define it first/,
+ op,
+ );
+ }
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
+test("remove and unsuppress still work for a tag whose def was deleted", () => {
+ // The other half of the guard: those two only ever take a claim away, and an
+ // operator must be able to clean up after deleting a definition.
+ withPaths((paths) => {
+ writeGlobalTags(paths, {
+ version: 1,
+ tags: [],
+ assignments: {
+ "legal-mindset/XZqL6k9IHGA": { manual: ["gone"] },
+ "legal-mindset/AAAAAAAAAAA": { suppressed: ["gone"] },
+ },
+ });
+ applyTagAssignments(paths, { op: "remove", tag: "gone", videos: [V1] });
+ applyTagAssignments(paths, { op: "unsuppress", tag: "gone", videos: [V2] });
+ assert.deepEqual(readGlobalTags(paths).assignments, {});
+ });
+});
+
test("applyTagAssignments lowercases the tag id it stores", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
applyTagAssignments(paths, { op: "add", tag: " EVA-Collab ", videos: [V1] });
assert.deepEqual(readGlobalTags(paths).assignments["legal-mindset/XZqL6k9IHGA"].manual, [
"eva-collab",
@@ -227,6 +274,7 @@ test("applyTagAssignments lowercases the tag id it stores", () => {
test("add pins with provenance and is idempotent", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
const first = applyTagAssignments(paths, {
op: "add",
tag: "eva-collab",
@@ -266,6 +314,7 @@ test("add pins with provenance and is idempotent", () => {
test("add clears a suppression of the same tag", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
const a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
@@ -276,6 +325,7 @@ test("add clears a suppression of the same tag", () => {
test("remove unpins only — a rule hit survives it, which is why suppress exists", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
applyTagAssignments(paths, { op: "remove", tag: "eva-collab", videos: [V1] });
const cfg = readGlobalTags(paths);
@@ -287,6 +337,7 @@ test("remove unpins only — a rule hit survives it, which is why suppress exist
test("suppress rejects a rule-derived tag and records provenance", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-topic");
const res = applyTagAssignments(paths, {
op: "suppress",
tag: "eva-topic",
@@ -307,6 +358,7 @@ test("suppress rejects a rule-derived tag and records provenance", () => {
test("suppress also unpins, and unsuppress clears the rejection", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1] });
applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
let a = assignmentFor(readGlobalTags(paths), V1.channelSlug, V1.id)!;
@@ -333,6 +385,7 @@ test("an op that changes nothing writes no file at all", () => {
test("other tags on the same video, and other videos, are untouched", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab", "eva-topic");
applyTagAssignments(paths, { op: "add", tag: "eva-collab", videos: [V1, V2] });
applyTagAssignments(paths, { op: "add", tag: "eva-topic", videos: [V1] });
applyTagAssignments(paths, { op: "suppress", tag: "eva-collab", videos: [V1] });
@@ -356,6 +409,7 @@ test("the tag vocabulary survives an assignment write", () => {
test("a bulk call is ONE write, and reports exactly the keys it changed", () => {
withPaths((paths) => {
+ seedVocab(paths, "eva-collab");
const many = Array.from({ length: 50 }, (_, i) => ({
channelSlug: "legal-mindset",
id: `v${i}`,
@@ -380,7 +434,10 @@ test("provenance for a tag that is neither pinned nor suppressed is pruned", ()
paths.globalTagsFile,
JSON.stringify({
version: 1,
- tags: [],
+ tags: [
+ { id: "eva-collab", label: "Collab" },
+ { id: "eva-topic", label: "Discussed" },
+ ],
assignments: {
"legal-mindset/XZqL6k9IHGA": {
manual: ["eva-collab"],
diff --git a/common/lib/curatedTagsStore.ts b/common/lib/curatedTagsStore.ts
@@ -3,9 +3,16 @@
// - per-site: sites/<siteId>/tags.json (siteTagsFile)
// The global file is AUTHORITATIVE for both the vocabulary and the assignments
// (an assignment is a fact about a video, not a presentation choice). The
-// per-site file is presentation — relabel/recolour/reorder/hide — plus
-// site-only rules and site-only tags. See common/lib/curatedTags.ts for the
-// pure model, the coercion and mergeTagDefs.
+// per-site file is PRESENTATION — relabel/recolour/reorder/hide — plus
+// site-only tags. See common/lib/curatedTags.ts for the pure model, the
+// coercion and mergeTagDefs.
+//
+// **A site-layer RULE never tags a record.** mergeTagDefs will append one (it
+// is harmless on a published /tags.json def), but rules are evaluated once at
+// index time over records SHARED by every site that carries the channel —
+// there is no per-site record for a per-site hit to live in. Rules bind only
+// from transcripts/tags.json; promote one there to make it fire. This is why
+// the editor offers no site-side rule editor.
//
// Unlike aliasesStore there are NO seeded defaults: a fresh install has no
// tags, and an absent global file reads as an empty config.
@@ -142,8 +149,11 @@ export type ApplyTagAssignmentsResult = {
touched: string[];
};
+// Order-preserving de-dupe, Set-backed: a bulk "tag selected" can carry
+// thousands of refs, and `keys.includes` would make that quadratic.
function dedupeRefs(videos: TagVideoRef[]): string[] {
const keys: string[] = [];
+ const seen = new Set<string>();
for (const v of videos) {
if (!v || typeof v.channelSlug !== "string" || typeof v.id !== "string") {
throw new Error("each video needs a channelSlug and an id");
@@ -156,7 +166,9 @@ function dedupeRefs(videos: TagVideoRef[]): string[] {
);
}
const key = assignmentKey(slug, id);
- if (!keys.includes(key)) keys.push(key);
+ if (seen.has(key)) continue;
+ seen.add(key);
+ keys.push(key);
}
return keys;
}
@@ -166,11 +178,21 @@ function withoutTag(list: string[] | undefined, tag: string): string[] {
}
// Apply one op for one tag to any number of videos, in a SINGLE atomic write of
-// the global file. Validates the tag id and the op, and records provenance for
-// every pin and suppression it writes. Throws on invalid input (an unknown op,
-// a malformed tag id or video ref) — this is the one write path all three
-// writers (editor UI, ops route, umtool) funnel through, so it refuses rather
-// than silently storing junk a sanitize pass would drop later.
+// the global file. Validates the tag id, the op and the video refs, checks the
+// tag EXISTS in the vocabulary for the two ops that add a claim, and records
+// provenance for every pin and suppression it writes. It throws rather than
+// silently storing junk a sanitize pass would drop later: this is the one write
+// path all three writers (editor UI, ops route, umtool) funnel through, and a
+// mistyped id here would put a tag on records that no /tags.json can ever show.
+//
+// `remove` and `unsuppress` deliberately skip the vocabulary check — they only
+// ever take a claim away, and an operator must be able to clean up after a tag
+// whose definition has already been deleted.
+//
+// The read-modify-write below is atomic against other writers in THIS process
+// only because this function is synchronous end to end — there is no await
+// between readGlobalTags and writeGlobalTags for another handler to interleave
+// in. Introducing one would silently make concurrent bulk tagging lossy.
//
// Returns which keys changed, so the caller can dirty exactly those videos.
export function applyTagAssignments(
@@ -195,6 +217,14 @@ export function applyTagAssignments(
const at = input.at ?? new Date().toISOString();
const config = readGlobalTags(paths);
+ if (
+ (op === "add" || op === "suppress") &&
+ !config.tags.some((def) => def.id === tag)
+ ) {
+ throw new Error(
+ `unknown tag "${tag}" — define it first (pnpm ops tags / the /tags page)`,
+ );
+ }
const changed: string[] = [];
for (const key of keys) {
diff --git a/common/lib/publishedTags.test.ts b/common/lib/publishedTags.test.ts
@@ -0,0 +1,123 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import {
+ coercePublishedTags,
+ groupPublishedTags,
+ selectableTags,
+ tagLabels,
+} from "./publishedTags";
+
+const tag = (over: Record<string, unknown> = {}) => ({
+ id: "eva-collab",
+ label: "Collab",
+ count: 2,
+ channels: { "legal-mindset": 2 },
+ ...over,
+});
+
+test("coercePublishedTags reads the document compose-site writes", () => {
+ const tags = coercePublishedTags({
+ version: 1,
+ tags: [
+ tag({ group: "eva", groupLabel: "Eva", color: "#b48ead", order: 1 }),
+ tag({ id: "eva-topic", label: "Discussed", order: 2, count: 1 }),
+ ],
+ });
+ assert.equal(tags.length, 2);
+ assert.deepEqual(tags.map((t) => t.id), ["eva-collab", "eva-topic"]);
+ assert.equal(tags[0].groupLabel, "Eva");
+ assert.equal(tags[0].count, 2);
+ assert.deepEqual(tags[0].channels, { "legal-mindset": 2 });
+});
+
+test("absence and junk both read as 'this site publishes no tags'", () => {
+ // The two shapes of absence a viewer actually meets: the file is a 404 (a
+ // site with nothing shippable, or one built before corpus spec 4) and the
+ // file is there but unrecognisable. Neither may throw.
+ for (const raw of [null, undefined, "", 42, {}, { tags: null }, { tags: {} }]) {
+ assert.deepEqual(coercePublishedTags(raw), []);
+ }
+});
+
+test("a malformed entry is dropped, not fatal", () => {
+ const tags = coercePublishedTags({
+ version: 1,
+ tags: [
+ null,
+ { label: "no id" },
+ tag({ id: "Not A Tag Id" }),
+ tag({ count: undefined }), // no count — cannot be offered as a chip
+ tag({ count: "many" }),
+ tag(),
+ ],
+ });
+ assert.deepEqual(tags.map((t) => t.id), ["eva-collab"]);
+});
+
+test("ids are lowercased and the first definition of an id wins", () => {
+ const tags = coercePublishedTags({
+ tags: [tag({ id: "EVA-COLLAB", label: "First" }), tag({ label: "Second" })],
+ });
+ assert.equal(tags.length, 1);
+ assert.equal(tags[0].id, "eva-collab");
+ assert.equal(tags[0].label, "First");
+});
+
+test("a tag with no label falls back to its id", () => {
+ const tags = coercePublishedTags({ tags: [tag({ label: " " })] });
+ assert.equal(tags[0].label, "eva-collab");
+});
+
+test("tags sort by order, then id — the same row on every site", () => {
+ const tags = coercePublishedTags({
+ tags: [
+ tag({ id: "c", order: 3 }),
+ tag({ id: "a", order: 1 }),
+ tag({ id: "b", order: 1 }),
+ ],
+ });
+ assert.deepEqual(tags.map((t) => t.id), ["a", "b", "c"]);
+});
+
+test("selectableTags refuses a chip that would match nothing", () => {
+ const tags = coercePublishedTags({
+ tags: [tag(), tag({ id: "stale", count: 0, channels: {} })],
+ });
+ assert.deepEqual(selectableTags(tags).map((t) => t.id), ["eva-collab"]);
+});
+
+test("groupPublishedTags keeps tag order and sinks the ungrouped bucket", () => {
+ const tags = coercePublishedTags({
+ tags: [
+ tag({ id: "loose", order: 1 }),
+ tag({ id: "eva-collab", group: "eva", groupLabel: "Eva", order: 2 }),
+ tag({ id: "eva-topic", group: "eva", order: 3 }),
+ ],
+ });
+ const groups = groupPublishedTags(tags);
+ assert.deepEqual(
+ groups.map((g) => [g.id, g.label, g.tags.map((t) => t.id)]),
+ [
+ ["eva", "Eva", ["eva-collab", "eva-topic"]],
+ ["", "", ["loose"]],
+ ],
+ );
+});
+
+test("a group label carried by a later member is still used", () => {
+ const groups = groupPublishedTags(
+ coercePublishedTags({
+ tags: [
+ tag({ id: "a", group: "eva", order: 1 }),
+ tag({ id: "b", group: "eva", groupLabel: "Eva", order: 2 }),
+ ],
+ }),
+ );
+ assert.equal(groups[0].label, "Eva");
+});
+
+test("tagLabels falls back to the id for a tag this site does not publish", () => {
+ const labels = tagLabels(coercePublishedTags({ tags: [tag()] }));
+ assert.equal(labels.get("eva-collab"), "Collab");
+ assert.equal(labels.get("eva-topic"), undefined);
+});
diff --git a/common/lib/publishedTags.ts b/common/lib/publishedTags.ts
@@ -0,0 +1,151 @@
+// The READER side of the published /tags.json.
+//
+// `lib/curatedTags.ts` is the authoring model — defs, rules, assignments, and
+// the folds the index build runs. This file is the other end of the wire: what
+// a viewer, an MCP source or any other client does with the small presentation
+// document compose-site writes out. It is deliberately a separate module from
+// the frozen one:
+//
+// - the authoring model never travels to a browser (rules, regexes,
+// provenance and every assignment stay server-side), and
+// - a published document is UNTRUSTED input to its reader — a federated hub
+// reads /tags.json from origins it does not control — so the coercion has
+// to be as tolerant as coerceAliasConfig, dropping anything malformed
+// rather than throwing.
+//
+// Absence is a legitimate empty state at every layer: compose-site writes no
+// file for a site with nothing shippable (exactly like duplicates.json), and a
+// site built before corpus spec 4 has never heard of tags at all. Both arrive
+// here as `[]`, and every caller must treat that as "this site publishes no
+// curated tags" rather than as an error.
+
+import type { PublishedTag, PublishedTags } from "./curatedTags";
+import { TAG_ID_RE } from "./curatedTags";
+
+function str(v: unknown): string | undefined {
+ if (typeof v !== "string") return undefined;
+ const s = v.trim();
+ return s === "" ? undefined : s;
+}
+
+function num(v: unknown): number | undefined {
+ return typeof v === "number" && Number.isFinite(v) ? v : undefined;
+}
+
+function coerceChannels(v: unknown): Record<string, number> {
+ const out: Record<string, number> = {};
+ if (!v || typeof v !== "object") return out;
+ for (const [slug, n] of Object.entries(v as Record<string, unknown>)) {
+ const count = num(n);
+ if (!slug.trim() || count === undefined || count < 0) continue;
+ out[slug] = Math.floor(count);
+ }
+ return out;
+}
+
+function coerceTag(raw: unknown): PublishedTag | null {
+ if (!raw || typeof raw !== "object") return null;
+ const r = raw as Record<string, unknown>;
+ const id = str(r.id)?.toLowerCase();
+ if (!id || !TAG_ID_RE.test(id)) return null;
+ // A count is the one field the chip row cannot invent: the operator's rule
+ // is that a tag is offered only when it has matches ON THIS SITE, so a
+ // countless entry is dropped rather than shown as a chip that finds nothing.
+ const count = num(r.count);
+ if (count === undefined || count < 0) return null;
+ return {
+ id,
+ label: str(r.label) ?? id,
+ ...(str(r.group) ? { group: str(r.group) } : {}),
+ ...(str(r.groupLabel) ? { groupLabel: str(r.groupLabel) } : {}),
+ ...(str(r.color) ? { color: str(r.color) } : {}),
+ ...(num(r.order) !== undefined ? { order: num(r.order) } : {}),
+ count: Math.floor(count),
+ channels: coerceChannels(r.channels),
+ };
+}
+
+// Parse a published /tags.json into the tag list, sorted the way chips render:
+// by `order` (defaulting to the document's own position) then by id, so two
+// sites that ship the same vocabulary show it in the same order. NEVER throws;
+// anything unrecognisable yields [].
+export function coercePublishedTags(raw: unknown): PublishedTag[] {
+ if (!raw || typeof raw !== "object") return [];
+ const list = (raw as Partial<PublishedTags>).tags;
+ if (!Array.isArray(list)) return [];
+ const out: PublishedTag[] = [];
+ const seen = new Set<string>();
+ list.forEach((entry) => {
+ const tag = coerceTag(entry);
+ if (!tag) return;
+ if (seen.has(tag.id)) return; // first definition of an id wins
+ seen.add(tag.id);
+ out.push(tag);
+ });
+ const rank = new Map<string, number>();
+ out.forEach((t, i) => rank.set(t.id, t.order ?? i));
+ return out.sort((a, b) => {
+ const ra = rank.get(a.id)!;
+ const rb = rank.get(b.id)!;
+ if (ra !== rb) return ra - rb;
+ return a.id < b.id ? -1 : a.id > b.id ? 1 : 0;
+ });
+}
+
+// The chips the UI may offer: visible tags with at least one video ON THIS
+// SITE. compose-site already drops hidden and zero-count tags, so this is a
+// belt-and-braces re-check on the reader side — a hand-written or stale
+// /tags.json must not be able to put a chip that matches nothing in front of
+// someone.
+export function selectableTags(
+ tags: readonly PublishedTag[],
+): PublishedTag[] {
+ return tags.filter((t) => t.count > 0);
+}
+
+// Group the selectable tags for a chip row, preserving the tag order within
+// each group and ordering the groups by their first member. Ungrouped tags
+// fall into one trailing bucket with no label, which is what a site using
+// free-form ids and no `group` gets.
+export type PublishedTagGroup = {
+ // "" for the ungrouped bucket.
+ id: string;
+ label: string;
+ tags: PublishedTag[];
+};
+
+export function groupPublishedTags(
+ tags: readonly PublishedTag[],
+): PublishedTagGroup[] {
+ const out: PublishedTagGroup[] = [];
+ const byId = new Map<string, PublishedTagGroup>();
+ for (const tag of tags) {
+ const id = tag.group ?? "";
+ let group = byId.get(id);
+ if (!group) {
+ group = { id, label: "", tags: [] };
+ byId.set(id, group);
+ out.push(group);
+ }
+ // The first EXPLICIT groupLabel wins; a later member may carry the one the
+ // first omitted, so the id fallback is applied after the whole walk rather
+ // than on first sight (which would make the upgrade unreachable).
+ if (id && !group.label && tag.groupLabel) group.label = tag.groupLabel;
+ group.tags.push(tag);
+ }
+ for (const group of out) {
+ if (group.id && !group.label) group.label = group.id;
+ }
+ // The ungrouped bucket always sorts last: it is the leftovers, not a group.
+ return out.sort((a, b) => (a.id === "" ? 1 : b.id === "" ? -1 : 0));
+}
+
+// id -> label, for rendering a record's `curatedTags` on a card. A tag the
+// site does not publish (hidden, zero-count here, or defined after this page
+// was built) falls back to its id — the fact is on the record either way, and
+// showing the raw id is more honest than dropping it.
+export function tagLabels(
+ tags: readonly PublishedTag[],
+): ReadonlyMap<string, string> {
+ return new Map(tags.map((t) => [t.id, t.label]));
+}
diff --git a/common/lib/search/evalTree.test.ts b/common/lib/search/evalTree.test.ts
@@ -321,3 +321,56 @@ test("needsAvailability and filterIsSelective are both false for an all-permissi
true,
);
});
+
+// ─── curated tags (tg) ───
+
+test("the curated-tag filter is an OR over the selection", () => {
+ const untagged = { uploadDate: "20250601" };
+ const collab = { uploadDate: "20250601", curatedTags: ["eva-collab"] };
+ const both = {
+ uploadDate: "20250601",
+ curatedTags: ["eva-collab", "eva-topic"],
+ };
+
+ // ANY of the selected tags is enough — the chips union, they do not intersect.
+ const one = openFilters({ curatedTags: ["eva-collab"] });
+ assert.equal(passesFilters(collab, one, undefined), true);
+ assert.equal(passesFilters(both, one, undefined), true);
+ assert.equal(passesFilters(untagged, one, undefined), false);
+
+ const two = openFilters({ curatedTags: ["eva-collab", "eva-in-chat"] });
+ assert.equal(passesFilters(collab, two, undefined), true);
+ assert.equal(
+ passesFilters({ uploadDate: "20250601", curatedTags: ["eva-in-chat"] }, two, undefined),
+ true,
+ );
+ assert.equal(passesFilters(untagged, two, undefined), false);
+
+ // A tag nothing carries excludes everything rather than erroring.
+ const none = openFilters({ curatedTags: ["nobody"] });
+ assert.equal(passesFilters(both, none, undefined), false);
+});
+
+test("an empty curated-tag selection is no filter at all", () => {
+ // The chip row starts empty and hands its state straight through, so [] must
+ // read as "unfiltered" — never as "matches nothing".
+ const empty = openFilters({ curatedTags: [] });
+ assert.equal(passesFilters({ uploadDate: "20250601" }, empty, undefined), true);
+ assert.equal(filterIsSelective(empty), false);
+});
+
+test("a record from a pre-spec-4 site carries no tags and passes no tag filter", () => {
+ // Records built before curated tags existed have no `curatedTags` key at all.
+ // That is not "matches every tag" — it is "carries none".
+ const legacy = { uploadDate: "20250601", isLivestream: false };
+ assert.equal(
+ passesFilters(legacy, openFilters({ curatedTags: ["eva-collab"] }), undefined),
+ false,
+ );
+});
+
+test("a non-empty curated-tag selection makes a filter selective", () => {
+ // filterIsSelective gates the MCP's filter-first page pruner: without this,
+ // a tag-only query would plan a full scan.
+ assert.equal(filterIsSelective(openFilters({ curatedTags: ["eva-collab"] })), true);
+});
diff --git a/common/lib/search/evalTree.ts b/common/lib/search/evalTree.ts
@@ -321,6 +321,14 @@ export type SearchFilters = {
// fdf / fdt — inclusive upload-date bounds, "YYYYMMDD".
dateFrom?: string;
dateTo?: string;
+ // tg — curated tag ids (common/lib/curatedTags.ts), ORed together: a record
+ // passes when it carries ANY of them. NOT the yt-dlp keywords in
+ // `TranscriptSummary.tags` — see the naming note there.
+ //
+ // Absent or empty means "no tag filter", which is why an empty array must
+ // never read as "matches nothing": the UI hands this straight through from a
+ // chip row that starts empty.
+ curatedTags?: string[];
};
// Typed on the three fields it actually reads rather than on TranscriptDetail,
@@ -330,6 +338,10 @@ export type FilterableRecord = {
isLivestream?: boolean;
ageRestricted?: boolean;
uploadDate: string;
+ // Curated tag ids carried by the record. OPTIONAL and omitted-when-empty on
+ // the wire, so a record from a site built before corpus spec 4 simply has
+ // none — and, correctly, passes no tag filter.
+ curatedTags?: string[];
};
export function passesFilters(
@@ -346,6 +358,14 @@ export function passesFilters(
// fdf / fdt — upload-date range (lexicographic on YYYYMMDD)
if (f.dateFrom && rec.uploadDate < f.dateFrom) return false;
if (f.dateTo && rec.uploadDate > f.dateTo) return false;
+ // tg — curated tags. OR across the selection (a video tagged either
+ // "eva-collab" OR "eva-in-chat" passes both-chips-selected), which is the
+ // operator's decision: the chips are a union, not an intersection.
+ if (f.curatedTags && f.curatedTags.length > 0) {
+ const has = rec.curatedTags;
+ if (!has || has.length === 0) return false;
+ if (!f.curatedTags.some((t) => has.includes(t))) return false;
+ }
return true;
}
@@ -364,5 +384,6 @@ export function filterIsSelective(f: SearchFilters | null | undefined): boolean
if (!VIDEO_STATES.every((s) => f.states.has(s))) return true;
if (!f.videos || !f.livestreams) return true;
if (!f.allAges || !f.restricted) return true;
+ if (f.curatedTags && f.curatedTags.length > 0) return true;
return Boolean(f.dateFrom || f.dateTo);
}
diff --git a/common/lib/transcripts-server.test.ts b/common/lib/transcripts-server.test.ts
@@ -0,0 +1,55 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { toDisplaySummary } from "./transcripts-server";
+import type { TranscriptSummary } from "./transcripts";
+
+// Run with: node_modules/.bin/tsx --test common/lib/transcripts-server.test.ts
+
+function summary(over: Partial<TranscriptSummary> = {}): TranscriptSummary {
+ return {
+ slug: "test-channel/v1",
+ id: "v1",
+ channelSlug: "test-channel",
+ title: "A stream",
+ uploadDate: "20260101",
+ duration: 3600,
+ channel: "Test Channel",
+ description: "",
+ tags: ["keyword-one", "keyword-two"],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: "https://example.com/v1",
+ ...over,
+ };
+}
+
+test("toDisplaySummary omits curatedTags entirely when there are none", () => {
+ // The byte-identical-pages contract: an untagged corpus's summaries pages
+ // must not gain a key, or every one of them is rewritten on upgrade.
+ const display = toDisplaySummary(summary());
+ assert.equal("curatedTags" in display, false);
+ assert.equal("curatedTags" in toDisplaySummary(summary({ curatedTags: [] })), false);
+ // And the platform's keywords do NOT leak in under the new name.
+ assert.equal(JSON.stringify(display).includes("keyword-one"), false);
+});
+
+test("toDisplaySummary passes curatedTags through in order", () => {
+ const display = toDisplaySummary(
+ summary({ curatedTags: ["eva-collab", "eva-topic"] }),
+ );
+ assert.deepEqual(display.curatedTags, ["eva-collab", "eva-topic"]);
+});
+
+test("toDisplaySummary still derives the state fields it always did", () => {
+ const available = toDisplaySummary(summary());
+ assert.equal(available.isDeleted, false);
+ assert.equal(available.isUnlisted, false);
+ assert.equal("state" in available, false);
+ const deleted = toDisplaySummary(summary({ curatedTags: ["x"] }), {
+ state: "deleted",
+ });
+ assert.equal(deleted.state, "deleted");
+ assert.equal(deleted.isDeleted, true);
+ assert.deepEqual(deleted.curatedTags, ["x"]);
+});
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -220,5 +220,11 @@ export function toDisplaySummary(
...(state === "available" ? {} : { state }),
platform: t.platform,
webpageUrl: t.webpageUrl,
+ // Curated tags ride through to the browse index, so a list can be narrowed
+ // to "streams where X collabs" without loading a transcript page. Omitted
+ // when empty for the same byte-identical reason as `state`.
+ ...(t.curatedTags && t.curatedTags.length > 0
+ ? { curatedTags: t.curatedTags }
+ : {}),
};
}
diff --git a/export/e2e/tag-chips.spec.ts b/export/e2e/tag-chips.spec.ts
@@ -0,0 +1,321 @@
+import { expect, test, type Page } from "@playwright/test";
+import { installRoutes, installTagRoutes } from "./helpers";
+import {
+ CHANNEL_SLUG,
+ TAG_COLLAB,
+ TAG_TOPIC,
+ VIDEO_CHAT_LARGE,
+ VIDEO_CHAT_SMALL,
+ VIDEO_TRANSCRIPT_ONLY,
+} from "./fixtures/data";
+
+// A composite query tree, seeded through `?qt=` exactly as posts-search.spec
+// does. Only the two leaf kinds this file needs.
+function qt(
+ children: { q: string; s: "metadata" | "posts" }[],
+ op: "AND" | "OR" = "OR",
+): string {
+ return encodeURIComponent(
+ JSON.stringify({
+ k: "g",
+ o: op,
+ c: children.map((c) => ({ k: "l", q: c.q, s: c.s })),
+ }),
+ );
+}
+
+// Curated per-video tags (common/lib/curatedTags.ts) in the export viewer: the
+// filter panel's chip row, the `tg` URL param, and the tags a result card
+// carries. Modelled on channel-group-chips.spec.ts, but driven from the shared
+// fixture, because the point of this feature is that the row is built from what
+// THIS SITE publishes — so the fixture's /tags.json and the fixture's
+// `curatedTags` have to be the same two facts the app sees.
+//
+// The fixture tags two of its three videos (see fixtures/data.ts):
+// vid-chat-small → eva-collab
+// vid-chat-large → eva-collab, eva-topic
+// vid-transcript-only → nothing at all (the "omitted when empty" control)
+
+const slugOf = (id: string) => `${CHANNEL_SLUG}/${id}`;
+
+async function waitForHydration(page: Page) {
+ await page.getByTestId("query-builder").waitFor();
+}
+
+function chipRow(page: Page) {
+ return page.getByTestId("tag-chip-row");
+}
+
+function chip(page: Page, id: string) {
+ return page.locator(`[data-testid="tag-chip"][data-tag-id="${id}"]`);
+}
+
+function card(page: Page, id: string) {
+ return page.locator(`[data-result-slug="${slugOf(id)}"]`);
+}
+
+// Tag chips are an ordinary filter: the panel edits a draft, Search commits it.
+async function apply(page: Page) {
+ await page.getByTestId("search-submit").click();
+}
+
+test.describe("curated tag chips", () => {
+ test.describe("on a site that publishes tags", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ // AFTER installRoutes, which leaves /tags.json a 404 by default.
+ await installTagRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+ });
+
+ test("the row is grouped and every chip carries this site's count", async ({
+ page,
+ }) => {
+ await expect(chipRow(page)).toBeVisible();
+
+ // The group label comes from the document, not from anything in the
+ // code: "Eva: Collab · Discussed".
+ const group = page.locator('[data-testid="tag-chip-group"][data-group-id="eva"]');
+ await expect(group).toContainText("Eva");
+
+ // Counts are the site's own, and they agree with the summaries: two
+ // videos carry eva-collab, one carries eva-topic.
+ await expect(chip(page, TAG_COLLAB)).toContainText("Collab");
+ await expect(chip(page, TAG_COLLAB)).toContainText("2");
+ await expect(chip(page, TAG_TOPIC)).toContainText("Discussed");
+ await expect(chip(page, TAG_TOPIC)).toContainText("1");
+
+ // Nothing is selected until someone selects it.
+ await expect(chip(page, TAG_COLLAB)).toHaveAttribute("aria-pressed", "false");
+ });
+
+ test("a tagged card shows its tags; an untagged one shows none", async ({
+ page,
+ }) => {
+ await expect(card(page, VIDEO_CHAT_LARGE).getByTestId("card-tag")).toHaveCount(2);
+ await expect(
+ card(page, VIDEO_CHAT_LARGE).locator('[data-testid="card-tag"][data-tag-id="' + TAG_TOPIC + '"]'),
+ ).toContainText("Discussed");
+ // The label comes from /tags.json, so the card and the chip agree.
+ await expect(card(page, VIDEO_CHAT_SMALL).getByTestId("card-tag")).toHaveCount(1);
+ await expect(
+ card(page, VIDEO_TRANSCRIPT_ONLY).getByTestId("card-tag"),
+ ).toHaveCount(0);
+ });
+
+ test("selecting a chip narrows the list and puts tg on the URL", async ({
+ page,
+ }) => {
+ // All three videos before any tag filter.
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toBeVisible();
+
+ await chip(page, TAG_TOPIC).click();
+ await expect(chip(page, TAG_TOPIC)).toHaveAttribute("aria-pressed", "true");
+ await apply(page);
+
+ // Only the one video carrying eva-topic survives.
+ await expect(card(page, VIDEO_CHAT_LARGE)).toBeVisible();
+ await expect(card(page, VIDEO_CHAT_SMALL)).toHaveCount(0);
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+
+ // …and the selection is on the URL, so the view is linkable.
+ await expect
+ .poll(() => new URL(page.url()).searchParams.getAll("tg"))
+ .toEqual([TAG_TOPIC]);
+ });
+
+ test("two chips are ORed, not ANDed", async ({ page }) => {
+ // No fixture video carries eva-topic alone, and only one carries both —
+ // an AND would leave a single card, which is the bug this pins.
+ await chip(page, TAG_COLLAB).click();
+ await chip(page, TAG_TOPIC).click();
+ await apply(page);
+
+ await expect(card(page, VIDEO_CHAT_SMALL)).toBeVisible();
+ await expect(card(page, VIDEO_CHAT_LARGE)).toBeVisible();
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+ await expect
+ .poll(() => new URL(page.url()).searchParams.getAll("tg").sort())
+ .toEqual([TAG_COLLAB, TAG_TOPIC].sort());
+ });
+
+ test("a reload restores the selection from the URL", async ({ page }) => {
+ await chip(page, TAG_COLLAB).click();
+ await apply(page);
+ await expect
+ .poll(() => new URL(page.url()).searchParams.getAll("tg"))
+ .toEqual([TAG_COLLAB]);
+
+ await page.reload();
+ await waitForHydration(page);
+
+ await expect(chip(page, TAG_COLLAB)).toHaveAttribute("aria-pressed", "true");
+ await expect(card(page, VIDEO_CHAT_SMALL)).toBeVisible();
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+ });
+
+ test("clicking a selected chip clears it and drops the param", async ({
+ page,
+ }) => {
+ await chip(page, TAG_COLLAB).click();
+ await apply(page);
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+
+ // The selected chip IS the clear affordance — there is no second control.
+ await chip(page, TAG_COLLAB).click();
+ await expect(chip(page, TAG_COLLAB)).toHaveAttribute("aria-pressed", "false");
+ await apply(page);
+
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toBeVisible();
+ await expect
+ .poll(() => new URL(page.url()).searchParams.getAll("tg"))
+ .toEqual([]);
+ });
+
+ test("a tag filter takes an explicit posts-scope leaf out of the search", async ({
+ page,
+ }) => {
+ // A post carries no curated tags. The DEFAULT scope build already knew
+ // that, but a scopes:['posts'] leaf is fed its own slug set, so the posts
+ // corpus was still being searched under a tag filter: the rows were then
+ // dropped by the display fold while `totalHits` — which sums the
+ // pipeline's hits BEFORE that fold — still counted them. The header is
+ // the assertion because the header is where it showed: N videos and a
+ // hit count that no card on the page accounts for.
+ //
+ // "chat" matches all three video titles; "alpha" matches two posts.
+ const tree = qt([
+ { q: "chat", s: "metadata" },
+ { q: "alpha", s: "posts" },
+ ]);
+ await page.goto(`/?qt=${tree}&tg=${TAG_COLLAB}`);
+ await waitForHydration(page);
+
+ // Only the two tagged videos, and only their two hits.
+ await expect(page.getByTestId("results-summary")).toHaveText(
+ "Matching videos (2 videos, 2 hits)",
+ );
+ await expect(card(page, VIDEO_CHAT_SMALL)).toBeVisible();
+ await expect(card(page, VIDEO_CHAT_LARGE)).toBeVisible();
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+ });
+
+ test("the same query without a tag filter does search the posts leaf", async ({
+ page,
+ }) => {
+ // The control for the case above: the posts leaf is only silenced BY the
+ // tag filter, never by this change.
+ const tree = qt([
+ { q: "chat", s: "metadata" },
+ { q: "alpha", s: "posts" },
+ ]);
+ await page.goto(`/?qt=${tree}`);
+ await waitForHydration(page);
+
+ await expect(page.getByTestId("results-summary")).toHaveText(
+ "Matching videos (5 videos, 5 hits)",
+ );
+ });
+
+ test("a tg naming an id this site does not publish still gets a chip", async ({
+ page,
+ }) => {
+ // A valid id from another site in the family, or from this one before a
+ // rebuild. Without a chip the Filters badge would say a filter is on
+ // while nothing in the row was lit, and the reader would have no way to
+ // clear the thing that emptied their results.
+ await page.goto("/?tg=not-on-this-site");
+ await waitForHydration(page);
+
+ const orphan = chip(page, "not-on-this-site");
+ await expect(orphan).toBeVisible();
+ await expect(orphan).toHaveAttribute("data-unpublished", "true");
+ await expect(orphan).toContainText("not on this site");
+ await expect(orphan).toHaveAttribute("aria-pressed", "true");
+ // It really is filtering: nothing carries it.
+ await expect(page.getByTestId("results-summary")).toHaveText("All videos (0)");
+
+ // One click clears it, and the param goes with it.
+ await orphan.click();
+ await apply(page);
+ await expect(page.getByTestId("tag-chip")).toHaveCount(2);
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toBeVisible();
+ await expect
+ .poll(() => new URL(page.url()).searchParams.getAll("tg"))
+ .toEqual([]);
+ });
+
+ test("a malformed tg token is dropped at the URL, not carried", async ({
+ page,
+ }) => {
+ // The URL is an entry point like the MCP's parseSearchArgs and agrees
+ // with it about what an id is. A token that can never match must not
+ // become a filter that silently returns nothing.
+ await page.goto("/?tg=Not%20An%20Id");
+ await waitForHydration(page);
+
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toBeVisible();
+ await expect(page.locator("[data-unpublished]")).toHaveCount(0);
+ });
+
+ test("a tg link arrives already filtered", async ({ page }) => {
+ await page.goto(`/?tg=${TAG_TOPIC}`);
+ await waitForHydration(page);
+
+ await expect(chip(page, TAG_TOPIC)).toHaveAttribute("aria-pressed", "true");
+ await expect(card(page, VIDEO_CHAT_LARGE)).toBeVisible();
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+ });
+ });
+
+ // The Filters chip only exists below xl — from xl the panel is inline under
+ // the bar and a chip would be a second copy of a control already on screen.
+ test.describe("on a phone", () => {
+ test.use({ viewport: { width: 390, height: 844 } });
+
+ test("the Filters badge counts a tag selection", async ({ page }) => {
+ await installRoutes(page);
+ await installTagRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+
+ const trigger = page.getByTestId("filters-trigger");
+ // Nothing narrowed yet: the badge is absent, not zero.
+ await expect(trigger).toHaveText("Filters");
+
+ await trigger.click();
+ await chip(page, TAG_COLLAB).click();
+ await page.getByRole("button", { name: "Apply filters" }).click();
+
+ // Without this the reader could apply a filter from the sheet and see no
+ // sign of it on the chip that opens the sheet.
+ await expect(trigger).toContainText("1");
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toHaveCount(0);
+ });
+ });
+
+ test.describe("on a site that publishes none", () => {
+ test("there is no chip row at all", async ({ page }) => {
+ // installRoutes alone: /tags.json 404s, which is the real default for a
+ // site with nothing shippable AND for every site built before corpus
+ // spec 4. No chips, and no empty "Tags" header either — a row offering
+ // nothing is worse than no row.
+ await installRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+
+ await expect(card(page, VIDEO_TRANSCRIPT_ONLY)).toBeVisible();
+ await expect(chipRow(page)).toHaveCount(0);
+ await expect(page.getByTestId("tag-chip")).toHaveCount(0);
+
+ // The records still carry their tags — a site choosing not to publish a
+ // vocabulary does not erase a fact about a video — so the card still
+ // shows them, by their raw ids, because there are no labels to use.
+ await expect(card(page, VIDEO_CHAT_LARGE).getByTestId("card-tag")).toHaveCount(2);
+ await expect(card(page, VIDEO_CHAT_LARGE).getByTestId("card-tag").first()).toHaveText(
+ TAG_COLLAB,
+ );
+ });
+ });
+});
diff --git a/export/service-worker/site-sw.js b/export/service-worker/site-sw.js
@@ -11,8 +11,8 @@
* PAGES — every published JSON document: the per-channel shard trees
* (manifest.json + page-NNNN.json), the site-wide flat trees
* (summaries, stats) and the root documents (corpus.json, site.json,
- * search-aliases.json, duplicates.json). Per-channel pages are
- * cache-first; per-channel manifests are network-first so a rebuild is
+ * search-aliases.json, duplicates.json, tags.json). Per-channel
+ * pages are cache-first; per-channel manifests are network-first so a rebuild is
* seen, with invalidation keyed off manifest.generatedAt — when a
* channel's manifest generatedAt changes, that channel's cached page
* shards are evicted (the shards have stable, non-content-hashed URLs,
@@ -40,7 +40,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
// Activate immediately — no precache list (corpus is too large to bundle).
diff --git a/export/service-worker/sw-hub.js b/export/service-worker/sw-hub.js
@@ -33,7 +33,7 @@ const SHARD_RE = /^\/(transcripts|subs|posts|digests)\/([^/]+)\/(.+)$/;
// Flat trees: one manifest and its pages at the tree root, no channel level.
const FLAT_RE = /^\/(summaries|stats)\/(.+)$/;
// The root documents a reader fetches by name.
-const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json)$/;
+const ROOT_RE = /^\/(corpus\.json|site\.json|search-aliases\.json|duplicates\.json|tags\.json)$/;
self.addEventListener("install", () => {
self.skipWaiting();
diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts
@@ -192,8 +192,11 @@ export function buildSweepInstructions(
`\`["deleted","private","members_only","unlisted","maybe_missing"]\` for ` +
`"what did the videos that are now GONE say"), \`date_from\`/` +
`\`date_to\`, \`media_type\`, \`age\`, \`exclude\` (video-level NOT, ` +
- `for "cup" but not "world cup"), and \`scopes\` to search descriptions, ` +
- `tags or live chat instead of captions. Use them when the question ` +
+ `for "cup" but not "world cup"), \`tags\` — the operator's CURATED ` +
+ `per-video tags, cutting across channels ("every stream where X is on ` +
+ `mic"); call \`list_tags\` for the ids this corpus publishes rather ` +
+ `than guessing one — and \`scopes\` to search descriptions, ` +
+ `keywords or live chat instead of captions. Use them when the question ` +
`implies them: a filtered scan reads only the shard pages that can hold ` +
`a match, which is the difference between seconds and a minute per ` +
`query — and it makes the answer narrower and more honest at the same ` +
diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts
@@ -32,6 +32,7 @@ const TSX = path.join(HERE, "..", "node_modules", ".bin", "tsx");
// or an accidental reshuffle has to be a deliberate edit here.
const EXPECTED_TOOLS = [
"list_channels",
+ "list_tags",
"search_transcripts",
"enumerate_matches",
"get_transcript",
@@ -47,7 +48,7 @@ const EXPECTED_TOOLS = [
];
// A minimal composed public dir: one channel, two videos, real shard layout.
-async function writeFixture(): Promise<string> {
+async function writeFixture(opts: { tags?: boolean } = {}): Promise<string> {
const dir = await mkdtemp(path.join(os.tmpdir(), "mcp-protocol-"));
await writeFile(
path.join(dir, "corpus.json"),
@@ -68,7 +69,41 @@ async function writeFixture(): Promise<string> {
slugToPage: { a1: 0, a2: 0 },
}),
);
- const rec = (id: string, title: string, text: string) => ({
+ // Curated tags, opt-in per fixture. The DEFAULT is a site with no
+ // /tags.json, because that is what every site built before corpus spec 4
+ // looks like and it is the path most likely to be got wrong.
+ if (opts.tags) {
+ await writeFile(
+ path.join(dir, "tags.json"),
+ JSON.stringify({
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ count: 1,
+ channels: { "chan-a": 1 },
+ },
+ {
+ id: "loose-tag",
+ label: "Loose",
+ order: 2,
+ count: 1,
+ channels: { "chan-a": 1 },
+ },
+ ],
+ }),
+ );
+ }
+ const rec = (
+ id: string,
+ title: string,
+ text: string,
+ curatedTags?: string[],
+ ) => ({
slug: `chan-a/${id}`,
id,
channelSlug: "chan-a",
@@ -82,13 +117,18 @@ async function writeFixture(): Promise<string> {
ageRestricted: false,
platform: "youtube",
webpageUrl: `https://example.test/${id}`,
+ // Omitted when empty, exactly as the index build writes it — so the
+ // no-tags fixture's records are byte-identical to a pre-spec-4 site's.
+ ...(curatedTags && curatedTags.length > 0 ? { curatedTags } : {}),
cues: [{ start: 12, end: 15, text }],
});
+ // The records agree with the /tags.json above: one video per tag, which is
+ // what its counts claim.
await writeFile(
path.join(chDir, pageFileName(0)),
JSON.stringify([
- rec("a1", "Coffee one", "i love coffee"),
- rec("a2", "Tea two", "i love tea"),
+ rec("a1", "Coffee one", "i love coffee", opts.tags ? ["eva-collab"] : undefined),
+ rec("a2", "Tea two", "i love tea", opts.tags ? ["loose-tag"] : undefined),
]),
);
return dir;
@@ -181,6 +221,96 @@ test("stdio: the legacy (2025) handshake still serves the same tools", async (t)
assert.match(firstText(res), /Tea two/);
});
+test("stdio: list_tags reports the vocabulary a site publishes", async (t) => {
+ const dir = await writeFixture({ tags: true });
+ t.after(() => rm(dir, { recursive: true, force: true }));
+ const s = await connect(dir, { mode: "auto" });
+ t.after(() => s.close());
+
+ const out = firstText(await s.client.callTool({ name: "list_tags", arguments: {} }));
+ assert.match(out, /2 curated tag\(s\)/);
+ // Grouped, with the label, this site's count and the per-channel breakdown.
+ assert.match(out, /### Eva/);
+ assert.match(out, /- eva-collab · Collab · 1 video\(s\) — chan-a 1/);
+ // An ungrouped tag still appears, under a bucket that says so.
+ assert.match(out, /### \(ungrouped\)/);
+ assert.match(out, /- loose-tag · Loose · 1 video\(s\) — chan-a 1/);
+ // And it names the filter it feeds, so the id never has to be guessed.
+ assert.match(out, /tags:\["eva-collab"\]/);
+});
+
+test("stdio: a site with no /tags.json says so instead of returning nothing", async (t) => {
+ const dir = await writeFixture();
+ t.after(() => rm(dir, { recursive: true, force: true }));
+ const s = await connect(dir, { mode: "auto" });
+ t.after(() => s.close());
+
+ const out = firstText(await s.client.callTool({ name: "list_tags", arguments: {} }));
+ assert.match(out, /publishes no curated tags/);
+ assert.match(out, /corpus spec 4/);
+
+ // …and a search that filters by a tag anyway is warned, not quietly empty.
+ // This is the pre-spec-4 path: correct (no record carries a tag) but
+ // indistinguishable from "searched and found nothing" without the warning.
+ const search = firstText(
+ await s.client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", tags: ["eva-collab"] },
+ }),
+ );
+ assert.match(search, /No matches for "coffee"/);
+ assert.match(search, /publishes no \/tags\.json/);
+ assert.match(search, /not evidence of absence/);
+});
+
+test("stdio: a tag the site does not publish is named, not silently empty", async (t) => {
+ const dir = await writeFixture({ tags: true });
+ t.after(() => rm(dir, { recursive: true, force: true }));
+ const s = await connect(dir, { mode: "auto" });
+ t.after(() => s.close());
+
+ const out = firstText(
+ await s.client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", tags: ["eva-collab", "not-a-real-tag"] },
+ }),
+ );
+ assert.match(out, /tag\(s\) this source does not publish: not-a-real-tag/);
+ // The filter it DID understand is echoed too.
+ assert.match(out, /tags: eva-collab OR not-a-real-tag/);
+});
+
+test("stdio: enumerate_matches filters by tag and names the filter", async (t) => {
+ const dir = await writeFixture({ tags: true });
+ t.after(() => rm(dir, { recursive: true, force: true }));
+ const s = await connect(dir, { mode: "auto" });
+ t.after(() => s.close());
+
+ // Both records match "i love"; only a1 carries eva-collab.
+ const all = firstText(
+ await s.client.callTool({
+ name: "enumerate_matches",
+ arguments: { query: "i love" },
+ }),
+ );
+ assert.match(all, /2 match\(es\)/);
+
+ const tagged = firstText(
+ await s.client.callTool({
+ name: "enumerate_matches",
+ arguments: { query: "i love", tags: ["eva-collab"] },
+ }),
+ );
+ assert.match(tagged, /1 match\(es\)/);
+ assert.match(tagged, /- a1 \| video/);
+ assert.doesNotMatch(tagged, /- a2 \| video/);
+ // The worklist's footer names the filter that produced it, so a count can
+ // never be quoted without the scope that made it.
+ assert.match(tagged, /filters — tags: eva-collab/);
+ // …and the complete-set line is still the honest one.
+ assert.match(tagged, /complete set: yes/);
+});
+
test("stdio: prompts are served on both eras", async (t) => {
const dir = await writeFixture();
t.after(() => rm(dir, { recursive: true, force: true }));
diff --git a/mcp/src/scanPlan.test.ts b/mcp/src/scanPlan.test.ts
@@ -50,6 +50,7 @@ function video(
isLivestream?: boolean;
deleted?: boolean;
text?: string;
+ curatedTags?: string[];
} = {},
): TranscriptDetail {
return {
@@ -66,6 +67,7 @@ function video(
webpageUrl: `https://www.youtube.com/watch?v=${id}`,
description: "",
tags: [],
+ ...(opts.curatedTags ? { curatedTags: opts.curatedTags } : {}),
cues: cues([10, opts.text ?? "they filed a lawsuit today"]),
};
}
@@ -160,6 +162,9 @@ function indexed(
uploadDate: rec.uploadDate,
isLivestream: rec.isLivestream === true,
ageRestricted: rec.ageRestricted === true,
+ ...(rec.curatedTags && rec.curatedTags.length > 0
+ ? { curatedTags: rec.curatedTags }
+ : {}),
},
];
}
@@ -380,6 +385,71 @@ test("a states filter reaches only the pages holding videos in those states", as
assert.equal(result.hits[0].videoId, "gone1");
});
+test("a curated-tag filter reaches only the pages holding tagged videos", async () => {
+ const p0 = [video("chan", "plain")];
+ const p1 = [video("chan", "collab", { curatedTags: ["eva-collab"] })];
+ const p2 = [video("chan", "chatty", { curatedTags: ["eva-in-chat"] })];
+ const source = new CountingSource(
+ { chan: [p0, p1, p2] },
+ new Map([indexed(p0[0]), indexed(p1[0]), indexed(p2[0])]),
+ );
+ const channels = await source.listChannels();
+ const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] };
+
+ const plan = await buildScanPlan(source, channels, filters);
+ assert.equal(plan.pruned, true);
+ assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]);
+ assert.equal(plan.unknownVideos, 0);
+
+ const result = await searchTranscripts(source, {
+ query: "lawsuit",
+ contentTypes: ["video"],
+ filters,
+ });
+ assert.deepEqual(source.reads, ["chan:1"]);
+ assert.equal(result.total, 1);
+ assert.equal(result.hits[0].videoId, "collab");
+});
+
+test("a video the index has no tags for still gets its page read", async () => {
+ // The pre-spec-4 case in miniature: the summaries set is older than the
+ // transcripts and has never heard of `ghost`. "I don't know about this
+ // video" must mean READ THE PAGE and let the record predicate decide — the
+ // planner's one invariant — not "it carries no tags, skip it". `plain` IS
+ // known and genuinely untagged, so it is correctly pruned.
+ const p0 = [video("chan", "plain")];
+ const p1 = [video("chan", "ghost", { curatedTags: ["eva-collab"] })];
+ const source = new CountingSource(
+ { chan: [p0, p1] },
+ new Map([indexed(p0[0])]),
+ );
+ const channels = await source.listChannels();
+ const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] };
+
+ const plan = await buildScanPlan(source, channels, filters);
+ assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]);
+ assert.equal(plan.unknownVideos, 1);
+
+ const result = await searchTranscripts(source, {
+ query: "lawsuit",
+ contentTypes: ["video"],
+ filters,
+ });
+ assert.deepEqual(source.reads, ["chan:1"]);
+ assert.equal(result.total, 1, "the record the index missed still matched");
+});
+
+test("an empty curated-tag list is not a filter and does not plan", async () => {
+ const p0 = [video("chan", "a1")];
+ const source = new CountingSource({ chan: [p0] }, new Map());
+ const plan = await buildScanPlan(source, await source.listChannels(), {
+ ...ALL_STATES,
+ curatedTags: [],
+ });
+ assert.equal(plan.pruned, false);
+ assert.equal(plan.pagesPlanned, 1);
+});
+
// ─── 3. duplicate collapsing ───
test("a recording mirrored across channels is counted once and its mirror named", async () => {
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -107,10 +107,16 @@ const CHAN_A: TranscriptDetail[] = [
vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), {
description: "a pour over brewing guide",
tags: ["espresso", "beans"],
+ // CURATED tags (lib/curatedTags.ts) — the operator's vocabulary, which is
+ // a different field from the yt-dlp keywords one line above and is
+ // deliberately set on the same record so a test that confuses the two
+ // fails loudly.
+ curatedTags: ["eva-collab"],
}),
vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])),
vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), {
isLivestream: true,
+ curatedTags: ["eva-in-chat"],
}),
vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), {
ageRestricted: true,
@@ -623,6 +629,34 @@ test("spec filter fa: keep only age-restricted → a4", async () => {
assert.deepEqual(ids, ["a4"]);
});
+test("spec filter tg: a curated tag keeps only the videos carrying it", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, curatedTags: ["eva-collab"] },
+ });
+ assert.deepEqual(ids, ["a1"]);
+});
+
+test("spec filter tg: several tags are ORed, never ANDed", async () => {
+ // a1 carries eva-collab, a3 carries eva-in-chat and neither carries both:
+ // an AND here would return nothing, which is the bug this pins.
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, curatedTags: ["eva-collab", "eva-in-chat"] },
+ });
+ assert.deepEqual(ids, ["a1", "a3"]);
+});
+
+test("spec filter tg: an untagged record carries none and passes none", async () => {
+ // Every record on a site built before corpus spec 4 looks like this, so the
+ // pre-spec-4 path is "an honest empty result", not "everything matches".
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, curatedTags: ["never-assigned"] },
+ });
+ assert.deepEqual(ids, []);
+});
+
test("spec filter fav: available-only drops every missing state", async () => {
const src = new StubSource();
const ids = await specIds(src, leafTree("coffee", "transcripts"), {
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -11,6 +11,8 @@ import type { VideoStat } from "yt-dlp-transcript-common/lib/stats";
import { momentUrl, momentBaseUrl } from "yt-dlp-transcript-common/lib/momentUrl";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
+import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags";
+import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags";
import {
VIDEO_STATES,
isVideoState,
@@ -215,6 +217,20 @@ const SEARCH_FILTER_ARGS = {
"state. ('maybe_missing' = fell out of the channel listing but was never " +
"individually confirmed.)",
},
+ tags: {
+ type: "array",
+ items: { type: "string" },
+ description:
+ "Keep only videos carrying one of these CURATED TAGS — the operator's " +
+ "cross-channel vocabulary (e.g. tags:['eva-collab'] is 'every stream " +
+ "where she is on mic', on any channel, including channels that are not " +
+ "hers). ORed: naming two tags keeps a video with EITHER. Call " +
+ "`list_tags` for the ids this source publishes and how many videos each " +
+ "one has — do not guess an id. NOT the yt-dlp keywords searched by " +
+ "scopes:['tags']; those are metadata from the uploader, these are " +
+ "curation. A source that publishes no tags says so in the footer rather " +
+ "than returning a quiet zero.",
+ },
date_from: {
type: "string",
description: "Keep only videos uploaded on or after this date (YYYYMMDD).",
@@ -306,6 +322,27 @@ export const TOOLS: Tool[] = [
},
},
{
+ name: "list_tags",
+ description:
+ "List the CURATED TAGS this archive publishes — the operator's " +
+ "cross-channel vocabulary for marking individual videos (e.g. " +
+ "'eva-collab' = she is on mic), which is what `tags` on " +
+ "search_transcripts / enumerate_matches filters by. Returns each tag's " +
+ "id, label, group, how many videos carry it ON THIS SOURCE, and the " +
+ "per-channel breakdown — so a tag can be scoped, counted and cited " +
+ "without guessing an id. These are NOT the yt-dlp keywords searched by " +
+ "scopes:['tags']: those come from the uploader's metadata, these are " +
+ "curation applied after the fact and cut across channels. A source that " +
+ "publishes none (nothing curated yet, or a site built before corpus " +
+ "spec 4) says so plainly rather than returning an empty list that reads " +
+ "like an answer.",
+ inputSchema: {
+ type: "object",
+ properties: { ...SOURCE_ARG },
+ additionalProperties: false,
+ },
+ },
+ {
name: "search_transcripts",
description:
"Search the archive for a term or phrase. The corpus holds video " +
@@ -934,6 +971,8 @@ export function createServer(
switch (name) {
case "list_channels":
return handleListChannels(source, args);
+ case "list_tags":
+ return handleListTags(source);
case "search_transcripts":
return handleSearch(source, args);
case "enumerate_matches":
@@ -1087,6 +1126,94 @@ async function handleListChannels(
);
}
+// The curated-tag vocabulary a source publishes, with its own counts.
+//
+// The empty case is answered in WORDS, not with an empty list. An agent that
+// gets `[]` back reads it as "no tags here" and moves on; what it actually
+// needs to know is that a tag filter against this source will match nothing and
+// why — the site publishes none, or it was built before the corpus spec that
+// carries them. That is the same failure the footer warning below exists to
+// stop, said once up front.
+async function handleListTags(source: ShardSource): Promise<ToolResult> {
+ const tags = await loadSourceTags(source);
+ if (tags.length === 0) {
+ return text(
+ `${source.label} publishes no curated tags.\n\n` +
+ `Either nothing has been tagged for this site yet, or it was built ` +
+ `before curated tags existed (corpus spec 4) — /tags.json is absent. ` +
+ `A \`tags\` filter against this source would match nothing, so do not ` +
+ `use one here; scope with channel/group/date instead.`,
+ );
+ }
+
+ const groups = groupPublishedTags(tags);
+ const sections = groups.map((g) => {
+ const head = `### ${g.label || "(ungrouped)"}`;
+ const lines = g.tags.map((t) => {
+ const per = Object.entries(t.channels)
+ .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
+ .map(([slug, n]) => `${slug} ${n}`)
+ .join(", ");
+ return (
+ `- ${t.id} · ${t.label} · ${t.count} video(s)` +
+ (per ? ` — ${per}` : "")
+ );
+ });
+ return `${head}\n${lines.join("\n")}`;
+ });
+
+ return text(
+ `${tags.length} curated tag(s) in ${source.label}:\n\n` +
+ `${sections.join("\n\n")}\n\n` +
+ `(filter with tags:["${tags[0].id}"] on search_transcripts or ` +
+ `enumerate_matches; several tags are ORed. A video may carry more than ` +
+ `one tag, so these counts do not sum to a video count.)`,
+ );
+}
+
+// Every tag read goes through here so "this source cannot report tags at all"
+// (an in-memory stub, which has no such concept) and "this source publishes
+// none" land in the same place, as the same empty list.
+async function loadSourceTags(source: ShardSource): Promise<PublishedTag[]> {
+ if (typeof source.loadTags !== "function") return [];
+ try {
+ return await source.loadTags();
+ } catch {
+ return [];
+ }
+}
+
+// A tag filter against a source that publishes no /tags.json matches nothing.
+// That is CORRECT — a record with no curated tags carries none — but a bare
+// zero is indistinguishable from "searched, found nothing", so it is named.
+//
+// The scan itself is unaffected and stays honest: the page planner only ever
+// PRUNES, and a video its index has never heard of gets its page read anyway
+// (search.ts's load-bearing invariant), so this is a reporting fix, never a
+// coverage one.
+async function describeTagCoverage(
+ source: ShardSource,
+ parsed: ParsedSearchArgs,
+): Promise<string> {
+ const asked = parsed.filters?.curatedTags ?? [];
+ if (asked.length === 0) return "";
+ const published = await loadSourceTags(source);
+ if (published.length === 0) {
+ return (
+ `⚠ this source publishes no /tags.json (nothing curated, or built ` +
+ `before corpus spec 4) — a tag filter matches NOTHING here, so this ` +
+ `result is not evidence of absence`
+ );
+ }
+ const known = new Set(published.map((t) => t.id));
+ const unknown = asked.filter((t) => !known.has(t));
+ if (unknown.length === 0) return "";
+ return (
+ `⚠ tag(s) this source does not publish: ${unknown.join(", ")} — they ` +
+ `match nothing here; call list_tags for the ids it has`
+ );
+}
+
async function handleSearch(
source: ShardSource,
args: Record<string, unknown>,
@@ -1119,6 +1246,7 @@ async function handleSearch(
const scopeNote = describeScope(result.selection);
const postsNote = describePostsPass(result);
const filterNote = describeSearchFilters(parsed);
+ const tagNote = await describeTagCoverage(source, parsed);
const prunedNote = describeCoverage(result);
const dupNote = describeDuplicates(result);
const rangeStart = result.total === 0 ? 0 : result.offset + 1;
@@ -1136,6 +1264,7 @@ async function handleSearch(
(filterNote ? `; ${filterNote}` : "") +
(postsNote ? `; ${postsNote}` : "") +
(aliasNote ? `; ${aliasNote}` : "") +
+ (tagNote ? `; ${tagNote}` : "") +
parsed.warnings.map((w) => `; ⚠ ${w}`).join("") +
")";
@@ -1321,6 +1450,7 @@ async function handleEnumerateMatches(
const aliasNote = describeFiredAliases(result.firedAliases);
const postsNote = describePostsPass(result);
const filterNote = describeSearchFilters(parsed);
+ const tagNote = await describeTagCoverage(source, parsed);
const prunedNote = describeCoverage(result);
const dupNote = describeDuplicates(result);
@@ -1335,6 +1465,7 @@ async function handleEnumerateMatches(
(filterNote ? `; ${filterNote}` : "") +
(postsNote ? `; ${postsNote}` : "") +
(aliasNote ? `; ${aliasNote}` : "") +
+ (tagNote ? `; ${tagNote}` : "") +
parsed.warnings.map((w) => `; ⚠ ${w}`).join("") +
")";
@@ -1426,12 +1557,35 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs {
const from = /^\d{8}$/.test(dateFrom) ? dateFrom : undefined;
const to = /^\d{8}$/.test(dateTo) ? dateTo : undefined;
+ // Curated tags. A malformed id is REPORTED, never silently dropped: dropping
+ // the only tag in the list would widen the search back to the whole corpus
+ // and read as a legitimate result, which is the same failure a typo'd state
+ // would cause.
+ const rawTags = strArray(args.tags);
+ let tags: string[] | undefined;
+ if (rawTags) {
+ const normalized = rawTags.map((t) => t.trim().toLowerCase());
+ tags = normalized.filter((t) => isTagId(t));
+ const bad = normalized.filter((t) => !isTagId(t));
+ if (bad.length > 0) {
+ warnings.push(
+ `not a tag id, ignored: ${bad.join(", ")} (ids are lowercase, ` +
+ `[a-z0-9._-]; call list_tags)`,
+ );
+ }
+ if (tags.length === 0) {
+ tags = undefined;
+ warnings.push("no valid tags given — the tag filter was NOT applied");
+ }
+ }
+
const anyFilter =
states !== undefined ||
mediaType !== undefined ||
age !== undefined ||
from !== undefined ||
- to !== undefined;
+ to !== undefined ||
+ tags !== undefined;
const filters: SearchFilters | null = anyFilter
? {
@@ -1442,6 +1596,7 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs {
states: new Set<VideoState>(states ?? VIDEO_STATES),
...(from ? { dateFrom: from } : {}),
...(to ? { dateTo: to } : {}),
+ ...(tags ? { curatedTags: tags } : {}),
}
: null;
@@ -1490,6 +1645,9 @@ function describeSearchFilters(p: ParsedSearchArgs): string {
if (f.dateFrom || f.dateTo) {
parts.push(`uploaded ${f.dateFrom ?? "…"}–${f.dateTo ?? "…"}`);
}
+ if (f.curatedTags && f.curatedTags.length > 0) {
+ parts.push(`tags: ${f.curatedTags.join(" OR ")}`);
+ }
}
if (p.exclude && p.exclude.length > 0) {
parts.push(`excluding ${p.exclude.map((e) => `"${e}"`).join(", ")}`);
@@ -2190,6 +2348,17 @@ async function handleResolveSource(
` reachable: yes — ${channels.length} channel(s)` +
(groups.length > 0 ? `, ${groups.length} group(s)` : ""),
);
+ // Named here so a caller learns whether a `tags` filter is even available
+ // BEFORE it writes one and reads the empty result as an answer. The two
+ // states are different and both are said: some tags, or none at all.
+ const tags = await loadSourceTags(resolved.source);
+ lines.push(
+ tags.length > 0
+ ? ` curated tags: ${tags.length} — ` +
+ tags.map((t) => `${t.id} (${t.count})`).join(", ") +
+ ` · list_tags for the breakdown`
+ : ` curated tags: none published (a \`tags\` filter matches nothing here)`,
+ );
} catch (e) {
return errorText(
`${lines.join("\n")}\n reachable: NO — ${(e as Error).message}`,
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -139,6 +139,17 @@ track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
**`report` already means three things**: `reportDebouncePreset` in settings, the
`refresh-report` job kind, and `/ask`'s `ReportPanel`. Use `feedback` / `review` instead.
+**`tags` means three unrelated things, and the record field is `curatedTags`.**
+
+| Which | Where | What it is |
+| --- | --- | --- |
+| yt-dlp keywords | `common/lib/transcripts.ts:21` — `TranscriptSummary.tags` | The platform's own keywords from `metadata.info.json`, searchable via the `"tags"` scope. The UI label for this scope is being changed to **"Keywords"** (`QueryLeafView.tsx:46,156-159`); the code token, the URL and the MCP `scopes` enum stay `"tags"`. |
+| AI digest topic tags | `common/lib/digests.ts` — `VideoDigest.tags` | Model-generated topics on a digested video. |
+| **Curated tags** | `common/lib/curatedTags.ts`; record field **`curatedTags`** | The operator's cross-channel vocabulary ("every stream where X collabs"), published at `/tags.json`. |
+
+Never name the curated field `tags`. Never assume a `tags.json` is the keyword list:
+`transcripts/tags.json` and `sites/<id>/tags.json` are curated tags.
+
---
## Reusable helpers (do not rewrite these)
@@ -160,6 +171,9 @@ track. The exclusion list in `readSubTracks` (`:238-256`) is hardcoded. Use
| ENOENT → friendly message | `common/social/xGalleryDlFetcher.ts:207-213` | `/ENOENT/.test(message) ? "X not found (set X_BIN)" : message`. |
| ETA estimator | `editor/app/jobs/active/buildActiveJobs.ts:38` | `computeEtaSeconds`. |
| Binary search over cue starts | `common/components/TranscriptModal.tsx:473-491` | `findActiveIndex`. |
+| Curated-tag model (pure) | `common/lib/curatedTags.ts` | `sanitizeTagsConfig` / `mergeTagDefs` / `effectiveTagsFor` / `compileTagRules` + `evaluateCompiledRules`. Compile ONCE per build, evaluate per video. |
+| Curated-tag persistence | `common/lib/curatedTagsStore.ts:198` | `applyTagAssignments` — the ONE write path for every tag writer (editor, ops, umtool); one atomic write per call, and it refuses to pin a tag the vocabulary does not define. |
+| A chat cue's author | `common/lib/curatedTags.ts:495` — `chatAuthorOf` | The `<author>: ` prefix; there is no `author` field on `Cue`. |
---
@@ -187,9 +201,9 @@ Exposed to the settings form via `listTranscriptionApps()` (`:272`), consumed at
| Constant | Where | Value |
| --- | --- | --- |
-| `SCHEMA_VERSION` | `buildIndex.ts:116` | 12 |
+| `SCHEMA_VERSION` | `buildIndex.ts:165` | **13** — and curated tags deliberately did NOT bump it (2026-09-21); see below |
| `CUES_FILE_VERSION` | `normalizeTranscript.ts:33` | 2 |
-| `CORPUS_SPEC_VERSION` | `common/lib/corpus.ts:16` | 2 |
+| `CORPUS_SPEC_VERSION` | `common/lib/archive/contract.ts:35` (re-exported `corpus.ts:59`) | **4** (3 → 4 for `/tags.json`, 2026-09-21) |
| `MANIFEST_VERSION` (site summaries) | `common/lib/manifest.ts:36` | 3 |
| `TRANSCRIPTS_MANIFEST_VERSION` | `manifest.ts:43` | 1 |
| `SUBS_MANIFEST_VERSION` | `manifest.ts:59` | 4 |
@@ -210,6 +224,75 @@ three times (`:667` transcripts, `:675` subs, `:686` posts); it passes
Transcript shards ship the **full `cues` array inline** — `buildIndex.ts:830-832`.
+### Curated tags invalidate through `meta`, not through mtimes (2026-09-21)
+
+`curatedTags` is DERIVED at build time — `(rule hits ∪ pins) − suppressions` — and **rule
+hits are never persisted**. So nothing in the per-video mtime diff can see that a rule was
+edited or a video pinned, and the mechanism is four keys in the existing `meta` sub-DB
+(`common/controller/curatedTagsIndex.ts:44-52`), checked **unconditionally on every build**:
+
+| Key | Covers | Deliberately excludes |
+| --- | --- | --- |
+| `curatedRulesHash` | every def's id + `order`, and every **enabled** rule's kind, pattern, channel scope and date range | label / colour / group / hidden — relabelling a tag must not re-page 30,000 videos |
+| `curatedAssignHash` | every pin and suppression | — |
+| `curatedAssignSigs` | per-key `manual|suppressed` signature | not a hash: it is what makes an assignment-only edit re-derive just the videos that moved |
+| `curatedPagesPending` | "records were re-derived, the shared pages have not been written yet" | set when the hashes are recorded WITH changes, cleared only after the shared page build — without it a Ctrl-C between the two would store the hashes and leave the shards stale for ever, since the next build would compare equal and dirty nothing |
+
+`reapplyCuratedTags` (`:279`) runs after the mtime diff: rules changed → every video is a
+candidate (cue reads gated per video by channel/date scope); assignments changed → only the
+symmetric difference of the signatures, located by a key-only `byChannel` range scan per
+affected channel. It re-derives out of LMDB — **no video directory is read twice** — and
+whatever it changes flips `sharedNeedsBuild`, because the page writer's sha1 skip is what
+then leaves the untouched pages untouched — and `curatedPagesPending` is what covers the
+build that was interrupted between the two. The per-site fingerprint (`buildIndex.ts:1765`)
+includes both hashes, or a tag-only edit would leave a site reporting "up to date" with
+yesterday's counts. **No new sub-DB**, so the `clearAsync()` enumeration is unchanged.
+
+**And no `SCHEMA_VERSION` bump — on purpose.** Every prior bump existed because something
+would otherwise never be derived at all (v13 populates a new sub-DB; nothing re-derives it
+lazily). Curated tags re-derive lazily *by design*: `curatedRulesHash` is absent on an index
+built before them, so it can never equal the current hash and the first build re-derives
+every record on its own — ~80 s across 30k videos, out of LMDB, and **zero page writes on an
+untagged corpus** because an empty derivation leaves each summary byte-identical (tested:
+"an UNTAGGED corpus cold-starts to zero changes"). A bump would instead wipe the cache and
+re-read 76k video directories to reach exactly the same state.
+
+**The site layer is presentation-only.** `sites/<id>/tags.json` relabels, recolours,
+reorders, hides and may add site-only tags; it can never delete a corpus rule or an
+assignment. A site-layer RULE is accepted by `mergeTagDefs` (harmless on a published def)
+but **never tags a record**: rules are evaluated once at index time over records SHARED by
+every site carrying the channel, so there is no per-site record for a per-site hit to live
+in. Rules bind only from `transcripts/tags.json` — promote one there to make it fire. The
+editor therefore offers no site-side rule editor.
+
+**A chat cue's author is a string prefix, not a field.** `common/lib/liveChat.ts:100-105`
+emits every live-chat cue as ``text: author ? `${author}: ${text}` : text`` (`:104`) — `Cue` is
+`{start, end, text}` (`vtt.ts:1`) and has no `author`. So a `chat-author` rule matches
+`chatAuthorOf(text)` (`curatedTags.ts:495`): everything before the FIRST `": "`, and `null`
+when there is none (a cue with no author can never match). Anything that wants the author
+of a chat line must use this, not a second copy of the split.
+
+### Measured: what a curated rule costs (2026-09-21, real corpus)
+
+`legal-mindset`, 518 videos / 1.37 M caption cues / 1.47 M live-chat cues, the three seed
+rules (metadata + chat-author + caption), run offline against the production LMDB (warm):
+
+| Rules | Wall | LMDB decode | Regex |
+| --- | --- | --- | --- |
+| metadata only | 18 ms | 0 | 4 ms |
+| chat-author only | 848 ms | 727 ms | 110 ms |
+| caption only | 421 ms | 302 ms | 109 ms |
+| all three | 1.28 s | 1.03 s | 232 ms |
+
+**The cost is the cue decode, not the regex** (4:1). A metadata-only vocabulary is free; a
+cue-backed rule costs about 2.5 ms per video with cues, so a whole-corpus re-derive at
+Jeralyzer's scale (30,886 videos) projects to ~80 s — a rule edit, not a rebuild. Scoping a
+rule with `channels`/`dateFrom` skips the decode entirely for everything out of scope.
+
+Hits on that channel: 7 `eva-collab` (metadata), 96 `eva-in-chat` (chat author), 1
+`eva-topic` (caption `\belf ?pire\b|elfpyre|legal loli`) — the caption number is a pattern
+problem, not a plumbing one: a spoken name rarely appears spelled that way in ASR output.
+
---
## Digest bake-off (measured 2026-07-26)
diff --git a/plans/curated-tags.md b/plans/curated-tags.md
@@ -21,7 +21,9 @@ with her in chat.
(re-evaluated at index build; operator pins/suppresses exceptions), by **import from a umtool project**.
2. **One tag per kind**: `eva-collab` (on mic), `eva-in-chat`, `eva-topic`; filter is OR across chips.
3. **Vocabulary both corpus-wide and per site, like aliases** — corpus file authoritative for
- definitions AND assignments; the site file is presentation (relabel/hide/order/colour) + site-only rules.
+ definitions AND assignments; the site file is presentation ONLY (relabel/hide/order/colour). **Amended
+ 2026-09-21 during S1**: a site-layer rule can never tag a record — a record is shared by every site carrying
+ its channel — so rules bind only from the corpus file and the site tab offers no rule editor.
4. **Surfaces — all four**: export lists (channel + all-videos), export search, MCP filter, editor lists.
5. **Universal + flexible, three writers**: free-form tag ids for any subject, optional `group` so chips
read "Eva: Collab · In chat · Discussed"; operator UI, `pnpm ops` (agents; MCP stays read-only per
@@ -90,7 +92,8 @@ scope and date range are applied before compiling; each kind compiles once per b
video from the `cues`/`subs` sub-DBs (no disk re-read). Returns the changed `indexKey`s, which are unioned
into the dirty-channel set so transcript/subs/summaries pages rewrite — `common/lib/channelSignature.ts`
hashes only mtimes (FACTS ~:203-207), so this explicit dirtying is what keeps compose from skipping them.
- `SCHEMA_VERSION` 13 → 14 (`:143`). Build log prints `curated tags: rules <hash8> (changed), re-derived N`.
+ `SCHEMA_VERSION` stays **13** (amended during S1: an absent `curatedRulesHash` can never match, so the first
+ build re-derives every record from LMDB in ~80 s; a bump would re-read 76k video dirs for the same state). Build log prints `curated tags: rules <hash8> (changed), re-derived N`.
- **Counts**: while streaming `sums` into per-site summaries pages (`:1712`) accumulate per-tag totals and
per-channel breakdown → `tag-counts.json` in `siteIndexDir(paths, siteId)` (`common/lib/site.ts:138`).
- **Compose** (`common/bin/compose-site.ts:744-753`, beside aliases): `effectiveSiteTags(paths, siteId)` +
@@ -109,16 +112,16 @@ scope and date range are applied before compiling; each kind compiles once per b
`export/e2e/fixtures/data.ts` emits `curatedTags` on two summaries and a `/tags.json` fixture. Merge to
main immediately — S2 and S3 branch from it.
2. `common/lib/curatedTagsStore.ts` (mirror `common/lib/aliasesStore.ts`): `readGlobalTags/writeGlobalTags/
- readSiteTags/writeSiteTags/effectiveSiteTags` + **`applyTagAssignments(paths, {op: add|remove|replace, tag,
+ readSiteTags/writeSiteTags/effectiveSiteTags` + **`applyTagAssignments(paths, {op: add|remove|suppress|unsuppress, tag,
videos, source, at})`**; `paths.globalTagsFile` (`common/lib/paths.ts:105,219`), `siteTagsFile`
(`common/lib/site.ts:132`); tests: sanitize, layering, both-lists conflict, provenance.
3. `curatedTagsIndex.ts` + `buildIndex.ts` hook at `:697`, meta hashes, `reapplyCuratedTags`,
- `SCHEMA_VERSION` 14, per-site counts at `:1712`; `toDisplaySummary`. Measure rule cost on legal-mindset
+ NO schema bump (see Derivation), per-site counts at `:1712`; `toDisplaySummary`. Measure rule cost on legal-mindset
(478 chat streams) and record it in FACTS.
4. Contract + compose: `contract.ts` spec 4 + `ROOT_FILES`; `corpus.ts:204,:320`; `compose-site.ts:744-753`
writes `/tags.json`; `corpus.test.ts`.
5. `plans/FACTS.md` (the three meanings of "tags" in the naming-hazards table; chat-author prefix contract;
- the two meta hashes; SCHEMA_VERSION 14) + `AGENTS.md` (tags.json is curated data; never hand-edit).
+ the two meta hashes; SCHEMA_VERSION unchanged) + `AGENTS.md` (tags.json is curated data; never hand-edit).
### S2 — export UI + MCP (branch `tags/site`, needs only S1.1)
6. `common/lib/search/evalTree.ts`: `curatedTags?` on `SearchFilters` (`:311`) and `FilterableRecord`
@@ -146,7 +149,8 @@ scope and date range are applied before compiling; each kind compiles once per b
`previewTagRuleAction` = dry-run `evaluateTagRules` over LMDB `sums/cues/subs`, listing title/date/channel
with Pin all / Pin / Unpin, `applyTagAssignmentsAction`) — pattern `editor/app/sites/[siteId]/aliases/`.
Nav entry (twelve → thirteen; `editor/scripts/measure-nav.mjs`).
-13. Per-site overlay tab `editor/app/sites/[siteId]/tags/page.tsx` (presentation fields + site-only rules).
+13. Per-site overlay tab `editor/app/sites/[siteId]/tags/page.tsx` (presentation fields only — no rule editor,
+ see decision 3 as amended).
14. Video page `videos/[id]/components/TagsPanel.tsx` beside `OperationPanel` (`page.tsx:251`);
`toggleVideoTagAction` in `videoActions.ts` (shows each tag's provenance).
15. `VideoListPane.tsx`: `tag`/`untag` in `BULK_ACTION_OPTIONS` (`:38-49`) with a tag picker via