commit 3d35ffa94ca2823d02739f698612486f58f784d9
parent a8c66c780b41c336dabbea5d5ec6c6f8b65c0969
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 13:43:16 -0400
tags review: an interrupted build owes the pages a rewrite, and must know it
The hole the review found: reapplyCuratedTags records the meta hashes BEFORE
the shared page build runs. Ctrl-C in between and the next build compares the
hashes equal, re-derives nothing, never flips sharedNeedsBuild — and the shards
ship stale curatedTags for ever, while the per-site summaries (whose
fingerprint is written after their own build) quietly refresh. The two halves
of the archive disagree, permanently, with nothing to notice it.
`curatedPagesPending` closes it: set whenever the hashes are recorded with
changes, flushed before the page build begins, OR'd into sharedNeedsBuild, and
cleared only once BOTH shared trees are written. A schema bump records it too —
an interrupt mid-rebuild is no different.
Proved by a new integration test over a temp corpus driving the REAL buildIndex
and the real compose script: build an untagged corpus (no curatedTags key
anywhere), add a rule touching no video and no mtime, rebuild, and assert the
tag reached BOTH the channel's transcripts/page-0000.json and the site's
summaries/page-0000.json plus tag-counts.json; compose and assert /tags.json
and corpus.json's spec-4 tags pointer; drop the vocabulary and assert the file
and the pointer are removed rather than emptied; then reconstruct the interrupt
exactly (re-derive through LMDB, hashes recorded, flag set, pages untouched)
and assert the next build repairs the shards and settles the flag.
Also from the review: reapplyCuratedTags returns `changedCount` and collects
the keys only when asked (a first build over 30k records was holding 30k tuples
to answer a boolean); the duplicated scope predicate is gone, both callers now
ask ruleApplies; toDisplaySummary gets its own test file for the
omitted-when-empty contract; FACTS anchors re-checked line by line after these
edits.
Co-Authored-By: Claude Opus <noreply@anthropic.com>
Diffstat:
6 files changed, 508 insertions(+), 52 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -131,6 +131,7 @@ import { loadDigest, loadDigestOverrides } from "../lib/digest-server";
import {
TAG_COUNTS_FILENAME,
applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
createTagCounter,
deriveCuratedTags,
loadCuratedTagsRuntime,
@@ -894,7 +895,15 @@ export async function buildIndex({
allFresh: schemaBumped,
log,
});
- if (curatedReapply.changed.length > 0) sharedNeedsBuild = true;
+ // `pagesPending` also covers a PREVIOUS build that re-derived records and was
+ // then interrupted before writing the pages: its hashes are already stored,
+ // so without the flag nothing would ever ask for those shards again and they
+ // would ship stale curatedTags for ever.
+ if (curatedReapply.pagesPending) sharedNeedsBuild = true;
+ // Commit the flag (and the re-derived summaries) BEFORE the page build, so an
+ // interrupt during it finds the debt recorded rather than lost in a pending
+ // write batch.
+ await meta.flushed;
await sums.flushed;
await cues.flushed;
@@ -1357,6 +1366,13 @@ export async function buildIndex({
log(`Shared sub pages up to date; ${channelStats.size} channels with subs.`);
}
+ // Both shared trees are now written (or were already current), so the
+ // curated-tag page debt is settled. Deliberately AFTER the page build and not
+ // beside the hashes: an interrupt anywhere above must leave the flag standing
+ // so the next build rewrites the shards.
+ clearCuratedPagesPending(meta);
+ await meta.flushed;
+
// ---------------------------------------------------------------------------
// The social-post corpus: a parallel per-channel page tree beside transcripts
// and subs. Social channels have no `data/` dir and so never appear in the
diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts
@@ -0,0 +1,287 @@
+// Integration: a curated-tag edit reaching the published files, through the
+// REAL buildIndex and the real compose script, over a temp corpus.
+//
+// The unit tests around reapplyCuratedTags prove the derivation. This proves
+// the thing that actually breaks in production: a tag edit moves no video and
+// no mtime, so every incremental short-circuit in the build has a chance to
+// decide nothing happened and ship yesterday's shards.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/curatedTagsBuild.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { execFile } from "node:child_process";
+import { mkdtempSync, writeFileSync, mkdirSync, readFileSync, existsSync } from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import { promisify } from "node:util";
+import { open } from "lmdb";
+
+// getPaths() is lazy and cached, and nothing above calls it at import time, so
+// pointing the whole path graph at a temp root here is enough to isolate this
+// file's process from the real corpus. Every path the build touches is derived
+// from these three.
+const ROOT = mkdtempSync(path.join(tmpdir(), "curated-tags-build-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public");
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { writeGlobalTags } = await import("../lib/curatedTagsStore");
+const {
+ curatedTagsRuntime,
+ reapplyCuratedTags,
+ META_PAGES_PENDING,
+} = await import("./curatedTagsIndex");
+
+const paths = getPaths();
+const REPO = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..");
+const exec = promisify(execFile);
+
+// compose-site.ts is a SCRIPT (it runs main() on import), so it is exercised
+// the way the build runs it: as a child process with SITE_ID, inheriting this
+// file's temp-root env. `.bin/tsx` is a shell wrapper, so the CLI entry is
+// invoked directly.
+const composeSite = () =>
+ exec(
+ process.execPath,
+ [
+ path.join(REPO, "node_modules", "tsx", "dist", "cli.mjs"),
+ path.join(REPO, "common", "bin", "compose-site.ts"),
+ ],
+ { env: { ...process.env, SITE_ID: SITE }, cwd: REPO },
+ );
+
+const CHANNEL = "test-channel";
+const SITE = "testsite";
+const HIT = "vid-hit";
+const MISS = "vid-miss";
+
+function writeJson(file: string, value: unknown): void {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+}
+
+function videoDir(id: string): string {
+ return path.join(paths.channelsDir, CHANNEL, "data", id);
+}
+
+function seedCorpus(): void {
+ writeFileSync(paths.settingsFile, JSON.stringify({}));
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Test Channel",
+ url: "https://www.youtube.com/@example/videos",
+ });
+ const video = (id: string, title: string, uploadDate: string) => {
+ writeJson(path.join(videoDir(id), "metadata.info.json"), {
+ id,
+ title,
+ channel: "Test Channel",
+ upload_date: uploadDate,
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ writeFileSync(
+ path.join(videoDir(id), "transcript.en.vtt"),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nA line of transcript.\n",
+ );
+ };
+ video(HIT, "Stream with Elfpire Eva", "20260102");
+ video(MISS, "An ordinary stream", "20260101");
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+}
+
+const build = () => buildIndex({ paths, onLog: () => {} });
+
+function transcriptPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSharedTranscriptsDir, CHANNEL, "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function summariesPage(): { id: string; curatedTags?: string[] }[] {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "summaries", "page-0000.json"),
+ "utf8",
+ ),
+ );
+}
+
+function tagCounts(): { tags: Record<string, { count: number; channels: Record<string, number> }> } {
+ return JSON.parse(
+ readFileSync(
+ path.join(paths.exportSitesIndexDir, SITE, "tag-counts.json"),
+ "utf8",
+ ),
+ );
+}
+
+const recordIn = (page: { id: string; curatedTags?: string[] }[], id: string) =>
+ page.find((r) => r.id === id)!;
+
+const EVA = {
+ version: 1,
+ tags: [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ rules: [
+ {
+ id: "meta",
+ kind: "metadata" as const,
+ pattern: "elfpire",
+ enabled: true,
+ },
+ ],
+ },
+ ],
+ assignments: {},
+};
+
+// The whole slice, in one sequence, because each step's fixture is the previous
+// step's output. node:test runs top-level tests in order within a file.
+
+test("build 1: an untagged corpus ships no curatedTags key at all", async () => {
+ seedCorpus();
+ const res = await build();
+ assert.equal(res.totalCount, 2);
+ for (const page of [transcriptPage(), summariesPage()]) {
+ for (const record of page) {
+ assert.equal(
+ "curatedTags" in record,
+ false,
+ "omitted-when-empty, or every untagged corpus re-pages on upgrade",
+ );
+ }
+ }
+ assert.deepEqual(tagCounts().tags, {});
+});
+
+test("build 2: a rule edit with NO mtime change reaches the shards AND the site pages", async () => {
+ // The only thing that changes between build 1 and build 2. No video
+ // directory is touched, so the mtime diff sees +0 ~0 -0.
+ writeGlobalTags(paths, EVA);
+
+ const res = await build();
+ assert.equal(res.added, 0);
+ assert.equal(res.changed, 0);
+ assert.equal(res.removed, 0);
+
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(transcriptPage(), MISS), false);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.equal("curatedTags" in recordIn(summariesPage(), MISS), false);
+ assert.deepEqual(tagCounts().tags["eva-collab"], {
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ });
+});
+
+test("compose 1: the site publishes /tags.json and corpus.json points at it", async () => {
+ await composeSite();
+ const tagsFile = path.join(paths.exportPublicDir, "tags.json");
+ const published = JSON.parse(readFileSync(tagsFile, "utf8"));
+ assert.deepEqual(published.tags, [
+ {
+ id: "eva-collab",
+ label: "Collab",
+ group: "eva",
+ groupLabel: "Eva",
+ order: 1,
+ count: 1,
+ channels: { [CHANNEL]: 1 },
+ },
+ ]);
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.spec, 4);
+ assert.equal(corpus.tags.videoField, "curatedTags");
+ assert.match(corpus.tags.url, /\/tags\.json$/);
+ assert.match(
+ readFileSync(path.join(paths.exportPublicDir, "llms.txt"), "utf8"),
+ /tags\.json/,
+ );
+});
+
+test("build 3 + compose 2: dropping the vocabulary removes the file and the pointer", async () => {
+ writeGlobalTags(paths, { version: 1, tags: [], assignments: {} });
+ await build();
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+ assert.deepEqual(tagCounts().tags, {});
+
+ await composeSite();
+ assert.equal(
+ existsSync(path.join(paths.exportPublicDir, "tags.json")),
+ false,
+ "a site with nothing to say must stop serving the file, not serve an empty one",
+ );
+ const corpus = JSON.parse(
+ readFileSync(path.join(paths.exportPublicDir, "corpus.json"), "utf8"),
+ );
+ assert.equal(corpus.tags, undefined);
+});
+
+test("an interrupted build's page debt is paid on the next build", async () => {
+ // Reconstruct exactly what a Ctrl-C between the re-derivation and the page
+ // build leaves behind: records re-derived in LMDB, hashes recorded,
+ // curatedPagesPending set — and pages still holding the OLD content.
+ writeGlobalTags(paths, EVA);
+ const runtime = curatedTagsRuntime(EVA);
+ const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const sums = root.openDB<unknown, [string, string, string]>({ name: "sums", encoding: "msgpack" });
+ const cues = root.openDB<unknown, [string, string, string]>({ name: "cues", encoding: "msgpack" });
+ const subs = root.openDB<unknown, [string, string, string]>({ name: "subs", encoding: "msgpack" });
+ const byChannel = root.openDB<number, [string, string, string]>({ name: "byChannel", encoding: "msgpack" });
+ const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ const res = reapplyCuratedTags({
+ runtime,
+ // The sub-DB handles are structurally what the pass needs.
+ sums: sums as never,
+ cues: cues as never,
+ subs: subs as never,
+ byChannel: byChannel as never,
+ meta: meta as never,
+ });
+ assert.equal(res.changedCount, 1, "the interrupted build did re-derive a record");
+ await sums.flushed;
+ await meta.flushed;
+ assert.equal(meta.get(META_PAGES_PENDING), true);
+ await root.close();
+
+ // The shards are stale: the interrupted build never wrote them.
+ assert.equal("curatedTags" in recordIn(transcriptPage(), HIT), false);
+
+ // Next build: no mtime moved AND the hashes now match, so nothing but the
+ // pending flag can save these pages.
+ await build();
+ assert.deepEqual(recordIn(transcriptPage(), HIT).curatedTags, ["eva-collab"]);
+ assert.deepEqual(recordIn(summariesPage(), HIT).curatedTags, ["eva-collab"]);
+
+ const after = open({ path: paths.lmdbPath, maxDbs: 18, compression: true });
+ const meta2 = after.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
+ assert.equal(meta2.get(META_PAGES_PENDING), false, "the debt is settled");
+ await after.close();
+});
diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts
@@ -5,6 +5,7 @@ import type { TranscriptSummary } from "../lib/transcripts";
import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags";
import {
applyCuratedTagsToSummary,
+ clearCuratedPagesPending,
createTagCounter,
curatedTagsRuntime,
deriveCuratedTags,
@@ -14,6 +15,7 @@ import {
reapplyCuratedTags,
META_ASSIGN_HASH,
META_ASSIGN_SIGS,
+ META_PAGES_PENDING,
META_RULES_HASH,
} from "./curatedTagsIndex";
@@ -199,12 +201,14 @@ test("unchanged hashes do no work at all", () => {
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.meta.put(META_RULES_HASH, f.runtime.rulesHash);
f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash);
- const res = reapplyCuratedTags({ ...f });
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
assert.deepEqual(res, {
rulesChanged: false,
assignmentsChanged: false,
+ changedCount: 0,
changed: [],
examined: 0,
+ pagesPending: false,
});
});
@@ -216,10 +220,11 @@ test("an index with no stored hashes re-derives every record", () => {
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.put(summary("lm", "v2", "20260102", { title: "ordinary" }));
assert.equal(f.meta.get(META_RULES_HASH), undefined);
- const res = reapplyCuratedTags({ ...f });
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
assert.equal(res.rulesChanged, true);
assert.equal(res.examined, 2, "every record is a candidate on a cold index");
- assert.deepEqual(res.changed.map((k) => k[2]), ["v1"]);
+ assert.equal(res.changedCount, 1);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v1"]);
});
test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () => {
@@ -232,8 +237,12 @@ test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () =
for (let i = 0; i < 10; i++) f.put(summary("lm", `v${i}`, `2026010${i}`));
const before = JSON.stringify(Array.from(f.sums.map.values()));
const res = reapplyCuratedTags({ ...f });
- assert.deepEqual(res.changed, []);
+ assert.equal(res.changedCount, 0);
+ assert.equal(res.pagesPending, false, "nothing moved, so nothing to re-page");
assert.equal(res.examined, 10);
+ // The keys are not collected unless asked for — a 30k-record first build must
+ // not hold 30k tuples to answer a boolean.
+ assert.equal(res.changed, undefined);
assert.equal(JSON.stringify(Array.from(f.sums.map.values())), before);
for (const { value } of f.sums.map.values()) {
assert.equal("curatedTags" in value, false);
@@ -247,11 +256,12 @@ test("a rule change re-derives the whole corpus from LMDB and reports the movers
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" }));
f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" }));
- const res = reapplyCuratedTags({ ...f });
+ const res = reapplyCuratedTags({ ...f, collectChanged: true });
assert.equal(res.rulesChanged, true);
assert.equal(res.examined, 3);
+ assert.equal(res.changedCount, 2);
assert.deepEqual(
- res.changed.map((k) => k[2]).sort(),
+ res.changed?.map((k) => k[2]).sort(),
["v1", "v3"],
);
assert.deepEqual(f.sums.get(["20260101", "lm", "v1"])!.curatedTags, ["eva-collab"]);
@@ -268,7 +278,7 @@ test("removing the last rule strips the tag back off every record", () => {
assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]);
const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) };
const res = reapplyCuratedTags(gone);
- assert.deepEqual(res.changed.length, 1);
+ assert.equal(res.changedCount, 1);
assert.equal(f.sums.get(ik)!.curatedTags, undefined);
});
@@ -286,12 +296,12 @@ test("an assignment-only change touches ONLY the videos whose assignment moved",
cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }),
),
};
- const res = reapplyCuratedTags(pinned);
+ const res = reapplyCuratedTags({ ...pinned, collectChanged: true });
assert.equal(res.rulesChanged, false);
assert.equal(res.assignmentsChanged, true);
// 21 videos in the corpus; exactly one was even looked at.
assert.equal(res.examined, 1);
- assert.deepEqual(res.changed.map((k) => k[2]), ["v7"]);
+ assert.deepEqual(res.changed?.map((k) => k[2]), ["v7"]);
assert.deepEqual(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, ["eva-collab"]);
// Clearing it again is also a one-video pass, and the key comes back off.
@@ -307,6 +317,7 @@ test("videos the worker already derived this build are not done twice", () => {
const res = reapplyCuratedTags({
...f,
alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]),
+ collectChanged: true,
});
assert.equal(res.examined, 0);
assert.deepEqual(res.changed, []);
@@ -316,12 +327,50 @@ test("a schema bump records the hashes and re-applies nothing", () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
const res = reapplyCuratedTags({ ...f, allFresh: true });
- assert.equal(res.changed.length, 0);
+ assert.equal(res.changedCount, 0);
assert.equal(res.examined, 0);
+ // The pages ARE rebuilt after a wipe, so the debt is recorded all the same —
+ // an interrupt mid-rebuild must not be forgotten.
+ assert.equal(res.pagesPending, true);
assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
assert.deepEqual(f.meta.get(META_ASSIGN_SIGS), {});
});
+// ─── the interrupted-build flag ───
+
+test("a change sets curatedPagesPending, and only clearing it settles the debt", () => {
+ const f = fixture(cfg([COLLAB_RULE]));
+ f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
+ const first = reapplyCuratedTags({ ...f });
+ assert.equal(first.changedCount, 1);
+ assert.equal(first.pagesPending, true);
+ assert.equal(f.meta.get(META_PAGES_PENDING), true);
+
+ // THE INTERRUPT: the hashes are stored but the page build never ran, so the
+ // flag was never cleared. The next build finds the hashes equal — it
+ // re-derives nothing — and must STILL report that the pages owe a rewrite.
+ const afterCrash = reapplyCuratedTags({ ...f });
+ assert.equal(afterCrash.rulesChanged, false);
+ assert.equal(afterCrash.changedCount, 0);
+ assert.equal(
+ afterCrash.pagesPending,
+ true,
+ "hashes match but the shards are still stale",
+ );
+
+ // The page build completed this time.
+ clearCuratedPagesPending(f.meta);
+ assert.equal(f.meta.get(META_PAGES_PENDING), false);
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+});
+
+test("no change means no debt — an untagged corpus never sets the flag", () => {
+ const f = fixture(cfg([]));
+ f.put(summary("lm", "v1", "20260101"));
+ assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+ assert.notEqual(f.meta.get(META_PAGES_PENDING), true);
+});
+
test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => {
const capRule: CuratedTagDef = {
id: "eva-topic",
diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts
@@ -31,7 +31,7 @@ import {
compileTagRules,
effectiveTagsFor,
evaluateCompiledRules,
- type CompiledTagRule,
+ ruleApplies,
type CompiledTagRules,
type CuratedTagAssignment,
type CuratedTagDef,
@@ -44,6 +44,12 @@ import { readGlobalTags } from "../lib/curatedTagsStore";
export const META_RULES_HASH = "curatedRulesHash";
export const META_ASSIGN_HASH = "curatedAssignHash";
export const META_ASSIGN_SIGS = "curatedAssignSigs";
+// "Records were re-derived; the shared pages do not reflect them yet." Set when
+// the hashes are recorded with changes, cleared only once the shared page build
+// has finished. Without it, a Ctrl-C between the two would leave the hashes
+// stored and the shards stale for ever: the next build would compare equal,
+// re-derive nothing, and never dirty the pages.
+export const META_PAGES_PENDING = "curatedPagesPending";
export const TAG_COUNTS_FILENAME = "tag-counts.json";
export const TAG_COUNTS_VERSION = 1;
@@ -129,29 +135,18 @@ export function curatedTagsRuntime(config: CuratedTagsConfig): CuratedTagsRuntim
};
}
-// Cheap pre-checks, so a video on a channel no rule scopes to never pays for a
-// cue read. Mirrors the scoping evaluateCompiledRules does internally.
-function scopeAllows(
- rule: CompiledTagRule,
- channelSlug: string,
- uploadDate: string | undefined,
-): boolean {
- if (rule.channels && !rule.channels.has(channelSlug)) return false;
- if (rule.dateFrom || rule.dateTo) {
- const date = (uploadDate ?? "").replace(/\D/g, "");
- if (!date) return false;
- if (rule.dateFrom && date < rule.dateFrom) return false;
- if (rule.dateTo && date > rule.dateTo) return false;
- }
- return true;
-}
-
+// Cheap pre-checks, so a video on a channel no rule is scoped to never pays for
+// a cue read. They ask `ruleApplies` — the SAME predicate
+// evaluateCompiledRules uses — so this can never answer "no cues needed" for a
+// rule that would then have fired.
export function needsCaptionCues(
runtime: CuratedTagsRuntime,
channelSlug: string,
uploadDate?: string,
): boolean {
- return runtime.compiled.caption.some((r) => scopeAllows(r, channelSlug, uploadDate));
+ return runtime.compiled.caption.some((r) =>
+ ruleApplies(r, { channelSlug, uploadDate }),
+ );
}
export function needsChatCues(
@@ -160,7 +155,7 @@ export function needsChatCues(
uploadDate?: string,
): boolean {
return runtime.compiled.chatAuthor.some((r) =>
- scopeAllows(r, channelSlug, uploadDate),
+ ruleApplies(r, { channelSlug, uploadDate }),
);
}
@@ -231,16 +226,30 @@ export type ReapplyOptions = {
// A schema bump cleared the cache and the worker re-derived everything, so
// there is nothing to re-apply — just record the hashes.
allFresh?: boolean;
+ // Return the changed index keys as well as the count. Off by default: the
+ // build only needs the boolean, and a first build over 30k records would
+ // otherwise hold 30k three-element tuples for no reason. Tests and any future
+ // per-channel dirtying ask for them.
+ collectChanged?: boolean;
log?: (msg: string) => void;
};
export type ReapplyResult = {
rulesChanged: boolean;
assignmentsChanged: boolean;
- // Videos whose stored curatedTags moved. Their channels must be re-paged.
- changed: IndexKey[];
+ // How many videos' stored curatedTags moved. Non-zero means the shared pages
+ // must be rewritten.
+ changedCount: number;
+ // Which ones — only when `collectChanged` was asked for.
+ changed?: IndexKey[];
// Videos examined (the cost).
examined: number;
+ // True when the shared pages still owe a rewrite for curated tags: either
+ // this call changed records, or an earlier build did and was interrupted
+ // before its pages were written. The caller MUST OR this into whatever
+ // decides to rebuild the shared trees, and clear it (clearCuratedPagesPending)
+ // only once that build has finished.
+ pagesPending: boolean;
};
function indexKeyId(k: IndexKey): string {
@@ -264,8 +273,9 @@ function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[]
// found by a key-only range scan of `byChannel` per
// affected channel, which is the cheap idiom
//
-// Returns the changed index keys so the caller can force those pages to be
-// rewritten: nothing in the mtime diff knows this happened.
+// Returns how many videos changed (and which, on request) so the caller can
+// force those pages to be rewritten: nothing in the mtime diff knows this
+// happened. It also sets `curatedPagesPending` — see the flag's own note.
export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
const { runtime, sums, cues, subs, byChannel, meta } = opts;
const log = opts.log ?? (() => {});
@@ -274,25 +284,50 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
const prevSigs = (meta.get(META_ASSIGN_SIGS) as Record<string, string> | undefined) ?? {};
const rulesChanged = prevRules !== runtime.rulesHash;
const assignmentsChanged = prevAssign !== runtime.assignHash;
+ // A previous build re-derived records and was then interrupted before it
+ // finished writing the shared pages. The hashes are already stored, so this
+ // build would otherwise see "nothing changed" and ship stale shards forever.
+ const pendingBefore = meta.get(META_PAGES_PENDING) === true;
- const record = () => {
+ const record = (pages: boolean) => {
meta.put(META_RULES_HASH, runtime.rulesHash);
meta.put(META_ASSIGN_HASH, runtime.assignHash);
meta.put(META_ASSIGN_SIGS, runtime.sigs);
+ // Set BEFORE the pages are written and cleared only after they are; never
+ // un-set here, or an interrupted build's debt would be forgotten.
+ if (pages) meta.put(META_PAGES_PENDING, true);
};
+ const result = (over: Partial<ReapplyResult> & { changedCount: number }): ReapplyResult => ({
+ rulesChanged,
+ assignmentsChanged,
+ examined: 0,
+ pagesPending: pendingBefore || over.changedCount > 0,
+ ...over,
+ });
if (!rulesChanged && !assignmentsChanged) {
- return { rulesChanged, assignmentsChanged, changed: [], examined: 0 };
+ return result({ changedCount: 0, ...(opts.collectChanged ? { changed: [] } : {}) });
}
if (opts.allFresh) {
- record();
+ // A schema bump: the worker re-derived every record, and the pages are
+ // being rebuilt anyway — but the debt is recorded all the same, so an
+ // interrupt mid-rebuild is not forgotten either.
+ record(true);
log(
`curated tags: rules ${runtime.rulesHash.slice(0, 8)}, assignments ${runtime.assignHash.slice(0, 8)} (full rebuild, nothing to re-apply)`,
);
- return { rulesChanged, assignmentsChanged, changed: [], examined: 0 };
+ return {
+ rulesChanged,
+ assignmentsChanged,
+ changedCount: 0,
+ ...(opts.collectChanged ? { changed: [] } : {}),
+ examined: 0,
+ pagesPending: true,
+ };
}
const changed: IndexKey[] = [];
+ let changedCount = 0;
let examined = 0;
const skip = opts.alreadyFresh;
@@ -323,7 +358,8 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
});
if (applyCuratedTagsToSummary(summary, tags)) {
sums.put(indexKey, summary);
- changed.push(indexKey);
+ changedCount++;
+ if (opts.collectChanged) changed.push(indexKey);
}
};
@@ -367,7 +403,7 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
}
}
- record();
+ record(changedCount > 0);
const what = [
rulesChanged ? `rules ${runtime.rulesHash.slice(0, 8)} (changed)` : null,
assignmentsChanged
@@ -377,9 +413,20 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
.filter(Boolean)
.join(", ");
log(
- `curated tags: ${what}, examined ${examined}, re-derived ${changed.length}`,
+ `curated tags: ${what}, examined ${examined}, re-derived ${changedCount}`,
);
- return { rulesChanged, assignmentsChanged, changed, examined };
+ return result({
+ changedCount,
+ ...(opts.collectChanged ? { changed } : {}),
+ examined,
+ });
+}
+
+// Clear the "the shared pages do not yet reflect the current tags" debt. Call
+// ONLY after the shared transcript/subs page build has completed: until then an
+// interrupt must leave the flag standing, which is the whole point of it.
+export function clearCuratedPagesPending(meta: MetaDb): void {
+ meta.put(META_PAGES_PENDING, false);
}
// ─── per-site counts ───
diff --git a/common/lib/transcripts-server.test.ts b/common/lib/transcripts-server.test.ts
@@ -0,0 +1,55 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { toDisplaySummary } from "./transcripts-server";
+import type { TranscriptSummary } from "./transcripts";
+
+// Run with: node_modules/.bin/tsx --test common/lib/transcripts-server.test.ts
+
+function summary(over: Partial<TranscriptSummary> = {}): TranscriptSummary {
+ return {
+ slug: "test-channel/v1",
+ id: "v1",
+ channelSlug: "test-channel",
+ title: "A stream",
+ uploadDate: "20260101",
+ duration: 3600,
+ channel: "Test Channel",
+ description: "",
+ tags: ["keyword-one", "keyword-two"],
+ isLivestream: false,
+ ageRestricted: false,
+ platform: "youtube",
+ webpageUrl: "https://example.com/v1",
+ ...over,
+ };
+}
+
+test("toDisplaySummary omits curatedTags entirely when there are none", () => {
+ // The byte-identical-pages contract: an untagged corpus's summaries pages
+ // must not gain a key, or every one of them is rewritten on upgrade.
+ const display = toDisplaySummary(summary());
+ assert.equal("curatedTags" in display, false);
+ assert.equal("curatedTags" in toDisplaySummary(summary({ curatedTags: [] })), false);
+ // And the platform's keywords do NOT leak in under the new name.
+ assert.equal(JSON.stringify(display).includes("keyword-one"), false);
+});
+
+test("toDisplaySummary passes curatedTags through in order", () => {
+ const display = toDisplaySummary(
+ summary({ curatedTags: ["eva-collab", "eva-topic"] }),
+ );
+ assert.deepEqual(display.curatedTags, ["eva-collab", "eva-topic"]);
+});
+
+test("toDisplaySummary still derives the state fields it always did", () => {
+ const available = toDisplaySummary(summary());
+ assert.equal(available.isDeleted, false);
+ assert.equal(available.isUnlisted, false);
+ assert.equal("state" in available, false);
+ const deleted = toDisplaySummary(summary({ curatedTags: ["x"] }), {
+ state: "deleted",
+ });
+ assert.equal(deleted.state, "deleted");
+ assert.equal(deleted.isDeleted, true);
+ assert.deepEqual(deleted.curatedTags, ["x"]);
+});
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -172,8 +172,8 @@ Never name the curated field `tags`. Never assume a `tags.json` is the keyword l
| ETA estimator | `editor/app/jobs/active/buildActiveJobs.ts:38` | `computeEtaSeconds`. |
| Binary search over cue starts | `common/components/TranscriptModal.tsx:473-491` | `findActiveIndex`. |
| Curated-tag model (pure) | `common/lib/curatedTags.ts` | `sanitizeTagsConfig` / `mergeTagDefs` / `effectiveTagsFor` / `compileTagRules` + `evaluateCompiledRules`. Compile ONCE per build, evaluate per video. |
-| Curated-tag persistence | `common/lib/curatedTagsStore.ts:176` | `applyTagAssignments` — the ONE write path for every tag writer (editor, ops, umtool); one atomic write per call. |
-| A chat cue's author | `common/lib/curatedTags.ts:488` — `chatAuthorOf` | The `<author>: ` prefix; there is no `author` field on `Cue`. |
+| Curated-tag persistence | `common/lib/curatedTagsStore.ts:198` | `applyTagAssignments` — the ONE write path for every tag writer (editor, ops, umtool); one atomic write per call, and it refuses to pin a tag the vocabulary does not define. |
+| A chat cue's author | `common/lib/curatedTags.ts:495` — `chatAuthorOf` | The `<author>: ` prefix; there is no `author` field on `Cue`. |
---
@@ -201,7 +201,7 @@ Exposed to the settings form via `listTranscriptionApps()` (`:272`), consumed at
| Constant | Where | Value |
| --- | --- | --- |
-| `SCHEMA_VERSION` | `buildIndex.ts:164` | **13** — and curated tags deliberately did NOT bump it (2026-09-21); see below |
+| `SCHEMA_VERSION` | `buildIndex.ts:165` | **13** — and curated tags deliberately did NOT bump it (2026-09-21); see below |
| `CUES_FILE_VERSION` | `normalizeTranscript.ts:33` | 2 |
| `CORPUS_SPEC_VERSION` | `common/lib/archive/contract.ts:35` (re-exported `corpus.ts:59`) | **4** (3 → 4 for `/tags.json`, 2026-09-21) |
| `MANIFEST_VERSION` (site summaries) | `common/lib/manifest.ts:36` | 3 |
@@ -228,21 +228,23 @@ Transcript shards ship the **full `cues` array inline** — `buildIndex.ts:830-8
`curatedTags` is DERIVED at build time — `(rule hits ∪ pins) − suppressions` — and **rule
hits are never persisted**. So nothing in the per-video mtime diff can see that a rule was
-edited or a video pinned, and the mechanism is two keys in the existing `meta` sub-DB
-(`common/controller/curatedTagsIndex.ts:44-46`), checked **unconditionally on every build**:
+edited or a video pinned, and the mechanism is four keys in the existing `meta` sub-DB
+(`common/controller/curatedTagsIndex.ts:44-52`), checked **unconditionally on every build**:
| Key | Covers | Deliberately excludes |
| --- | --- | --- |
| `curatedRulesHash` | every def's id + `order`, and every **enabled** rule's kind, pattern, channel scope and date range | label / colour / group / hidden — relabelling a tag must not re-page 30,000 videos |
| `curatedAssignHash` | every pin and suppression | — |
| `curatedAssignSigs` | per-key `manual|suppressed` signature | not a hash: it is what makes an assignment-only edit re-derive just the videos that moved |
+| `curatedPagesPending` | "records were re-derived, the shared pages have not been written yet" | set when the hashes are recorded WITH changes, cleared only after the shared page build — without it a Ctrl-C between the two would store the hashes and leave the shards stale for ever, since the next build would compare equal and dirty nothing |
-`reapplyCuratedTags` (`:269`) runs after the mtime diff: rules changed → every video is a
+`reapplyCuratedTags` (`:279`) runs after the mtime diff: rules changed → every video is a
candidate (cue reads gated per video by channel/date scope); assignments changed → only the
symmetric difference of the signatures, located by a key-only `byChannel` range scan per
affected channel. It re-derives out of LMDB — **no video directory is read twice** — and
whatever it changes flips `sharedNeedsBuild`, because the page writer's sha1 skip is what
-then leaves the untouched pages untouched. The per-site fingerprint (`buildIndex.ts:1741`)
+then leaves the untouched pages untouched — and `curatedPagesPending` is what covers the
+build that was interrupted between the two. The per-site fingerprint (`buildIndex.ts:1765`)
includes both hashes, or a tag-only edit would leave a site reporting "up to date" with
yesterday's counts. **No new sub-DB**, so the `clearAsync()` enumeration is unchanged.
@@ -264,9 +266,9 @@ in. Rules bind only from `transcripts/tags.json` — promote one there to make i
editor therefore offers no site-side rule editor.
**A chat cue's author is a string prefix, not a field.** `common/lib/liveChat.ts:100-105`
-emits every live-chat cue as ``text: author ? `${author}: ${text}` : text`` — `Cue` is
+emits every live-chat cue as ``text: author ? `${author}: ${text}` : text`` (`:104`) — `Cue` is
`{start, end, text}` (`vtt.ts:1`) and has no `author`. So a `chat-author` rule matches
-`chatAuthorOf(text)` (`curatedTags.ts:488`): everything before the FIRST `": "`, and `null`
+`chatAuthorOf(text)` (`curatedTags.ts:495`): everything before the FIRST `": "`, and `null`
when there is none (a cue with no author can never match). Anything that wants the author
of a chat line must use this, not a second copy of the split.