commit 319e9370383dad6caa89e7ccd3495069f251d7f4
parent f387da569a0809466bd6fc86c61af54b8f4beb7b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 16:12:16 -0400
tags: the re-apply pass must not hold the event loop for minutes
Observed on the live editor 2026-09-21: an unscoped caption rule turned
`reapplyCuratedTags` into 2-3 minutes of unbroken synchronous work — every
record in `sums` (77,224 here), each one decoding its cues out of LMDB at
roughly 500/second — and every page render in the same process timed out
while it ran. Only the cheapest API routes answered, in the gaps.
The pass is now async and hands the event loop back every 200 records walked,
in both loops. Nothing else moves: the same LMDB reads, the same `record()`
after the loops, the same `curatedPagesPending` contract (set before the
shared page build, cleared only after it).
The whole-corpus scan also passes `snapshot: false` to `sums.getRange`. That
is lmdb-js's documented idiom for an iterator held open a long time: the
default keeps ONE read transaction open for the whole walk, and a read txn
held across minutes of writes stops LMDB reusing the pages those writes free.
With it off the cursor renews and repositions on the key it last returned —
safe here because the only record the loop puts is the one under the cursor,
and the cursor only moves forward.
The assignment-only path collects its matches inside the key-only scan and
re-derives after it, so no cursor is held across an await at all; the match
set is bounded by the assignments that moved, not by the channel's size.
`ReapplyResult.yields` reports the count, which is what the two new tests
assert against a self-rescheduling setImmediate chain — an instrument that
can only advance if the loop actually reaches its check phase. The negative
control is the state of every untagged install: ten records, zero yields, not
one trip through the event loop.
Co-Authored-By: Claude Opus <noreply@anthropic.com>
Diffstat:
5 files changed, 190 insertions(+), 36 deletions(-)
diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts
@@ -884,7 +884,10 @@ export async function buildIndex({
// Anything that changed has to be re-paged, which is what the
// sharedNeedsBuild flip below buys: the page writer's sha1 skip then keeps
// the untouched pages untouched, so compose still copies only real changes.
- const curatedReapply = reapplyCuratedTags({
+ // It is awaited because it yields to the event loop as it walks: a rule edit
+ // makes this a whole-corpus pass, and when the editor shares the process a
+ // synchronous one blocks every page render for minutes.
+ const curatedReapply = await reapplyCuratedTags({
runtime: curated,
sums,
cues,
diff --git a/common/controller/curatedTagsBuild.test.ts b/common/controller/curatedTagsBuild.test.ts
@@ -300,7 +300,7 @@ test("an interrupted build's page debt is paid on the next build", async () => {
const subs = root.openDB<unknown, [string, string, string]>({ name: "subs", encoding: "msgpack" });
const byChannel = root.openDB<number, [string, string, string]>({ name: "byChannel", encoding: "msgpack" });
const meta = root.openDB<unknown, string>({ name: "meta", encoding: "msgpack" });
- const res = reapplyCuratedTags({
+ const res = await reapplyCuratedTags({
runtime,
// The sub-DB handles are structurally what the pass needs.
sums: sums as never,
diff --git a/common/controller/curatedTagsIndex.test.ts b/common/controller/curatedTagsIndex.test.ts
@@ -6,6 +6,7 @@ import type { CuratedTagDef, CuratedTagsConfig } from "../lib/curatedTags";
import {
applyCuratedTagsToSummary,
clearCuratedPagesPending,
+ CURATED_REAPPLY_YIELD_EVERY,
createTagCounter,
curatedTagsRuntime,
deriveCuratedTags,
@@ -196,18 +197,19 @@ test("deriveCuratedTags folds rule hits with pins and suppressions", () => {
// ─── reapplyCuratedTags ───
-test("unchanged hashes do no work at all", () => {
+test("unchanged hashes do no work at all", async () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.meta.put(META_RULES_HASH, f.runtime.rulesHash);
f.meta.put(META_ASSIGN_HASH, f.runtime.assignHash);
- const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ const res = await reapplyCuratedTags({ ...f, collectChanged: true });
assert.deepEqual(res, {
rulesChanged: false,
assignmentsChanged: false,
changedCount: 0,
changed: [],
examined: 0,
+ yields: 0,
pagesPending: false,
});
});
@@ -215,19 +217,19 @@ test("unchanged hashes do no work at all", () => {
// THE COLD START, and the reason curated tags need no SCHEMA_VERSION bump: an
// index built before they existed carries no hash at all, which can never equal
// the current one, so the first build re-derives every record by itself.
-test("an index with no stored hashes re-derives every record", () => {
+test("an index with no stored hashes re-derives every record", async () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.put(summary("lm", "v2", "20260102", { title: "ordinary" }));
assert.equal(f.meta.get(META_RULES_HASH), undefined);
- const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ const res = await reapplyCuratedTags({ ...f, collectChanged: true });
assert.equal(res.rulesChanged, true);
assert.equal(res.examined, 2, "every record is a candidate on a cold index");
assert.equal(res.changedCount, 1);
assert.deepEqual(res.changed?.map((k) => k[2]), ["v1"]);
});
-test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () => {
+test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", async () => {
// The state of every existing install the day this ships: no vocabulary, no
// assignments. The pass still runs (the hashes have to be recorded), but
// every record derives [] and comes out byte-identical, so nothing is written
@@ -236,7 +238,7 @@ test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () =
const f = fixture(cfg([]));
for (let i = 0; i < 10; i++) f.put(summary("lm", `v${i}`, `2026010${i}`));
const before = JSON.stringify(Array.from(f.sums.map.values()));
- const res = reapplyCuratedTags({ ...f });
+ const res = await reapplyCuratedTags({ ...f });
assert.equal(res.changedCount, 0);
assert.equal(res.pagesPending, false, "nothing moved, so nothing to re-page");
assert.equal(res.examined, 10);
@@ -248,15 +250,15 @@ test("an UNTAGGED corpus cold-starts to zero changes — no page rewrites", () =
assert.equal("curatedTags" in value, false);
}
// And the hashes are now recorded, so every later build is a true no-op.
- assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+ assert.equal((await reapplyCuratedTags({ ...f })).examined, 0);
});
-test("a rule change re-derives the whole corpus from LMDB and reports the movers", () => {
+test("a rule change re-derives the whole corpus from LMDB and reports the movers", async () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
f.put(summary("lm", "v2", "20260102", { title: "ordinary stream" }));
f.put(summary("other", "v3", "20260103", { description: "ELFPIRE guested" }));
- const res = reapplyCuratedTags({ ...f, collectChanged: true });
+ const res = await reapplyCuratedTags({ ...f, collectChanged: true });
assert.equal(res.rulesChanged, true);
assert.equal(res.examined, 3);
assert.equal(res.changedCount, 2);
@@ -268,27 +270,27 @@ test("a rule change re-derives the whole corpus from LMDB and reports the movers
assert.equal(f.sums.get(["20260102", "lm", "v2"])!.curatedTags, undefined);
// The hashes are recorded, so a second pass is a no-op.
assert.equal(f.meta.get(META_RULES_HASH), f.runtime.rulesHash);
- assert.equal(reapplyCuratedTags({ ...f }).examined, 0);
+ assert.equal((await reapplyCuratedTags({ ...f })).examined, 0);
});
-test("removing the last rule strips the tag back off every record", () => {
+test("removing the last rule strips the tag back off every record", async () => {
const f = fixture(cfg([COLLAB_RULE]));
const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
- reapplyCuratedTags({ ...f });
+ await reapplyCuratedTags({ ...f });
assert.deepEqual(f.sums.get(ik)!.curatedTags, ["eva-collab"]);
const gone = { ...f, runtime: curatedTagsRuntime(cfg([])) };
- const res = reapplyCuratedTags(gone);
+ const res = await reapplyCuratedTags(gone);
assert.equal(res.changedCount, 1);
assert.equal(f.sums.get(ik)!.curatedTags, undefined);
});
-test("an assignment-only change touches ONLY the videos whose assignment moved", () => {
+test("an assignment-only change touches ONLY the videos whose assignment moved", async () => {
const f = fixture(cfg([COLLAB_RULE]));
for (let i = 0; i < 20; i++) {
f.put(summary("lm", `v${i}`, `202601${String(i + 10).padStart(2, "0")}`));
}
f.put(summary("other", "w1", "20260201"));
- reapplyCuratedTags({ ...f }); // establish the baseline hashes
+ await reapplyCuratedTags({ ...f }); // establish the baseline hashes
const pinned = {
...f,
@@ -296,7 +298,7 @@ test("an assignment-only change touches ONLY the videos whose assignment moved",
cfg([COLLAB_RULE], { "lm/v7": { manual: ["eva-collab"] } }),
),
};
- const res = reapplyCuratedTags({ ...pinned, collectChanged: true });
+ const res = await reapplyCuratedTags({ ...pinned, collectChanged: true });
assert.equal(res.rulesChanged, false);
assert.equal(res.assignmentsChanged, true);
// 21 videos in the corpus; exactly one was even looked at.
@@ -306,15 +308,15 @@ test("an assignment-only change touches ONLY the videos whose assignment moved",
// Clearing it again is also a one-video pass, and the key comes back off.
const cleared = { ...f, runtime: curatedTagsRuntime(cfg([COLLAB_RULE])) };
- const res2 = reapplyCuratedTags(cleared);
+ const res2 = await reapplyCuratedTags(cleared);
assert.equal(res2.examined, 1);
assert.equal(f.sums.get(["20260117", "lm", "v7"])!.curatedTags, undefined);
});
-test("videos the worker already derived this build are not done twice", () => {
+test("videos the worker already derived this build are not done twice", async () => {
const f = fixture(cfg([COLLAB_RULE]));
const ik = f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
- const res = reapplyCuratedTags({
+ const res = await reapplyCuratedTags({
...f,
alreadyFresh: new Set([`${ik[0]}\x00${ik[1]}\x00${ik[2]}`]),
collectChanged: true,
@@ -323,10 +325,10 @@ test("videos the worker already derived this build are not done twice", () => {
assert.deepEqual(res.changed, []);
});
-test("a schema bump records the hashes and re-applies nothing", () => {
+test("a schema bump records the hashes and re-applies nothing", async () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
- const res = reapplyCuratedTags({ ...f, allFresh: true });
+ const res = await reapplyCuratedTags({ ...f, allFresh: true });
assert.equal(res.changedCount, 0);
assert.equal(res.examined, 0);
// The pages ARE rebuilt after a wipe, so the debt is recorded all the same —
@@ -338,10 +340,10 @@ test("a schema bump records the hashes and re-applies nothing", () => {
// ─── the interrupted-build flag ───
-test("a change sets curatedPagesPending, and only clearing it settles the debt", () => {
+test("a change sets curatedPagesPending, and only clearing it settles the debt", async () => {
const f = fixture(cfg([COLLAB_RULE]));
f.put(summary("lm", "v1", "20260101", { title: "Elfpire collab" }));
- const first = reapplyCuratedTags({ ...f });
+ const first = await reapplyCuratedTags({ ...f });
assert.equal(first.changedCount, 1);
assert.equal(first.pagesPending, true);
assert.equal(f.meta.get(META_PAGES_PENDING), true);
@@ -349,7 +351,7 @@ test("a change sets curatedPagesPending, and only clearing it settles the debt",
// THE INTERRUPT: the hashes are stored but the page build never ran, so the
// flag was never cleared. The next build finds the hashes equal — it
// re-derives nothing — and must STILL report that the pages owe a rewrite.
- const afterCrash = reapplyCuratedTags({ ...f });
+ const afterCrash = await reapplyCuratedTags({ ...f });
assert.equal(afterCrash.rulesChanged, false);
assert.equal(afterCrash.changedCount, 0);
assert.equal(
@@ -361,17 +363,17 @@ test("a change sets curatedPagesPending, and only clearing it settles the debt",
// The page build completed this time.
clearCuratedPagesPending(f.meta);
assert.equal(f.meta.get(META_PAGES_PENDING), false);
- assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+ assert.equal((await reapplyCuratedTags({ ...f })).pagesPending, false);
});
-test("no change means no debt — an untagged corpus never sets the flag", () => {
+test("no change means no debt — an untagged corpus never sets the flag", async () => {
const f = fixture(cfg([]));
f.put(summary("lm", "v1", "20260101"));
- assert.equal(reapplyCuratedTags({ ...f }).pagesPending, false);
+ assert.equal((await reapplyCuratedTags({ ...f })).pagesPending, false);
assert.notEqual(f.meta.get(META_PAGES_PENDING), true);
});
-test("cue-backed rules read cues out of LMDB, and only where they are scoped", () => {
+test("cue-backed rules read cues out of LMDB, and only where they are scoped", async () => {
const capRule: CuratedTagDef = {
id: "eva-topic",
label: "Discussed",
@@ -400,7 +402,7 @@ test("cue-backed rules read cues out of LMDB, and only where they are scoped", (
{ track: "live_chat", cues: [{ start: 0, end: 1, text: "ElfpireEva: hi" }] },
]);
- reapplyCuratedTags({ ...f });
+ await reapplyCuratedTags({ ...f });
// Order follows the vocabulary (neither def sets `order`, so it is the
// config's own order), not the alphabet.
assert.deepEqual(f.sums.get(a)!.curatedTags, ["eva-topic", "eva-in-chat"]);
@@ -409,6 +411,82 @@ test("cue-backed rules read cues out of LMDB, and only where they are scoped", (
assert.equal(f.sums.get(b)!.curatedTags, undefined);
});
+// ─── yielding ───
+
+// The bug this section exists for: on the live editor a single unscoped caption
+// rule turned this pass into two to three minutes of unbroken synchronous work
+// — 77k records, each one decoding its cues out of LMDB — and every page render
+// in the same process timed out while it ran. The fix is not to make it faster;
+// it is to make it interruptible.
+//
+// `ticks` is the instrument: a self-rescheduling setImmediate chain can only
+// advance if the event loop actually reaches its check phase, which it cannot
+// do while a synchronous loop holds the thread.
+function immediateTicker() {
+ let ticks = 0;
+ let running = true;
+ const tick = () => {
+ if (!running) return;
+ ticks++;
+ setImmediate(tick);
+ };
+ setImmediate(tick);
+ return {
+ get ticks() {
+ return ticks;
+ },
+ stop() {
+ running = false;
+ },
+ };
+}
+
+test("the whole-corpus walk hands the event loop back as it goes", async () => {
+ const capRule: CuratedTagDef = {
+ id: "eva-topic",
+ label: "Discussed",
+ rules: [{ id: "cap", kind: "caption", pattern: "elfpire", enabled: true }],
+ };
+ const f = fixture(cfg([capRule]));
+ const n = 1000;
+ for (let i = 0; i < n; i++) {
+ const ik = f.put(summary("lm", `v${String(i).padStart(4, "0")}`, "20260101"));
+ // A caption rule is the expensive shape: every record pays a cue read.
+ f.cues.put(ik, [{ start: 0, end: 1, text: `nothing here ${i}` }]);
+ }
+
+ const ticker = immediateTicker();
+ const res = await reapplyCuratedTags({ ...f });
+ ticker.stop();
+
+ assert.equal(res.examined, n);
+ assert.equal(
+ res.yields,
+ Math.floor(n / CURATED_REAPPLY_YIELD_EVERY),
+ "one yield per CURATED_REAPPLY_YIELD_EVERY records walked",
+ );
+ assert.ok(
+ ticker.ticks > 0,
+ `the event loop never ran during the pass (${ticker.ticks} ticks)`,
+ );
+});
+
+// The negative control, and the state of every untagged install: the pass still
+// runs — the hashes have to be recorded — but it must not buy a round trip
+// through the event loop to walk ten records.
+test("a corpus below the yield interval returns without yielding once", async () => {
+ const f = fixture(cfg([]));
+ for (let i = 0; i < 10; i++) f.put(summary("lm", `v${i}`, `2026010${i}`));
+
+ const ticker = immediateTicker();
+ const res = await reapplyCuratedTags({ ...f });
+ ticker.stop();
+
+ assert.equal(res.examined, 10);
+ assert.equal(res.yields, 0);
+ assert.equal(ticker.ticks, 0, "not even one trip through the check phase");
+});
+
// ─── counts + publication ───
test("createTagCounter accumulates per tag and per channel", () => {
diff --git a/common/controller/curatedTagsIndex.ts b/common/controller/curatedTagsIndex.ts
@@ -212,6 +212,26 @@ type MetaDb = {
put(key: string, value: unknown): unknown;
};
+// How many records the re-apply pass walks between yields to the event loop.
+//
+// This pass is the one place in the build that touches EVERY record in a single
+// stretch, and for a caption or chat-author rule it decodes each video's cues
+// out of LMDB while doing it — measured at roughly 500 videos/second, so an
+// unscoped caption rule over a 77k-record corpus is two to three minutes of
+// uninterrupted synchronous work. Run inside the editor that is minutes with no
+// event loop: every page render times out while only the cheapest API routes
+// answer in the gaps. The rest of the build does not behave this way because
+// its per-video worker awaits.
+//
+// 200 keeps the longest uninterrupted chunk well under half a second even on
+// the cue-decoding path, while the setImmediate turns themselves stay noise
+// against the decode cost (~385 of them across the whole corpus).
+export const CURATED_REAPPLY_YIELD_EVERY = 200;
+
+function yieldToEventLoop(): Promise<void> {
+ return new Promise<void>((resolve) => setImmediate(resolve));
+}
+
export type ReapplyOptions = {
runtime: CuratedTagsRuntime;
sums: SumsDb;
@@ -244,6 +264,10 @@ export type ReapplyResult = {
changed?: IndexKey[];
// Videos examined (the cost).
examined: number;
+ // How many times the pass handed the event loop back. Zero is the normal
+ // answer on a small or untouched corpus; it is non-zero exactly when the pass
+ // would otherwise have blocked.
+ yields: number;
// True when the shared pages still owe a rewrite for curated tags: either
// this call changed records, or an earlier build did and was interrupted
// before its pages were written. The caller MUST OR this into whatever
@@ -276,7 +300,15 @@ function chatCuesOf(stored: { track: string; cues: Cue[] }[] | undefined): Cue[]
// Returns how many videos changed (and which, on request) so the caller can
// force those pages to be rewritten: nothing in the mtime diff knows this
// happened. It also sets `curatedPagesPending` — see the flag's own note.
-export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
+//
+// It is async ONLY to yield: every CURATED_REAPPLY_YIELD_EVERY records it hands
+// the event loop back, because the whole-corpus walk is otherwise minutes of
+// unbroken synchronous decoding and the editor serving the same process goes
+// dark for the duration. The LMDB reads are unchanged; the write ordering is
+// unchanged; `record()` still runs after the loops and before the log line.
+export async function reapplyCuratedTags(
+ opts: ReapplyOptions,
+): Promise<ReapplyResult> {
const { runtime, sums, cues, subs, byChannel, meta } = opts;
const log = opts.log ?? (() => {});
const prevRules = meta.get(META_RULES_HASH) as string | undefined;
@@ -301,6 +333,7 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
rulesChanged,
assignmentsChanged,
examined: 0,
+ yields: 0,
pagesPending: pendingBefore || over.changedCount > 0,
...over,
});
@@ -322,6 +355,7 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
changedCount: 0,
...(opts.collectChanged ? { changed: [] } : {}),
examined: 0,
+ yields: 0,
pagesPending: true,
};
}
@@ -329,8 +363,20 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
const changed: IndexKey[] = [];
let changedCount = 0;
let examined = 0;
+ let yields = 0;
+ let sinceYield = 0;
const skip = opts.alreadyFresh;
+ // Called once per record the loops walk — INCLUDING the ones `alreadyFresh`
+ // skips, because a 77k-record corpus where the worker already did every video
+ // still walks 77k keys.
+ const maybeYield = async (): Promise<void> => {
+ if (++sinceYield < CURATED_REAPPLY_YIELD_EVERY) return;
+ sinceYield = 0;
+ yields++;
+ await yieldToEventLoop();
+ };
+
// `known` is the value the cursor already decoded, when there is one — a
// second sums.get() per video would double the msgpack decodes on the
// whole-corpus path.
@@ -366,8 +412,17 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
if (rulesChanged) {
// Every video is a candidate. This walks `sums` keys only; the value comes
// from the same cursor, and cue reads are gated per video above.
- for (const { key, value } of sums.getRange()) {
+ //
+ // `snapshot: false` because this walk now awaits: lmdb-js otherwise keeps
+ // ONE read transaction open for the whole iteration, and a read txn held
+ // open across minutes of writes stops LMDB reusing the pages those writes
+ // free. With it off the cursor renews its read transaction and repositions
+ // on the key it last returned — the library's own long-duration-iterator
+ // idiom. Nothing written here can confuse it: the only record this loop
+ // puts is the one under the cursor, and the cursor only moves forward.
+ for (const { key, value } of sums.getRange({ snapshot: false })) {
rederive(key as IndexKey, value as TranscriptSummary);
+ await maybeYield();
}
} else {
// Assignment-only change: the symmetric difference of the signatures.
@@ -390,6 +445,11 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
byChannelSlug.set(slug, set);
}
for (const [slug, ids] of byChannelSlug) {
+ // Matched inside the scan, re-derived after it. The scan is key-only and
+ // cheap; the re-derivation is what costs, and keeping the await out of
+ // the cursor is free here because the match set is bounded by the
+ // assignments that moved, not by the channel's size.
+ const hits: IndexKey[] = [];
for (const { key } of byChannel.getRange({
start: [slug],
end: [slug, ""],
@@ -398,7 +458,11 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
const ck = key as ChannelKey;
if (ck[0] !== slug) continue;
if (!ids.has(ck[2])) continue;
- rederive([ck[1], ck[0], ck[2]]);
+ hits.push([ck[1], ck[0], ck[2]]);
+ }
+ for (const indexKey of hits) {
+ rederive(indexKey);
+ await maybeYield();
}
}
}
@@ -413,12 +477,14 @@ export function reapplyCuratedTags(opts: ReapplyOptions): ReapplyResult {
.filter(Boolean)
.join(", ");
log(
- `curated tags: ${what}, examined ${examined}, re-derived ${changedCount}`,
+ `curated tags: ${what}, examined ${examined}, re-derived ${changedCount}` +
+ (yields > 0 ? `, yielded ${yields}x` : ""),
);
return result({
changedCount,
...(opts.collectChanged ? { changed } : {}),
examined,
+ yields,
});
}
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -241,7 +241,14 @@ edited or a video pinned, and the mechanism is four keys in the existing `meta`
`reapplyCuratedTags` (`:279`) runs after the mtime diff: rules changed → every video is a
candidate (cue reads gated per video by channel/date scope); assignments changed → only the
symmetric difference of the signatures, located by a key-only `byChannel` range scan per
-affected channel. It re-derives out of LMDB — **no video directory is read twice** — and
+affected channel. **It is `async` and yields to the event loop every
+`CURATED_REAPPLY_YIELD_EVERY` (200) records walked**, in both loops — a rule edit makes this
+the one whole-corpus pass in the build, and at ~500 cue-decodes/second an unscoped caption
+rule is 2–3 minutes; run synchronously inside the editor that blocked every page render for
+the duration (observed live 2026-09-21, `/channels` `/jobs` `/tags` all >25 s). The
+whole-corpus scan therefore passes `snapshot: false` to `sums.getRange`, lmdb-js's
+long-duration-iterator idiom, so the read transaction is not held open across minutes of
+writes. `ReapplyResult.yields` reports the count; it is 0 below the interval. It re-derives out of LMDB — **no video directory is read twice** — and
whatever it changes flips `sharedNeedsBuild`, because the page writer's sha1 skip is what
then leaves the untouched pages untouched — and `curatedPagesPending` is what covers the
build that was interrupted between the two. The per-site fingerprint (`buildIndex.ts:1765`)