Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b48312187dba716598416f87d56557b92095dde3
parent ef0f538cf57463d8d9ddcf4ec44bd90130a8b98e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 08:47:39 -0400

index: a one-shot caption-track pass re-reads exactly the records the rule reaches

The index derives a caption record's cues through readEnglishVttCues, and
stats every caption input for change detection (the newest of each English
VTT and the pin).

Like the platform-labels pass, CAPTION_TRACK_KEY records the rule version the
index was built under. The first build that sees a new one queues every
caption record with more than one English VTT (the rule may now pick another)
or whose stored cues are empty (asked of the stored bytes, never decoded:
the cue-block shape used to parse to nothing) — and nothing else. It logs
"Caption track v1: N record(s) re-read." and, per channel, how many now read
different text and how many had none and now do; the result carries the same
report. The version is recorded only when no channel is held. A first build
or a schema bump has nothing to re-read.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/buildIndex.ts | 117++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-----
Acommon/controller/buildIndexCaptionTrack.test.ts | 156+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 266 insertions(+), 7 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -107,7 +107,11 @@ import { import { resolveChannelGroupId } from "../lib/channelGroups"; import type { Paths } from "../lib/paths"; import { + CAPTION_TRACK_RULE_VERSION, + captionInputs, + englishVttsByPreference, pickIndexTranscript, + readEnglishVttCues, readSubTracks, readVideoFiles, type IndexTranscript, @@ -205,6 +209,39 @@ const SCHEMA_VERSION = 13; const PLATFORM_LABELS_VERSION = 1; const PLATFORM_LABELS_KEY = "platformLabels"; +// CAPTION TRACK — the same one-shot shape, for the caption-track rule +// (CAPTION_TRACK_RULE_VERSION, lib/videoStatus.ts). When the rule changes, the +// first build that sees the new version re-reads, from disk, every caption +// record the change can reach — one with more than one English VTT (the rule +// may now pick another), or whose stored cues are empty (the cue-block shape +// used to parse to nothing) — and nothing else; then records the version, +// again only when no channel is held. Each re-read is reported per channel: +// how many now read different text, how many had none and now do. +const CAPTION_TRACK_KEY = "captionTrackRule"; + +// Whether the cue list stored under a key is empty or absent, WITHOUT decoding +// it: a non-empty list of cues is far longer than the few bytes an empty +// msgpack array takes, and decoding every caption record of the corpus to ask +// this would cost the full read the pass exists to avoid. +// A cue list's text, for "did the words change" — timing alone is not a +// different transcript. +function cueText(list: readonly Cue[]): string { + return list.map((c) => c.text).join("\n"); +} + +function storedCuesEmpty(db: { getBinaryFast(key: IndexKey): Buffer | undefined }, key: IndexKey): boolean { + const raw = db.getBinaryFast(key); + return raw === undefined || raw.length <= 4; +} + +export type CaptionTrackChannelReport = { + reread: number; + // Re-read records whose caption text is not what the index held. + changed: number; + // Of those, the ones the index held no text for. + zeroToText: number; +}; + // Per-channel post stats, persisted so per-site aggregates survive a no-op // rebuild that doesn't re-encode the post pages. Mirrors ChannelSubsStat. type ChannelPostsStat = { @@ -299,6 +336,8 @@ type LiveEntry = { transcriptPath: string; transcriptMs: number | null; transcriptKind: IndexTranscript["kind"] | null; + // How many English VTTs the dir holds (the caption-track pass's question). + englishVttCount: number; subTracks: SubTrack[]; subsMs: number | null; availabilityMs: number | null; @@ -429,10 +468,17 @@ async function scanSource( : path.join(fullVideoDir, "transcript.json"); let transcriptMs: number | null = null; if (picked) { - try { - transcriptMs = (await stat(transcriptPath)).mtimeMs; - } catch { - transcriptMs = null; + // Captions: the newest of every caption input (each English VTT and + // the operator's pin), since the cues may come from any of them. + const inputs = + picked.kind === "vtt" ? captionInputs(files.entries) : [picked.filename]; + for (const name of inputs) { + try { + const ms = (await stat(path.join(fullVideoDir, name))).mtimeMs; + if (transcriptMs === null || ms > transcriptMs) transcriptMs = ms; + } catch { + // ignore + } } } const subTracks = await readSubTracks(fullVideoDir); @@ -482,6 +528,7 @@ async function scanSource( transcriptPath, transcriptMs, transcriptKind: picked?.kind ?? null, + englishVttCount: englishVttsByPreference(files.entries).length, subTracks, subsMs, availabilityMs, @@ -542,6 +589,9 @@ export type BuildIndexResult = { // Channels whose media could not be read, so they were not rescanned: their // index records and shared pages were kept as they were (see the header). heldChannels: string[]; + // Per channel, what a caption-track pass re-read and changed; empty when no + // pass ran (CAPTION_TRACK_KEY). + captionTrack: Record<string, CaptionTrackChannelReport>; }; export type BuildIndexOptions = { @@ -803,6 +853,31 @@ export async function buildIndex({ ); } + // Caption records the caption-track rule may now read differently + // (CAPTION_TRACK_KEY). A schema bump re-reads everything anyway. + const captionTrackDue = + meta.get(CAPTION_TRACK_KEY) !== CAPTION_TRACK_RULE_VERSION; + // pathKeyIds of the records this pass re-reads. + const captionReread = new Set<string>(); + if (captionTrackDue && !schemaBumped) { + const queued = new Set( + [...added, ...changed].map((s) => pathKeyId([s.channelSlug, s.videoDir])), + ); + for (const s of live) { + if (s.transcriptKind !== "vtt") continue; + const pk: PathKey = [s.channelSlug, s.videoDir]; + const prev = mtimes.get(pk); + if (!prev) continue; + if (s.englishVttCount < 2 && !storedCuesEmpty(cues, prev.indexKey)) continue; + captionReread.add(pathKeyId(pk)); + if (!queued.has(pathKeyId(pk))) changed.push(s); + } + log( + `Caption track v${CAPTION_TRACK_RULE_VERSION}: ${captionReread.size} record(s) re-read.`, + ); + } + const captionReport = new Map<string, CaptionTrackChannelReport>(); + const anyMutations = added.length > 0 || changed.length > 0 || removed.length > 0; @@ -901,11 +976,11 @@ export async function buildIndex({ ); if (cueList === undefined && s.transcriptMs !== null && s.transcriptKind) { try { - const raw = await readFile(s.transcriptPath, "utf8"); cueList = s.transcriptKind === "vtt" - ? parseVtt(raw) - : parseTranscriptJson(raw); + ? // The caption-track rule (lib/videoStatus.ts). + (await readEnglishVttCues(videoFullDir))?.cues + : parseTranscriptJson(await readFile(s.transcriptPath, "utf8")); } catch { cueList = undefined; } @@ -923,6 +998,12 @@ export async function buildIndex({ const pk: PathKey = [s.channelSlug, s.videoDir]; const prev = mtimes.get(pk); + // What the last build held for this video's captions, for the + // caption-track report — read before the put below replaces it. + const heldBefore = + prev && captionReread.has(pathKeyId(pk)) + ? cueText(cues.get(prev.indexKey) ?? []) + : undefined; // What the last build held for this video's sub tracks, read before // a re-keyed record is removed: a tiered live chat whose raw cannot // be read now keeps the cues it had (below). @@ -938,6 +1019,17 @@ export async function buildIndex({ if (cueList) cues.put(indexKey, cueList); else cues.remove(indexKey); + if (heldBefore !== undefined) { + const r = captionReport.get(s.channelSlug) ?? { reread: 0, changed: 0, zeroToText: 0 }; + r.reread++; + const now = cueText(cueList ?? []); + if (heldBefore !== now) { + r.changed++; + if (heldBefore === "" && now !== "") r.zeroToText++; + } + captionReport.set(s.channelSlug, r); + } + const parsedSubs: StoredSubs = []; // A tiered track that could be neither read nor kept: `subsMs` is // stored null so the next build's scan sees a change and retries. @@ -1112,6 +1204,13 @@ export async function buildIndex({ } } + // What the caption-track pass changed, per channel (CAPTION_TRACK_KEY). + for (const [slug, r] of [...captionReport].sort(([a], [b]) => (a < b ? -1 : a > b ? 1 : 0))) { + log( + `Caption track v${CAPTION_TRACK_RULE_VERSION}: ${slug}: ${r.reread} re-read, ${r.changed} now read different text, ${r.zeroToText} had no text and now do.`, + ); + } + for (const { pathKey, indexKey } of removed) { sums.remove(indexKey); cues.remove(indexKey); @@ -2332,6 +2431,9 @@ export async function buildIndex({ if (relabelDue && held.size === 0) { await meta.put(PLATFORM_LABELS_KEY, PLATFORM_LABELS_VERSION); } + if (captionTrackDue && held.size === 0) { + await meta.put(CAPTION_TRACK_KEY, CAPTION_TRACK_RULE_VERSION); + } await meta.flushed; await root.close(); @@ -2356,5 +2458,6 @@ export async function buildIndex({ removed: removed.length, shortCircuited: !sharedNeedsBuild && sitesBuilt === 0, heldChannels: [...held.keys()], + captionTrack: Object.fromEntries(captionReport), }; } diff --git a/common/controller/buildIndexCaptionTrack.test.ts b/common/controller/buildIndexCaptionTrack.test.ts @@ -0,0 +1,156 @@ +// Integration: the caption-track pass (CAPTION_TRACK_RULE_VERSION) re-reads, +// through the REAL buildIndex over a temp corpus, exactly the records an older +// caption-track rule may have read differently — one with two English VTTs, +// one whose stored cues are empty — and leaves every other record alone. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildIndexCaptionTrack.test.ts + +import { after, test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +const ROOT = mkdtempSync(path.join(tmpdir(), "build-index-captions-")); +const PINNED: Record<string, string> = { + TRANSCRIPTS_DIR: path.join(ROOT, "transcripts"), + SAVED_VIDEOS_DIR: path.join(ROOT, "saved-videos"), + SITES_DIR: path.join(ROOT, "transcripts", "sites"), + SETTINGS_FILE: path.join(ROOT, "settings.json"), + EXPORT_PUBLIC_DIR: path.join(ROOT, "public"), + EXPORT_INDEX_DIR: path.join(ROOT, ".export-index"), + EXPORT_BUILDS_DIR: path.join(ROOT, ".export-builds"), + EDITOR_CHANGELOG_FILE: path.join(ROOT, "editor-CHANGELOG.md"), + EXPORT_CHANGELOG_FILE: path.join(ROOT, "export-CHANGELOG.md"), + CHARTS_CONFIG_FILE: path.join(ROOT, "chart-templates.json"), + SEARCH_ALIASES_FILE: path.join(ROOT, "transcripts", "search-aliases.json"), + CURATED_TAGS_FILE: path.join(ROOT, "transcripts", "tags.json"), + ARCHILYZER_CONFIG_DIR: path.join(ROOT, "config"), + ARCHILYZER_SOURCE_SCRATCH: path.join(ROOT, "source-scratch"), +}; +Object.assign(process.env, PINNED); +delete process.env.ARCHILYZER_INDEX_ALLOW_HELD; +after(() => rmSync(ROOT, { recursive: true, force: true })); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { open } = await import("lmdb"); + +const paths = getPaths(); +const CHANNEL = "example-channel"; +const SITE = "testsite"; +const BOTH = "BothTracks01"; // served en (cue blocks) + en-orig (rolling) +const BLOCKS = "CueBlocks001"; // a lone cue-block en +const ROLL = "RollingOnly1"; // a lone rolling en — nothing for the pass to do + +const fixture = (name: string) => + readFileSync(path.join(import.meta.dirname, "..", "lib", "__fixtures__", name), "utf8"); +const ROLLING = fixture("vtt-rolling.vtt"); +const CUE_BLOCKS = fixture("vtt-cue-blocks.vtt"); + +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const dirOf = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); + +function seed(): void { + rmSync(paths.transcriptsDir, { recursive: true, force: true }); + rmSync(PINNED.EXPORT_INDEX_DIR, { recursive: true, force: true }); + writeFileSync(paths.settingsFile, "{}"); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "youtube", + name: CHANNEL, + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); + for (const [i, id] of [BOTH, BLOCKS, ROLL].entries()) { + writeJson(path.join(dirOf(id), "metadata.info.json"), { + id, + title: `Video ${id}`, + upload_date: `2026060${i + 1}`, + duration: 30, + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + } + writeFileSync(path.join(dirOf(BOTH), "transcript.en.vtt"), CUE_BLOCKS); + writeFileSync(path.join(dirOf(BOTH), "transcript.en-orig.vtt"), ROLLING); + writeFileSync(path.join(dirOf(BLOCKS), "transcript.en.vtt"), CUE_BLOCKS); + writeFileSync(path.join(dirOf(ROLL), "transcript.en.vtt"), ROLLING); +} + +type Summary = { id: string }; +type Cue = { start: number; end: number; text: string }; + +function withIndex<T>(fn: (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>) => T): T { + const root = open({ path: paths.lmdbPath, maxDbs: 18, compression: true }); + try { + return fn((name) => root.openDB({ name, encoding: "msgpack" })); + } finally { + root.close(); + } +} +const keyOf = (db: (name: string) => ReturnType<ReturnType<typeof open>["openDB"]>, id: string) => { + for (const { key, value } of db("sums").getRange()) { + if ((value as Summary).id === id) return key; + } + throw new Error(`${id} is not indexed`); +}; +const cuesOf = (id: string) => withIndex((db) => db("cues").get(keyOf(db, id)) as Cue[] | undefined); + +async function runIndex(): Promise<string[]> { + const log: string[] = []; + await buildIndex({ paths, onLog: (s) => log.push(s) }); + return log; +} + +test("a fresh index reads en-orig beside a served en, and a lone cue-block en as text", async () => { + seed(); + const log = await runIndex(); + assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor"); + assert.equal(cuesOf(BLOCKS)?.length, 7); + assert.equal(cuesOf(ROLL)?.length, 3); + // A first build has nothing indexed under an older rule: no pass. + assert.equal(log.some((l) => /Caption track/.test(l)), false, log.join("\n")); +}); + +test("an index built under the old rule is re-read once, for exactly the records the rule reaches", async () => { + // The index as an older build left it: the served en's text for BOTH, no + // cues for BLOCKS, and no caption-track version recorded. + const heldRoll = cuesOf(ROLL); + withIndex((db) => { + const cues = db("cues"); + cues.putSync(keyOf(db, BOTH), [{ start: 0, end: 6, text: "served words" }]); + cues.putSync(keyOf(db, BLOCKS), []); + // ROLL's record is marked so a re-read would show. + cues.putSync(keyOf(db, ROLL), [{ start: 0, end: 1, text: "untouched" }]); + db("meta").removeSync("captionTrackRule"); + }); + + const log = await runIndex(); + assert.ok(log.includes("Caption track v1: 2 record(s) re-read."), log.join("\n")); + assert.ok( + log.includes( + `Caption track v1: ${CHANNEL}: 2 re-read, 2 now read different text, 1 had no text and now do.`, + ), + log.join("\n"), + ); + assert.equal(cuesOf(BOTH)?.[0].text, "are talking about the harbor"); + assert.equal(cuesOf(BLOCKS)?.length, 7); + assert.deepEqual(cuesOf(ROLL), [{ start: 0, end: 1, text: "untouched" }]); + assert.notDeepEqual(heldRoll, cuesOf(ROLL)); + + // Recorded: the next build does not look again. + const again = await runIndex(); + assert.equal(again.some((l) => /Caption track/.test(l)), false, again.join("\n")); +});