commit 688135130b1594890d6cf24bf4f6620d00abc675
parent bc5b8f9713050b4eb9f5974e2852655e31e21d5c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 28 Sep 2026 17:59:13 -0400
common: the stats cache keys on the index's transcript record, and a transcript always has a date (stats schema 6)
statsByPath was keyed on metadata.info.json's mtime alone, while hasTranscript,
cueCount, coverage and transcribedDate come from the index and the transcript
files. A transcript that arrived after a video was first seen (Whisper days
later, or a stats run made before build:index had the video) never reached its
stat, and a caption video with no transcript.cues.json had no date at all.
- The key is now the metadata mtime AND buildIndex's `mtimes.transcriptMs`
(NOT_INDEXED = -1 when the index has no record): one LMDB get per video, no
file I/O, on the unchanged path.
- transcribedDate: outcome sidecar -> the mtime of the transcript the index
took its cues from (transcript.json, else the caption VTT) -> cues.json ->
downloadedDate. Whisper videos resolve as before; caption videos are dated by
their captions' arrival, not by a later Normalize run.
- BuildStatsResult.unindexed, and a log line, for videos the index lacks.
- STATS_SCHEMA_VERSION 5 -> 6: one full re-extraction. The page shape is
unchanged (STATS_MANIFEST_VERSION stays 1).
buildStats.test.ts runs the real buildIndex + buildStats over a temp corpus.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
3 files changed, 431 insertions(+), 22 deletions(-)
diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts
@@ -0,0 +1,323 @@
+// Integration: the stats cache, through the REAL buildIndex and buildStats, over
+// a temp corpus.
+//
+// The stats cache (`statsByPath`) used to be keyed on metadata.info.json's
+// mtime alone, while hasTranscript / cueCount / coverage / transcribedDate come
+// from the index and the transcript files. So a transcript that arrived after a
+// video was first seen never reached its stat, and a caption video (a VTT, no
+// transcript.json, no outcome sidecar) could never be dated at all. Measured on
+// a real corpus: one site served 1,889 videos and the homepage said 0
+// transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds
+// the index's transcript record, and a transcript always has a date.
+//
+// Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { createRequire, syncBuiltinESMExports } from "node:module";
+import {
+ mkdirSync,
+ mkdtempSync,
+ rmSync,
+ utimesSync,
+ writeFileSync,
+} from "node:fs";
+import { tmpdir } from "node:os";
+import path from "node:path";
+
+// getPaths() is lazy and cached, and nothing above calls it at import time, so
+// pointing the whole path graph at a temp root here isolates this file's
+// process from the real corpus (the maybeMissingBuild.test.ts pattern).
+const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public");
+process.env.EXPORT_INDEX_DIR = path.join(ROOT, ".export-index");
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+
+const { getPaths } = await import("../lib/paths");
+const { buildIndex } = await import("./buildIndex");
+const { buildStats } = await import("./buildStats");
+const { normalizeTranscript } = await import("./normalizeTranscript");
+const { readStatsPages } = await import("./poolSummary");
+
+const paths = getPaths();
+const CHANNEL = "test-channel";
+const SITE = "testsite";
+const POOL = path.join(ROOT, "pool-stats");
+
+const at = (iso: string) => new Date(iso);
+const writeJson = (file: string, value: unknown) => {
+ mkdirSync(path.dirname(file), { recursive: true });
+ writeFileSync(file, JSON.stringify(value, null, 2));
+};
+const videoDir = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id);
+const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso));
+
+// A fresh corpus (and a fresh LMDB) per test: every count below is exact.
+function resetCorpus(): void {
+ rmSync(paths.transcriptsDir, { recursive: true, force: true });
+ rmSync(path.join(ROOT, ".export-index"), { recursive: true, force: true });
+ rmSync(POOL, { recursive: true, force: true });
+ mkdirSync(paths.transcriptsDir, { recursive: true });
+ writeFileSync(paths.settingsFile, JSON.stringify({}));
+ writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), {
+ handling: "youtube",
+ name: "Test Channel",
+ url: "https://www.youtube.com/@example/videos",
+ });
+ writeJson(path.join(paths.sitesDir, SITE, "site.json"), {
+ siteId: SITE,
+ siteTitle: "Test Site",
+ siteDescription: "fixture",
+ headerTitle: "Test Site",
+ homeTagline: "",
+ socialLinks: [],
+ groups: [{ id: "default", name: "All channels", selectedByDefault: true }],
+ defaultGroupId: "default",
+ channels: [{ slug: CHANNEL, groupId: "default" }],
+ });
+}
+
+// Metadata only — what a download leaves before any transcript exists.
+function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void {
+ const file = path.join(videoDir(id), "metadata.info.json");
+ writeJson(file, {
+ id,
+ title: `Video ${id}`,
+ channel: "Test Channel",
+ upload_date: "20260601",
+ duration: 120,
+ description: "fixture",
+ webpage_url: `https://www.youtube.com/watch?v=${id}`,
+ extractor_key: "Youtube",
+ });
+ touch(file, metaIso);
+}
+
+// YouTube's own captions, as a youtube-handled download writes them. parseVtt
+// keeps only lines carrying inline timing tags (YouTube's rolling-caption
+// shape), so the fixture has them.
+function addCaptions(id: string, iso: string): void {
+ const file = path.join(videoDir(id), "transcript.en.vtt");
+ writeFileSync(
+ file,
+ "WEBVTT\nKind: captions\nLanguage: en\n\n" +
+ "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" +
+ "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n\n" +
+ "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" +
+ "Second<00:01:10.000><c> caption</c><00:01:20.000><c> line.</c>\n",
+ );
+ touch(file, iso);
+}
+
+// A Whisper-family run: transcript.json, and (unless `outcome` is false) the
+// transcribe-outcome.json sidecar every run writes beside it.
+function addWhisper(id: string, iso: string, outcome = true): void {
+ const file = path.join(videoDir(id), "transcript.json");
+ writeJson(file, {
+ duration_seconds: 120,
+ chunks: 2,
+ text: "one two",
+ chunk_data: [
+ { start_time: 0, end_time: 5, text: "Spoken line one." },
+ { start_time: 60, end_time: 110, text: "Spoken line two." },
+ ],
+ });
+ touch(file, iso);
+ if (outcome) {
+ writeJson(path.join(videoDir(id), "transcribe-outcome.json"), {
+ videoId: id,
+ transcribedAt: iso,
+ });
+ }
+}
+
+type Stat = Awaited<ReturnType<typeof readStatsPages>>[number];
+
+async function runStats(log: string[] = []) {
+ const res = await buildStats({
+ paths,
+ onLog: (s) => log.push(s),
+ wholePoolStatsDir: POOL,
+ });
+ const byId = new Map<string, Stat>(
+ (await readStatsPages(POOL)).map((s) => [s.id, s]),
+ );
+ return { res, byId, log };
+}
+
+const runIndex = () => buildIndex({ paths, onLog: () => {} });
+
+function statOf(byId: Map<string, Stat>, id: string): Stat {
+ const s = byId.get(id);
+ assert.ok(s, `${id} has a stat`);
+ return s;
+}
+
+test("(a) a transcript that arrives after the stat was cached reaches it on the next run", async () => {
+ resetCorpus();
+ seedVideo("late");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "late").hasTranscript, false);
+
+ // Whisper runs days later. metadata.info.json — the old key — is untouched.
+ addWhisper("late", "2026-07-17T05:11:14Z");
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1, "the index's transcript record moved, so the stat is redone");
+ const s = statOf(second.byId, "late");
+ assert.equal(s.hasTranscript, true);
+ assert.equal(s.cueCount, 2);
+ assert.equal(s.transcribedDate, "20260717");
+ // The MCP's "covers only N% — truncated" note reads this field. A stale
+ // record said 0 for a complete transcript.
+ assert.ok(s.coverage != null && s.coverage > 0.9, `coverage ${s.coverage}`);
+});
+
+test("(b) stats built before the index had the video heal after the index build", async () => {
+ resetCorpus();
+ seedVideo("early");
+ addCaptions("early", "2026-07-11T12:00:00Z");
+ // The pool composers run buildStats against the index as it stands; this
+ // video was downloaded after the last index build.
+ const first = await runStats();
+ assert.equal(statOf(first.byId, "early").hasTranscript, false);
+ assert.equal(first.res.unindexed, 1);
+ assert.ok(
+ first.log.some((l) => l.startsWith("1 video(s) are not in the index yet")),
+ first.log.join("\n"),
+ );
+
+ await runIndex();
+ const second = await runStats();
+ assert.equal(second.res.changed, 1);
+ assert.equal(second.res.unindexed, 0);
+ const s = statOf(second.byId, "early");
+ assert.equal(s.hasTranscript, true);
+ assert.equal(s.transcribedDate, "20260711");
+});
+
+test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => {
+ resetCorpus();
+ // Captions only: no transcript.json, no outcome sidecar, never normalized.
+ seedVideo("vtt-only");
+ addCaptions("vtt-only", "2026-07-11T12:00:00Z");
+ // Captions that arrived the same day, normalized a month later.
+ seedVideo("normalized");
+ addCaptions("normalized", "2026-07-11T12:00:00Z");
+ const norm = await normalizeTranscript({
+ videoDir: videoDir("normalized"),
+ channelSlug: CHANNEL,
+ });
+ assert.equal(norm.status, "wrote");
+ touch(path.join(videoDir("normalized"), "transcript.cues.json"), "2026-08-10T13:44:00Z");
+
+ await runIndex();
+ const { byId } = await runStats();
+ for (const id of ["vtt-only", "normalized"]) {
+ const s = statOf(byId, id);
+ assert.equal(s.hasTranscript, true, id);
+ assert.equal(s.transcribedDate, "20260711", id);
+ }
+});
+
+test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else transcript.json", async () => {
+ resetCorpus();
+ // Transcribed before the stats first saw it, with and without the sidecar.
+ seedVideo("with-outcome");
+ addWhisper("with-outcome", "2026-06-01T10:00:00Z");
+ seedVideo("no-outcome");
+ addWhisper("no-outcome", "2026-06-02T10:00:00Z", false);
+ // The sidecar wins over every mtime, even a caption file's.
+ seedVideo("hybrid");
+ addCaptions("hybrid", "2026-05-01T10:00:00Z");
+ addWhisper("hybrid", "2026-06-04T10:00:00Z");
+ // Transcribed AFTER the stats first saw it.
+ seedVideo("after");
+ await runIndex();
+ await runStats();
+ addWhisper("after", "2026-06-03T10:00:00Z");
+ await runIndex();
+ const { byId } = await runStats();
+
+ const dates = Object.fromEntries(
+ ["with-outcome", "no-outcome", "hybrid", "after"].map((id) => [
+ id,
+ statOf(byId, id).transcribedDate,
+ ]),
+ );
+ assert.deepEqual(dates, {
+ "with-outcome": "20260601",
+ "no-outcome": "20260602",
+ hybrid: "20260604",
+ after: "20260603",
+ });
+});
+
+// Every fs/promises call inside a video dir, by function name.
+function spyFs(names: ("readFile" | "stat" | "readdir" | "open")[]) {
+ const calls: { fn: string; p: string }[] = [];
+ // The CJS exports object: patching it and syncing is what reaches the named
+ // ESM imports buildStats and its helpers hold.
+ const mod = createRequire(import.meta.url)("node:fs/promises") as Record<
+ string,
+ (...a: unknown[]) => unknown
+ >;
+ const orig = new Map(names.map((n) => [n, mod[n]]));
+ for (const n of names) {
+ const fn = orig.get(n)!;
+ mod[n] = (p: unknown, ...rest: unknown[]) => {
+ calls.push({ fn: n, p: String(p) });
+ return fn(p, ...rest);
+ };
+ }
+ syncBuiltinESMExports();
+ return {
+ calls,
+ restore() {
+ for (const [n, fn] of orig) mod[n] = fn;
+ syncBuiltinESMExports();
+ },
+ };
+}
+
+test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => {
+ resetCorpus();
+ seedVideo("w");
+ addWhisper("w", "2026-06-01T10:00:00Z");
+ seedVideo("c");
+ addCaptions("c", "2026-06-01T10:00:00Z");
+ seedVideo("m");
+ await runIndex();
+ const first = await runStats();
+ assert.equal(first.res.added, 3);
+
+ addWhisper("m", "2026-06-05T10:00:00Z");
+ await runIndex();
+ const heal = await runStats();
+ assert.equal(heal.res.added, 0);
+ assert.equal(heal.res.changed, 1, "only the video whose transcript arrived");
+
+ const dataDir = path.join(paths.channelsDir, CHANNEL, "data");
+ const spy = spyFs(["readFile", "stat", "readdir", "open"]);
+ let steady;
+ try {
+ steady = await runStats();
+ } finally {
+ spy.restore();
+ }
+ assert.equal(steady.res.added, 0);
+ assert.equal(steady.res.changed, 0);
+ assert.equal(steady.res.unindexed, 0);
+ const perVideo = spy.calls.filter((c) => c.p.startsWith(dataDir + path.sep));
+ assert.deepEqual(
+ perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(),
+ [
+ `stat ${path.join("c", "metadata.info.json")}`,
+ `stat ${path.join("m", "metadata.info.json")}`,
+ `stat ${path.join("w", "metadata.info.json")}`,
+ ],
+ "the unchanged path stats each metadata file and touches nothing else in a video dir",
+ );
+});
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -2,13 +2,23 @@
// page-NNNN}.json for the viewer charts feature. Engagement metrics
// (view/like/comment counts, follower count, categories, language) live in each
// video's metadata.info.json but are NOT carried by the search index, so this
-// reads the raw metadata. Incremental: per-video mtime state is kept in a
-// dedicated `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos.
+// reads the raw metadata. Incremental: per-video state is kept in a dedicated
+// `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos.
//
// Transcript presence + cue count are read from the index LMDB `cues` sub-DB,
// and each video's visibility from the `videoState` sub-DB — both populated by
// buildIndex, so this must run after build:index, which the export prebuild
-// guarantees by chaining build:index && build:stats.
+// guarantees by chaining build:index && build:stats. The pool composers
+// (compose-hub, compose-homepage) do NOT chain it: they read the index as it
+// stands, which the cache key below makes safe.
+//
+// THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own
+// per-video transcript record (buildIndex's `mtimes.transcriptMs`, or
+// NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while
+// hasTranscript / cueCount / coverage / transcribedDate come from the index and
+// the transcript files — so a transcript that arrived after a video was first
+// seen (Whisper days later, or a stats run before build:index had the video)
+// never reached its stat, and a whole channel could publish as untranscribed.
import path from "node:path";
import { createHash } from "node:crypto";
@@ -34,6 +44,11 @@ import {
import type { VideoState } from "../lib/availability";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
import { loadTranscribeOutcome } from "../lib/transcribeOutcome-server";
+import {
+ CUES_JSON_FILENAME,
+ pickIndexTranscript,
+ readVideoFiles,
+} from "../lib/videoStatus";
import type { VideoStatus } from "../lib/stats";
import type { ChannelConfig } from "../lib/channelConfig";
import { readChannelConfigFile } from "./channels";
@@ -51,7 +66,16 @@ import {
type IndexKey = [string, string, string];
type PathKey = [string, string];
-type StatsRecord = { metaMs: number; stat: VideoStat };
+
+// `idxMs` is the index's per-video transcript record as this stat saw it — see
+// indexTranscriptMs. Optional only because a record written before schema 6
+// has none; the schema bump clears those, and a missing value compares
+// unequal to every real one, so such a record would be recomputed anyway.
+type StatsRecord = { metaMs: number; idxMs?: number | null; stat: VideoStat };
+
+// buildIndex has no `mtimes` record for this video yet: it was downloaded after
+// the last index build. Distinct from `null` (indexed, no transcript file).
+const NOT_INDEXED = -1;
type ScanEntry = {
channelSlug: string;
@@ -68,6 +92,10 @@ export type BuildStatsResult = {
added: number;
changed: number;
removed: number;
+ // Videos on disk that the index does not have yet. Their stats say "no
+ // transcript" until the first run after the next index build, which
+ // recomputes them (the key moves from NOT_INDEXED).
+ unindexed: number;
pagesWritten: number;
shortCircuited: boolean;
durationMs: number;
@@ -116,6 +144,22 @@ async function fileMtimeMs(p: string): Promise<number | null> {
// "content added over time" progress charts. Prefers the explicit outcome
// sidecars (reliable across the shard rsync model, where file mtimes drift);
// falls back to file mtimes for content added before the sidecars existed.
+//
+// A TRANSCRIPT ALWAYS HAS A DATE (schema 6). The fallbacks, in order:
+// 1. transcribe-outcome.json's `transcribedAt` — every Whisper run writes it;
+// 2. the mtime of the transcript the index takes its cues from
+// (pickIndexTranscript: transcript.json, else the caption VTT);
+// 3. the mtime of transcript.cues.json;
+// 4. downloadedDate, which always resolves.
+// A Whisper video resolves exactly as before: 1, else transcript.json's mtime,
+// which is what (2) picks whenever transcript.json exists (the one difference:
+// a sidecar whose `transcribedAt` will not parse used to leave no date, and now
+// falls through to 2). A CAPTION-handled
+// video is dated by when its captions ARRIVED — the VTT's mtime — not by a later
+// Normalize run: (3) used to be the only file that could date one, so a
+// caption video either had no date (never normalized) or took the Normalize
+// run's date (1,683 of one channel's, all on one day). The homepage fold, the
+// charts and the recent rail all need `hasTranscript` ⇒ `transcribedDate`.
async function resolveAcquisitionDates(
videoDir: string,
metaMs: number,
@@ -128,13 +172,13 @@ async function resolveAcquisitionDates(
let transcribedDate: string | null = null;
if (hasTranscript) {
const tr = await loadTranscribeOutcome(videoDir);
- if (tr?.transcribedAt) {
- transcribedDate = ymdFromIso(tr.transcribedAt);
- } else {
+ transcribedDate = tr?.transcribedAt ? ymdFromIso(tr.transcribedAt) : null;
+ if (transcribedDate === null) {
+ const picked = pickIndexTranscript(await readVideoFiles(videoDir));
const mtime =
- (await fileMtimeMs(path.join(videoDir, "transcript.json"))) ??
- (await fileMtimeMs(path.join(videoDir, "transcript.cues.json")));
- transcribedDate = mtime != null ? ymdFromMs(mtime) : null;
+ (picked ? await fileMtimeMs(path.join(videoDir, picked.filename)) : null) ??
+ (await fileMtimeMs(path.join(videoDir, CUES_JSON_FILENAME)));
+ transcribedDate = (mtime != null ? ymdFromMs(mtime) : null) ?? downloadedDate;
}
}
return { downloadedDate, transcribedDate };
@@ -266,6 +310,21 @@ export async function buildStats({
encoding: "msgpack",
});
const meta = root.openDB<unknown, string>({ name: "statsMeta", encoding: "msgpack" });
+ // Read-only view of buildIndex's per-video mtime record, keyed like
+ // statsByPath. Only `transcriptMs` is read: the mtime of the transcript file
+ // the index took this video's cues from (null when it had none). buildIndex
+ // rewrites a video's `cues` when that number moves (buildIndex.ts, the
+ // added/changed diff), so it is exactly the "has the transcript this stat was
+ // computed from changed" signal — at the cost of one LMDB get per video, and
+ // no file I/O, on the unchanged path.
+ const indexMtimes = root.openDB<{ transcriptMs: number | null }, PathKey>({
+ name: "mtimes",
+ encoding: "msgpack",
+ });
+ const indexTranscriptMs = (k: PathKey): number | null => {
+ const rec = indexMtimes.get(k);
+ return rec ? (rec.transcriptMs ?? null) : NOT_INDEXED;
+ };
const storedSchema = meta.get("schema") as number | undefined;
const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION;
@@ -283,19 +342,35 @@ export async function buildStats({
const liveIds = new Set(
entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])),
);
- const toProcess: ScanEntry[] = [];
+ const toProcess: { e: ScanEntry; idxMs: number | null }[] = [];
let added = 0;
let changed = 0;
+ let unindexed = 0;
for (const e of entries) {
- const prev = statsByPath.get([e.channelSlug, e.videoDir]);
+ const pk: PathKey = [e.channelSlug, e.videoDir];
+ const idxMs = indexTranscriptMs(pk);
+ if (idxMs === NOT_INDEXED) unindexed++;
+ const prev = statsByPath.get(pk);
if (!prev) {
added++;
- toProcess.push(e);
- } else if (prev.metaMs !== e.metaMs) {
+ toProcess.push({ e, idxMs });
+ } else if (prev.metaMs !== e.metaMs || prev.idxMs !== idxMs) {
+ // The second half is the fix for stats frozen at first sight: a
+ // transcript that arrives later moves transcriptMs (and a video first
+ // seen before the index had it moves off NOT_INDEXED), while the
+ // metadata file — the old key's only input — is never touched.
changed++;
- toProcess.push(e);
+ toProcess.push({ e, idxMs });
}
}
+ if (unindexed > 0) {
+ // Not a failure: the pool composers read the index as it stands, and the
+ // editor downloads between index builds. Said so a published number that
+ // lags the disk has its reason in the log.
+ log(
+ `${unindexed} video(s) are not in the index yet; their transcripts reach the stats on the first run after the next index build.`,
+ );
+ }
const removedKeys: PathKey[] = [];
for (const { key } of statsByPath.getRange()) {
const k = key as PathKey;
@@ -308,7 +383,7 @@ export async function buildStats({
signal?.throwIfAborted();
const slice = toProcess.slice(i, i + BATCH);
await Promise.all(
- slice.map(async (e) => {
+ slice.map(async ({ e, idxMs }) => {
try {
const metaRaw = await readFile(e.metaPath, "utf8");
const parsedMeta = JSON.parse(metaRaw) as RawMetadata;
@@ -325,10 +400,11 @@ export async function buildStats({
const coverage = transcriptCoverage(cueList, base.duration).coverage;
// Placeholder. `status` is NOT a cached field any more: it is applied
// from buildIndex's `videoState` sub-DB at collection time below.
- // This record is keyed on metadata mtime alone, so caching a status
- // here meant a pure availability flip only reached the chart on the
- // next metadata touch or schema bump — a video deleted today did not
- // show as deleted today.
+ // This record's key does not see availability (metadata mtime and
+ // the index's transcript mtime only), so caching a status here meant
+ // a pure availability flip only reached the chart on the next
+ // metadata touch or schema bump — a video deleted today did not show
+ // as deleted today.
const status: VideoStatus = "available";
const hasTranscript = cueCount != null && cueCount > 0;
const { downloadedDate, transcribedDate } =
@@ -353,6 +429,7 @@ export async function buildStats({
);
await statsByPath.put([e.channelSlug, e.videoDir], {
metaMs: e.metaMs,
+ idxMs,
stat,
});
} catch (err) {
@@ -555,6 +632,7 @@ export async function buildStats({
added,
changed,
removed,
+ unindexed,
pagesWritten: aggregatePages,
shortCircuited: needBuild.length === 0,
durationMs: Date.now() - t0,
diff --git a/common/lib/stats.ts b/common/lib/stats.ts
@@ -3,8 +3,12 @@ import { pageFileName } from "./manifest";
import type { VideoState } from "./availability";
// Bumping this invalidates the LMDB `statsByPath` incremental cache and forces
-// a full re-extraction (e.g. when a new field is added below).
-export const STATS_SCHEMA_VERSION = 5;
+// a full re-extraction (e.g. when a new field is added below). It versions the
+// CACHE, not the published pages: STATS_MANIFEST_VERSION is theirs.
+// 6 — the cache key gained the index's transcript record, and a transcript
+// always has a `transcribedDate` (caption videos: the VTT's arrival).
+// The page shape did not change.
+export const STATS_SCHEMA_VERSION = 6;
export const STATS_MANIFEST_VERSION = 1;
// Visibility of a video on its source platform. One type with the viewer's, so
@@ -38,6 +42,10 @@ export type VideoStat = {
// "content added over time" progress charts; the time X-axis can bin on any of
// these date fields.
downloadedDate: string | null;
+ // Non-null whenever `hasTranscript` is, since schema 6 (a caption video takes
+ // its captions' arrival). A page written by an older build can still carry a
+ // transcript with a null date: readers count it and only leave it off a time
+ // axis (homepageSummary does exactly that).
transcribedDate: string | null;
timestamp: number | null; // unix seconds
duration: number; // seconds