Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 688135130b1594890d6cf24bf4f6620d00abc675
parent bc5b8f9713050b4eb9f5974e2852655e31e21d5c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 28 Sep 2026 17:59:13 -0400

common: the stats cache keys on the index's transcript record, and a transcript always has a date (stats schema 6)

statsByPath was keyed on metadata.info.json's mtime alone, while hasTranscript,
cueCount, coverage and transcribedDate come from the index and the transcript
files. A transcript that arrived after a video was first seen (Whisper days
later, or a stats run made before build:index had the video) never reached its
stat, and a caption video with no transcript.cues.json had no date at all.

- The key is now the metadata mtime AND buildIndex's `mtimes.transcriptMs`
  (NOT_INDEXED = -1 when the index has no record): one LMDB get per video, no
  file I/O, on the unchanged path.
- transcribedDate: outcome sidecar -> the mtime of the transcript the index
  took its cues from (transcript.json, else the caption VTT) -> cues.json ->
  downloadedDate. Whisper videos resolve as before; caption videos are dated by
  their captions' arrival, not by a later Normalize run.
- BuildStatsResult.unindexed, and a log line, for videos the index lacks.
- STATS_SCHEMA_VERSION 5 -> 6: one full re-extraction. The page shape is
  unchanged (STATS_MANIFEST_VERSION stays 1).

buildStats.test.ts runs the real buildIndex + buildStats over a temp corpus.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/controller/buildStats.test.ts | 323+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/buildStats.ts | 118+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------
Mcommon/lib/stats.ts | 12++++++++++--
3 files changed, 431 insertions(+), 22 deletions(-)

diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts @@ -0,0 +1,323 @@ +// Integration: the stats cache, through the REAL buildIndex and buildStats, over +// a temp corpus. +// +// The stats cache (`statsByPath`) used to be keyed on metadata.info.json's +// mtime alone, while hasTranscript / cueCount / coverage / transcribedDate come +// from the index and the transcript files. So a transcript that arrived after a +// video was first seen never reached its stat, and a caption video (a VTT, no +// transcript.json, no outcome sidecar) could never be dated at all. Measured on +// a real corpus: one site served 1,889 videos and the homepage said 0 +// transcripts, 0 channels, 0 hours. These cases pin the fix: the key also holds +// the index's transcript record, and a transcript always has a date. +// +// Run with: node_modules/.bin/tsx --test common/controller/buildStats.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createRequire, syncBuiltinESMExports } from "node:module"; +import { + mkdirSync, + mkdtempSync, + rmSync, + utimesSync, + writeFileSync, +} from "node:fs"; +import { tmpdir } from "node:os"; +import path from "node:path"; + +// getPaths() is lazy and cached, and nothing above calls it at import time, so +// pointing the whole path graph at a temp root here isolates this file's +// process from the real corpus (the maybeMissingBuild.test.ts pattern). +const ROOT = mkdtempSync(path.join(tmpdir(), "build-stats-")); +process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); +process.env.EXPORT_PUBLIC_DIR = path.join(ROOT, "public"); +process.env.EXPORT_INDEX_DIR = path.join(ROOT, ".export-index"); +process.env.SETTINGS_FILE = path.join(ROOT, "settings.json"); + +const { getPaths } = await import("../lib/paths"); +const { buildIndex } = await import("./buildIndex"); +const { buildStats } = await import("./buildStats"); +const { normalizeTranscript } = await import("./normalizeTranscript"); +const { readStatsPages } = await import("./poolSummary"); + +const paths = getPaths(); +const CHANNEL = "test-channel"; +const SITE = "testsite"; +const POOL = path.join(ROOT, "pool-stats"); + +const at = (iso: string) => new Date(iso); +const writeJson = (file: string, value: unknown) => { + mkdirSync(path.dirname(file), { recursive: true }); + writeFileSync(file, JSON.stringify(value, null, 2)); +}; +const videoDir = (id: string) => path.join(paths.channelsDir, CHANNEL, "data", id); +const touch = (file: string, iso: string) => utimesSync(file, at(iso), at(iso)); + +// A fresh corpus (and a fresh LMDB) per test: every count below is exact. +function resetCorpus(): void { + rmSync(paths.transcriptsDir, { recursive: true, force: true }); + rmSync(path.join(ROOT, ".export-index"), { recursive: true, force: true }); + rmSync(POOL, { recursive: true, force: true }); + mkdirSync(paths.transcriptsDir, { recursive: true }); + writeFileSync(paths.settingsFile, JSON.stringify({})); + writeJson(path.join(paths.channelsDir, CHANNEL, "config.json"), { + handling: "youtube", + name: "Test Channel", + url: "https://www.youtube.com/@example/videos", + }); + writeJson(path.join(paths.sitesDir, SITE, "site.json"), { + siteId: SITE, + siteTitle: "Test Site", + siteDescription: "fixture", + headerTitle: "Test Site", + homeTagline: "", + socialLinks: [], + groups: [{ id: "default", name: "All channels", selectedByDefault: true }], + defaultGroupId: "default", + channels: [{ slug: CHANNEL, groupId: "default" }], + }); +} + +// Metadata only — what a download leaves before any transcript exists. +function seedVideo(id: string, metaIso = "2026-07-11T11:00:00Z"): void { + const file = path.join(videoDir(id), "metadata.info.json"); + writeJson(file, { + id, + title: `Video ${id}`, + channel: "Test Channel", + upload_date: "20260601", + duration: 120, + description: "fixture", + webpage_url: `https://www.youtube.com/watch?v=${id}`, + extractor_key: "Youtube", + }); + touch(file, metaIso); +} + +// YouTube's own captions, as a youtube-handled download writes them. parseVtt +// keeps only lines carrying inline timing tags (YouTube's rolling-caption +// shape), so the fixture has them. +function addCaptions(id: string, iso: string): void { + const file = path.join(videoDir(id), "transcript.en.vtt"); + writeFileSync( + file, + "WEBVTT\nKind: captions\nLanguage: en\n\n" + + "00:00:00.000 --> 00:00:05.000 align:start position:0%\n" + + "First<00:00:01.000><c> caption</c><00:00:02.000><c> line.</c>\n\n" + + "00:01:00.000 --> 00:01:50.000 align:start position:0%\n" + + "Second<00:01:10.000><c> caption</c><00:01:20.000><c> line.</c>\n", + ); + touch(file, iso); +} + +// A Whisper-family run: transcript.json, and (unless `outcome` is false) the +// transcribe-outcome.json sidecar every run writes beside it. +function addWhisper(id: string, iso: string, outcome = true): void { + const file = path.join(videoDir(id), "transcript.json"); + writeJson(file, { + duration_seconds: 120, + chunks: 2, + text: "one two", + chunk_data: [ + { start_time: 0, end_time: 5, text: "Spoken line one." }, + { start_time: 60, end_time: 110, text: "Spoken line two." }, + ], + }); + touch(file, iso); + if (outcome) { + writeJson(path.join(videoDir(id), "transcribe-outcome.json"), { + videoId: id, + transcribedAt: iso, + }); + } +} + +type Stat = Awaited<ReturnType<typeof readStatsPages>>[number]; + +async function runStats(log: string[] = []) { + const res = await buildStats({ + paths, + onLog: (s) => log.push(s), + wholePoolStatsDir: POOL, + }); + const byId = new Map<string, Stat>( + (await readStatsPages(POOL)).map((s) => [s.id, s]), + ); + return { res, byId, log }; +} + +const runIndex = () => buildIndex({ paths, onLog: () => {} }); + +function statOf(byId: Map<string, Stat>, id: string): Stat { + const s = byId.get(id); + assert.ok(s, `${id} has a stat`); + return s; +} + +test("(a) a transcript that arrives after the stat was cached reaches it on the next run", async () => { + resetCorpus(); + seedVideo("late"); + await runIndex(); + const first = await runStats(); + assert.equal(statOf(first.byId, "late").hasTranscript, false); + + // Whisper runs days later. metadata.info.json — the old key — is untouched. + addWhisper("late", "2026-07-17T05:11:14Z"); + await runIndex(); + const second = await runStats(); + assert.equal(second.res.changed, 1, "the index's transcript record moved, so the stat is redone"); + const s = statOf(second.byId, "late"); + assert.equal(s.hasTranscript, true); + assert.equal(s.cueCount, 2); + assert.equal(s.transcribedDate, "20260717"); + // The MCP's "covers only N% — truncated" note reads this field. A stale + // record said 0 for a complete transcript. + assert.ok(s.coverage != null && s.coverage > 0.9, `coverage ${s.coverage}`); +}); + +test("(b) stats built before the index had the video heal after the index build", async () => { + resetCorpus(); + seedVideo("early"); + addCaptions("early", "2026-07-11T12:00:00Z"); + // The pool composers run buildStats against the index as it stands; this + // video was downloaded after the last index build. + const first = await runStats(); + assert.equal(statOf(first.byId, "early").hasTranscript, false); + assert.equal(first.res.unindexed, 1); + assert.ok( + first.log.some((l) => l.startsWith("1 video(s) are not in the index yet")), + first.log.join("\n"), + ); + + await runIndex(); + const second = await runStats(); + assert.equal(second.res.changed, 1); + assert.equal(second.res.unindexed, 0); + const s = statOf(second.byId, "early"); + assert.equal(s.hasTranscript, true); + assert.equal(s.transcribedDate, "20260711"); +}); + +test("(c) a caption-only video is dated by its captions' arrival, not by a later Normalize", async () => { + resetCorpus(); + // Captions only: no transcript.json, no outcome sidecar, never normalized. + seedVideo("vtt-only"); + addCaptions("vtt-only", "2026-07-11T12:00:00Z"); + // Captions that arrived the same day, normalized a month later. + seedVideo("normalized"); + addCaptions("normalized", "2026-07-11T12:00:00Z"); + const norm = await normalizeTranscript({ + videoDir: videoDir("normalized"), + channelSlug: CHANNEL, + }); + assert.equal(norm.status, "wrote"); + touch(path.join(videoDir("normalized"), "transcript.cues.json"), "2026-08-10T13:44:00Z"); + + await runIndex(); + const { byId } = await runStats(); + for (const id of ["vtt-only", "normalized"]) { + const s = statOf(byId, id); + assert.equal(s.hasTranscript, true, id); + assert.equal(s.transcribedDate, "20260711", id); + } +}); + +test("(d) a Whisper video resolves exactly as before: the outcome sidecar, else transcript.json", async () => { + resetCorpus(); + // Transcribed before the stats first saw it, with and without the sidecar. + seedVideo("with-outcome"); + addWhisper("with-outcome", "2026-06-01T10:00:00Z"); + seedVideo("no-outcome"); + addWhisper("no-outcome", "2026-06-02T10:00:00Z", false); + // The sidecar wins over every mtime, even a caption file's. + seedVideo("hybrid"); + addCaptions("hybrid", "2026-05-01T10:00:00Z"); + addWhisper("hybrid", "2026-06-04T10:00:00Z"); + // Transcribed AFTER the stats first saw it. + seedVideo("after"); + await runIndex(); + await runStats(); + addWhisper("after", "2026-06-03T10:00:00Z"); + await runIndex(); + const { byId } = await runStats(); + + const dates = Object.fromEntries( + ["with-outcome", "no-outcome", "hybrid", "after"].map((id) => [ + id, + statOf(byId, id).transcribedDate, + ]), + ); + assert.deepEqual(dates, { + "with-outcome": "20260601", + "no-outcome": "20260602", + hybrid: "20260604", + after: "20260603", + }); +}); + +// Every fs/promises call inside a video dir, by function name. +function spyFs(names: ("readFile" | "stat" | "readdir" | "open")[]) { + const calls: { fn: string; p: string }[] = []; + // The CJS exports object: patching it and syncing is what reaches the named + // ESM imports buildStats and its helpers hold. + const mod = createRequire(import.meta.url)("node:fs/promises") as Record< + string, + (...a: unknown[]) => unknown + >; + const orig = new Map(names.map((n) => [n, mod[n]])); + for (const n of names) { + const fn = orig.get(n)!; + mod[n] = (p: unknown, ...rest: unknown[]) => { + calls.push({ fn: n, p: String(p) }); + return fn(p, ...rest); + }; + } + syncBuiltinESMExports(); + return { + calls, + restore() { + for (const [n, fn] of orig) mod[n] = fn; + syncBuiltinESMExports(); + }, + }; +} + +test("(e) the key does not churn: a heal redoes one stat, then an unchanged run reads nothing per video", async () => { + resetCorpus(); + seedVideo("w"); + addWhisper("w", "2026-06-01T10:00:00Z"); + seedVideo("c"); + addCaptions("c", "2026-06-01T10:00:00Z"); + seedVideo("m"); + await runIndex(); + const first = await runStats(); + assert.equal(first.res.added, 3); + + addWhisper("m", "2026-06-05T10:00:00Z"); + await runIndex(); + const heal = await runStats(); + assert.equal(heal.res.added, 0); + assert.equal(heal.res.changed, 1, "only the video whose transcript arrived"); + + const dataDir = path.join(paths.channelsDir, CHANNEL, "data"); + const spy = spyFs(["readFile", "stat", "readdir", "open"]); + let steady; + try { + steady = await runStats(); + } finally { + spy.restore(); + } + assert.equal(steady.res.added, 0); + assert.equal(steady.res.changed, 0); + assert.equal(steady.res.unindexed, 0); + const perVideo = spy.calls.filter((c) => c.p.startsWith(dataDir + path.sep)); + assert.deepEqual( + perVideo.map((c) => `${c.fn} ${path.relative(dataDir, c.p)}`).sort(), + [ + `stat ${path.join("c", "metadata.info.json")}`, + `stat ${path.join("m", "metadata.info.json")}`, + `stat ${path.join("w", "metadata.info.json")}`, + ], + "the unchanged path stats each metadata file and touches nothing else in a video dir", + ); +}); diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts @@ -2,13 +2,23 @@ // page-NNNN}.json for the viewer charts feature. Engagement metrics // (view/like/comment counts, follower count, categories, language) live in each // video's metadata.info.json but are NOT carried by the search index, so this -// reads the raw metadata. Incremental: per-video mtime state is kept in a -// dedicated `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos. +// reads the raw metadata. Incremental: per-video state is kept in a dedicated +// `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos. // // Transcript presence + cue count are read from the index LMDB `cues` sub-DB, // and each video's visibility from the `videoState` sub-DB — both populated by // buildIndex, so this must run after build:index, which the export prebuild -// guarantees by chaining build:index && build:stats. +// guarantees by chaining build:index && build:stats. The pool composers +// (compose-hub, compose-homepage) do NOT chain it: they read the index as it +// stands, which the cache key below makes safe. +// +// THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own +// per-video transcript record (buildIndex's `mtimes.transcriptMs`, or +// NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while +// hasTranscript / cueCount / coverage / transcribedDate come from the index and +// the transcript files — so a transcript that arrived after a video was first +// seen (Whisper days later, or a stats run before build:index had the video) +// never reached its stat, and a whole channel could publish as untranscribed. import path from "node:path"; import { createHash } from "node:crypto"; @@ -34,6 +44,11 @@ import { import type { VideoState } from "../lib/availability"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; import { loadTranscribeOutcome } from "../lib/transcribeOutcome-server"; +import { + CUES_JSON_FILENAME, + pickIndexTranscript, + readVideoFiles, +} from "../lib/videoStatus"; import type { VideoStatus } from "../lib/stats"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfigFile } from "./channels"; @@ -51,7 +66,16 @@ import { type IndexKey = [string, string, string]; type PathKey = [string, string]; -type StatsRecord = { metaMs: number; stat: VideoStat }; + +// `idxMs` is the index's per-video transcript record as this stat saw it — see +// indexTranscriptMs. Optional only because a record written before schema 6 +// has none; the schema bump clears those, and a missing value compares +// unequal to every real one, so such a record would be recomputed anyway. +type StatsRecord = { metaMs: number; idxMs?: number | null; stat: VideoStat }; + +// buildIndex has no `mtimes` record for this video yet: it was downloaded after +// the last index build. Distinct from `null` (indexed, no transcript file). +const NOT_INDEXED = -1; type ScanEntry = { channelSlug: string; @@ -68,6 +92,10 @@ export type BuildStatsResult = { added: number; changed: number; removed: number; + // Videos on disk that the index does not have yet. Their stats say "no + // transcript" until the first run after the next index build, which + // recomputes them (the key moves from NOT_INDEXED). + unindexed: number; pagesWritten: number; shortCircuited: boolean; durationMs: number; @@ -116,6 +144,22 @@ async function fileMtimeMs(p: string): Promise<number | null> { // "content added over time" progress charts. Prefers the explicit outcome // sidecars (reliable across the shard rsync model, where file mtimes drift); // falls back to file mtimes for content added before the sidecars existed. +// +// A TRANSCRIPT ALWAYS HAS A DATE (schema 6). The fallbacks, in order: +// 1. transcribe-outcome.json's `transcribedAt` — every Whisper run writes it; +// 2. the mtime of the transcript the index takes its cues from +// (pickIndexTranscript: transcript.json, else the caption VTT); +// 3. the mtime of transcript.cues.json; +// 4. downloadedDate, which always resolves. +// A Whisper video resolves exactly as before: 1, else transcript.json's mtime, +// which is what (2) picks whenever transcript.json exists (the one difference: +// a sidecar whose `transcribedAt` will not parse used to leave no date, and now +// falls through to 2). A CAPTION-handled +// video is dated by when its captions ARRIVED — the VTT's mtime — not by a later +// Normalize run: (3) used to be the only file that could date one, so a +// caption video either had no date (never normalized) or took the Normalize +// run's date (1,683 of one channel's, all on one day). The homepage fold, the +// charts and the recent rail all need `hasTranscript` ⇒ `transcribedDate`. async function resolveAcquisitionDates( videoDir: string, metaMs: number, @@ -128,13 +172,13 @@ async function resolveAcquisitionDates( let transcribedDate: string | null = null; if (hasTranscript) { const tr = await loadTranscribeOutcome(videoDir); - if (tr?.transcribedAt) { - transcribedDate = ymdFromIso(tr.transcribedAt); - } else { + transcribedDate = tr?.transcribedAt ? ymdFromIso(tr.transcribedAt) : null; + if (transcribedDate === null) { + const picked = pickIndexTranscript(await readVideoFiles(videoDir)); const mtime = - (await fileMtimeMs(path.join(videoDir, "transcript.json"))) ?? - (await fileMtimeMs(path.join(videoDir, "transcript.cues.json"))); - transcribedDate = mtime != null ? ymdFromMs(mtime) : null; + (picked ? await fileMtimeMs(path.join(videoDir, picked.filename)) : null) ?? + (await fileMtimeMs(path.join(videoDir, CUES_JSON_FILENAME))); + transcribedDate = (mtime != null ? ymdFromMs(mtime) : null) ?? downloadedDate; } } return { downloadedDate, transcribedDate }; @@ -266,6 +310,21 @@ export async function buildStats({ encoding: "msgpack", }); const meta = root.openDB<unknown, string>({ name: "statsMeta", encoding: "msgpack" }); + // Read-only view of buildIndex's per-video mtime record, keyed like + // statsByPath. Only `transcriptMs` is read: the mtime of the transcript file + // the index took this video's cues from (null when it had none). buildIndex + // rewrites a video's `cues` when that number moves (buildIndex.ts, the + // added/changed diff), so it is exactly the "has the transcript this stat was + // computed from changed" signal — at the cost of one LMDB get per video, and + // no file I/O, on the unchanged path. + const indexMtimes = root.openDB<{ transcriptMs: number | null }, PathKey>({ + name: "mtimes", + encoding: "msgpack", + }); + const indexTranscriptMs = (k: PathKey): number | null => { + const rec = indexMtimes.get(k); + return rec ? (rec.transcriptMs ?? null) : NOT_INDEXED; + }; const storedSchema = meta.get("schema") as number | undefined; const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION; @@ -283,19 +342,35 @@ export async function buildStats({ const liveIds = new Set( entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])), ); - const toProcess: ScanEntry[] = []; + const toProcess: { e: ScanEntry; idxMs: number | null }[] = []; let added = 0; let changed = 0; + let unindexed = 0; for (const e of entries) { - const prev = statsByPath.get([e.channelSlug, e.videoDir]); + const pk: PathKey = [e.channelSlug, e.videoDir]; + const idxMs = indexTranscriptMs(pk); + if (idxMs === NOT_INDEXED) unindexed++; + const prev = statsByPath.get(pk); if (!prev) { added++; - toProcess.push(e); - } else if (prev.metaMs !== e.metaMs) { + toProcess.push({ e, idxMs }); + } else if (prev.metaMs !== e.metaMs || prev.idxMs !== idxMs) { + // The second half is the fix for stats frozen at first sight: a + // transcript that arrives later moves transcriptMs (and a video first + // seen before the index had it moves off NOT_INDEXED), while the + // metadata file — the old key's only input — is never touched. changed++; - toProcess.push(e); + toProcess.push({ e, idxMs }); } } + if (unindexed > 0) { + // Not a failure: the pool composers read the index as it stands, and the + // editor downloads between index builds. Said so a published number that + // lags the disk has its reason in the log. + log( + `${unindexed} video(s) are not in the index yet; their transcripts reach the stats on the first run after the next index build.`, + ); + } const removedKeys: PathKey[] = []; for (const { key } of statsByPath.getRange()) { const k = key as PathKey; @@ -308,7 +383,7 @@ export async function buildStats({ signal?.throwIfAborted(); const slice = toProcess.slice(i, i + BATCH); await Promise.all( - slice.map(async (e) => { + slice.map(async ({ e, idxMs }) => { try { const metaRaw = await readFile(e.metaPath, "utf8"); const parsedMeta = JSON.parse(metaRaw) as RawMetadata; @@ -325,10 +400,11 @@ export async function buildStats({ const coverage = transcriptCoverage(cueList, base.duration).coverage; // Placeholder. `status` is NOT a cached field any more: it is applied // from buildIndex's `videoState` sub-DB at collection time below. - // This record is keyed on metadata mtime alone, so caching a status - // here meant a pure availability flip only reached the chart on the - // next metadata touch or schema bump — a video deleted today did not - // show as deleted today. + // This record's key does not see availability (metadata mtime and + // the index's transcript mtime only), so caching a status here meant + // a pure availability flip only reached the chart on the next + // metadata touch or schema bump — a video deleted today did not show + // as deleted today. const status: VideoStatus = "available"; const hasTranscript = cueCount != null && cueCount > 0; const { downloadedDate, transcribedDate } = @@ -353,6 +429,7 @@ export async function buildStats({ ); await statsByPath.put([e.channelSlug, e.videoDir], { metaMs: e.metaMs, + idxMs, stat, }); } catch (err) { @@ -555,6 +632,7 @@ export async function buildStats({ added, changed, removed, + unindexed, pagesWritten: aggregatePages, shortCircuited: needBuild.length === 0, durationMs: Date.now() - t0, diff --git a/common/lib/stats.ts b/common/lib/stats.ts @@ -3,8 +3,12 @@ import { pageFileName } from "./manifest"; import type { VideoState } from "./availability"; // Bumping this invalidates the LMDB `statsByPath` incremental cache and forces -// a full re-extraction (e.g. when a new field is added below). -export const STATS_SCHEMA_VERSION = 5; +// a full re-extraction (e.g. when a new field is added below). It versions the +// CACHE, not the published pages: STATS_MANIFEST_VERSION is theirs. +// 6 — the cache key gained the index's transcript record, and a transcript +// always has a `transcribedDate` (caption videos: the VTT's arrival). +// The page shape did not change. +export const STATS_SCHEMA_VERSION = 6; export const STATS_MANIFEST_VERSION = 1; // Visibility of a video on its source platform. One type with the viewer's, so @@ -38,6 +42,10 @@ export type VideoStat = { // "content added over time" progress charts; the time X-axis can bin on any of // these date fields. downloadedDate: string | null; + // Non-null whenever `hasTranscript` is, since schema 6 (a caption video takes + // its captions' arrival). A page written by an older build can still carry a + // transcript with a null date: readers count it and only leave it off a time + // axis (homepageSummary does exactly that). transcribedDate: string | null; timestamp: number | null; // unix seconds duration: number; // seconds