// Extracts a compact per-video stats dataset into /{manifest, // page-NNNN}.json for the viewer charts feature. Engagement metrics // (view/like/comment counts, follower count, categories, language) live in each // video's metadata.info.json but are NOT carried by the search index, so this // reads the raw metadata. Incremental: per-video state is kept in a dedicated // `statsByPath` LMDB sub-DB so re-runs only re-parse changed videos. // // Transcript presence + cue count are read from the index LMDB `cues` sub-DB, // and each video's visibility from the `videoState` sub-DB — both populated by // buildIndex, so this must run after build:index, which the export prebuild // guarantees by chaining build:index && build:stats. The pool composers // (compose-hub, compose-homepage) do NOT chain it: they read the index as it // stands, which the cache key below makes safe. // // THE CACHE KEY IS TWO THINGS: the metadata file's mtime AND the index's own // per-video record (buildIndex's `mtimes` entry, as indexSignature, or // NOT_INDEXED). Until schema 6 it was the metadata mtime alone, while // hasTranscript / cueCount / coverage / transcribedDate come from the index and // the transcript files — so a transcript that arrived after a video was first // seen (Whisper days later, or a stats run before build:index had the video) // never reached its stat, and a whole channel could publish as untranscribed. // // A CHANNEL WHOSE MEDIA IS NOT REACHABLE (a relocated `data/` on an unmounted // drive, or one mid-relocation) is not rescanned: its cached stats are kept as // they are, rather than read as a channel with no videos and removed. This is // the stats build's own guard — the job registry's `needsMedia` check is per // channel, and this build is pool-wide — so the editor job and the CLI share it. // // ONE STATS BUILD AT A TIME. Nothing here takes a lock: two runs at once are // harmless unless one of them clears the cache (a schema change) after the // other scanned, when the other can publish truncated pages. The editor's build // jobs share the "build" queue by default; the CLI is outside every queue. import path from "node:path"; import { createHash } from "node:crypto"; import { mkdir, readdir, readFile, rename, rm, stat, writeFile, } from "node:fs/promises"; import type { Dirent } from "node:fs"; import { open } from "lmdb"; import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server"; import type { Cue } from "../lib/vtt"; import { transcriptCoverage } from "../lib/transcriptCoverage"; import { summarize, summarizeStats, type RawMetadata, } from "../lib/transcripts-server"; import type { VideoState } from "../lib/availability"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; import { loadTranscribeOutcome } from "../lib/transcribeOutcome-server"; import { CUES_JSON_FILENAME, pickIndexTranscript, readVideoFiles, } from "../lib/videoStatus"; import type { VideoStatus } from "../lib/stats"; import type { ChannelConfig } from "../lib/channelConfig"; import { readChannelConfigFile } from "./channels"; import { inspectChannelMedia } from "../lib/channelMedia"; import { HELD_WAYS_OUT, describeHeld, heldReason, } from "../lib/channelMediaHold"; import { getSettings } from "../lib/settings"; import type { Paths } from "../lib/paths"; import { channelsOnlyOnUnlistedSites, listSites, siteStatsDir, } from "../lib/site"; import { INDEX_SCANNED_AT_KEY, STATS_SCHEMA_VERSION, STATS_MANIFEST_VERSION, STATS_MAX_PAGE_BYTES, statsPageFileName, type VideoStat, type StatsManifest, type StatsChannelEntry, } from "../lib/stats"; type IndexKey = [string, string, string]; type PathKey = [string, string]; // `idx` is the index's per-video record as this stat saw it (indexSignature). // Optional only because a record written before schema 6 has none; the schema // bump clears those, and a missing value compares unequal to every real one, so // such a record would be recomputed anyway. type StatsRecord = { metaMs: number; idx?: string; stat: VideoStat }; // The part of buildIndex's `mtimes` record this cache reads: every input whose // change makes buildIndex re-process the video (and so rewrite its cues), and // the key it stored the cues under. type IndexRecord = { metaMs: number; transcriptMs: number | null; subsMs?: number | null; availabilityMs?: number | null; digestMs?: number | null; indexKey: IndexKey; }; // buildIndex has no `mtimes` record for this video. See the two counts below. const NOT_INDEXED = "-"; // The whole index record as one comparable string. Keying on all of it (not // the transcript mtime alone) means ANY re-index redoes the stat: a re-index // for another reason can read a fresher transcript.cues.json and change the cue // count, and a moved index key moves the cues. Numbers print as their shortest // round-trip form, so an unchanged record gives an identical string. function indexSignature(r: IndexRecord | undefined): string { if (!r) return NOT_INDEXED; const n = (v: number | null | undefined) => (v == null ? "" : String(v)); return [ n(r.metaMs), n(r.transcriptMs), n(r.subsMs), n(r.availabilityMs), n(r.digestMs), ...r.indexKey, ].join("\u0000"); } // Set to 1 (or true/yes/on) to let an OLDER build clear a stats cache a newer // one wrote — a deliberate rollback. Declared in lib/envVars.ts. export const STATS_DOWNGRADE_ENV = "ARCHILYZER_STATS_ALLOW_DOWNGRADE"; const TRUTHY = new Set(["1", "true", "yes", "on"]); function allowsStatsDowngrade( env: Record = process.env, ): boolean { const raw = env.ARCHILYZER_STATS_ALLOW_DOWNGRADE; return typeof raw === "string" && TRUTHY.has(raw.trim().toLowerCase()); } type ScanEntry = { channelSlug: string; videoDir: string; metaPath: string; metaMs: number; configName: string | undefined; }; export type BuildStatsResult = { totalCount: number; pageCount: number; channelCount: number; added: number; changed: number; removed: number; // Videos on disk with no index record, in two kinds. `notIndexedYet`: their // metadata is newer than the last completed index build's scan — downloaded // since — and the first stats run after the next index build redoes them. // `notIndexable`: older than the last index build, which did not index them // (no upload_date, a failure, or their channel's media unreachable during // that build); a stats run alone will not change that. notIndexedYet: number; notIndexable: number; // Channels whose media was not reachable, so their cached stats were kept. heldChannels: string[]; pagesWritten: number; shortCircuited: boolean; durationMs: number; }; export type BuildStatsOptions = { paths: Paths; onLog?: (msg: string) => void; signal?: AbortSignal; // When set, also write a whole-pool stats bundle (every non-excluded // channel, but for those only unlisted sites expose) into this dir as // {manifest,page-NNNN}.json. Used by the Archilyzer hub (compose-homepage), // whose cross-site charts need the full dataset rather than any one site's // filtered slice. Forces collection of the full dataset even when no per-site // bundle needs a rebuild. wholePoolStatsDir?: string; }; function pathKeyId(k: PathKey): string { return `${k[0]}\x00${k[1]}`; } // Format a Date as YYYYMMDD in UTC, matching uploadDate's shape so the chart // engine's time-binning treats all date fields identically. Returns null for an // invalid date. function ymdUtc(d: Date): string | null { const t = d.getTime(); if (!Number.isFinite(t)) return null; const y = d.getUTCFullYear(); const m = String(d.getUTCMonth() + 1).padStart(2, "0"); const day = String(d.getUTCDate()).padStart(2, "0"); return `${y}${m}${day}`; } const ymdFromIso = (iso: string): string | null => ymdUtc(new Date(iso)); const ymdFromMs = (ms: number): string | null => ymdUtc(new Date(ms)); async function fileMtimeMs(p: string): Promise { try { return (await stat(p)).mtimeMs; } catch { return null; } } // Resolve when this video's audio + transcript were acquired by us, for the // "content added over time" progress charts. Prefers the explicit outcome // sidecars (reliable across the shard rsync model, where file mtimes drift); // falls back to file mtimes for content added before the sidecars existed. // // A TRANSCRIPT ALWAYS HAS A DATE (schema 6). The fallbacks, in order: // 1. transcribe-outcome.json's `transcribedAt` — every Whisper run writes it; // 2. the mtime of the transcript the index takes its cues from // (pickIndexTranscript: transcript.json, else the caption VTT); // 3. the mtime of transcript.cues.json; // 4. downloadedDate, which always resolves. // A Whisper video resolves exactly as before: 1, else transcript.json's mtime, // which is what (2) picks whenever transcript.json exists (the one difference: // a sidecar whose `transcribedAt` will not parse used to leave no date, and now // falls through to 2). A CAPTION-handled // video is dated by when its captions ARRIVED — the VTT's mtime — not by a later // Normalize run: (3) used to be the only file that could date one, so a // caption video either had no date (never normalized) or took the Normalize // run's date (1,683 of one channel's, all on one day). The homepage fold, the // charts and the recent rail all need `hasTranscript` ⇒ `transcribedDate`. async function resolveAcquisitionDates( videoDir: string, metaMs: number, hasTranscript: boolean, ): Promise<{ downloadedDate: string | null; transcribedDate: string | null }> { const dl = await loadDownloadOutcome(videoDir); const downloadedDate = (dl?.finishedAt && ymdFromIso(dl.finishedAt)) || ymdFromMs(metaMs); let transcribedDate: string | null = null; if (hasTranscript) { const tr = await loadTranscribeOutcome(videoDir); transcribedDate = tr?.transcribedAt ? ymdFromIso(tr.transcribedAt) : null; if (transcribedDate === null) { const picked = pickIndexTranscript(await readVideoFiles(videoDir)); const mtime = (picked ? await fileMtimeMs(path.join(videoDir, picked.filename)) : null) ?? (await fileMtimeMs(path.join(videoDir, CUES_JSON_FILENAME))); transcribedDate = (mtime != null ? ymdFromMs(mtime) : null) ?? downloadedDate; } } return { downloadedDate, transcribedDate }; } // `held` maps each channel whose media is not reachable to why, in words with // no path in them: it is not scanned, and the caller keeps its cached stats // (see the file header). async function scanSource( channelsDir: string, log: (msg: string) => void, ): Promise<{ entries: ScanEntry[]; channels: Map; held: Map; }> { const locations = getSettings().storage.locations; const channels = new Map(); const entries: ScanEntry[] = []; const held = new Map(); let channelEntries: Dirent[]; try { channelEntries = await readdir(channelsDir, { withFileTypes: true }); } catch { return { entries, channels, held }; } for (const ch of channelEntries) { if (!ch.isDirectory()) continue; const channelDir = path.join(channelsDir, ch.name); const cfg = await readChannelConfigFile(path.join(channelDir, "config.json")); if (!cfg) { log(`Skipping channel ${ch.name}: missing or invalid config.json`); continue; } if (cfg.excludeFromBuild) continue; channels.set(ch.name, cfg); // An unreadable text tier is not an empty channel (lib/channelMedia.ts): // the readdir below would fail and every one of its stats would be removed. // THE TEXT GUARD (release 17): the stats build reads the text tier only and // is never held by the media tier — see buildIndex.ts's scanSource. const media = await inspectChannelMedia({ channelsDir }, ch.name, cfg, { fresh: true, }); if (!media.text.readable) { held.set(ch.name, heldReason(media, cfg.mediaDir ?? cfg.dataDir, locations)); continue; } const dataDir = path.join(channelDir, "data"); let videoEntries: Dirent[]; try { videoEntries = await readdir(dataDir, { withFileTypes: true }); } catch { continue; } for (const v of videoEntries) { if (!v.isDirectory()) continue; const metaPath = path.join(dataDir, v.name, "metadata.info.json"); let metaMs: number; try { metaMs = (await stat(metaPath)).mtimeMs; } catch { continue; } entries.push({ channelSlug: ch.name, videoDir: v.name, metaPath, metaMs, configName: cfg.name, }); } } return { entries, channels, held }; } // Buffered byte-capped page writer. Stats records are small, so a page fits in // memory comfortably (unlike transcript cues), keeping this far simpler than // the streaming writer in buildIndex.ts while honouring the same byte cap. async function writePages( outDir: string, stats: VideoStat[], maxPageBytes: number, ): Promise<{ pageCount: number; pagesWritten: number }> { await mkdir(outDir, { recursive: true }); const pages: string[][] = []; let cur: string[] = []; let curBytes = 2; // for the enclosing "[]" for (const s of stats) { const encoded = JSON.stringify(s); const delta = Buffer.byteLength(encoded, "utf8") + (cur.length ? 1 : 0); if (cur.length && curBytes + delta > maxPageBytes) { pages.push(cur); cur = []; curBytes = 2; } cur.push(encoded); curBytes += delta; } if (cur.length || pages.length === 0) pages.push(cur); for (let i = 0; i < pages.length; i++) { const outPath = path.join(outDir, statsPageFileName(i)); const tmp = `${outPath}.tmp-${process.pid}`; await writeFile(tmp, `[${pages[i].join(",")}]`); await rename(tmp, outPath); } // Remove any stale higher-numbered pages from a previous, larger run. for (let i = pages.length; ; i++) { const stale = path.join(outDir, statsPageFileName(i)); try { await stat(stale); await rm(stale, { force: true }); } catch { break; } } return { pageCount: pages.length, pagesWritten: pages.length }; } // Compact, no trailing newline — the stats manifests' historical bytes. function writeJsonAtomic(filePath: string, value: unknown): Promise { return writeJsonAtomicShared(filePath, value, { indent: 0, newline: false }); } export async function buildStats({ paths, onLog, signal, wholePoolStatsDir, }: BuildStatsOptions): Promise { const log = onLog ?? ((msg: string) => console.log(msg)); const t0 = Date.now(); await mkdir(path.dirname(paths.lmdbPath), { recursive: true }); await mkdir(paths.exportSitesIndexDir, { recursive: true }); const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true }); const statsByPath = root.openDB({ name: "statsByPath", encoding: "msgpack", }); // Read-only view of the cues populated by buildIndex, for cue counts. const cues = root.openDB({ name: "cues", encoding: "msgpack" }); // Read-only view of buildIndex's per-video VideoState (sparse — only // non-`available` videos). The status chart's authority; see the collection // loop below. const videoState = root.openDB({ name: "videoState", encoding: "msgpack", }); const meta = root.openDB({ name: "statsMeta", encoding: "msgpack" }); // Read-only views of buildIndex's per-video `mtimes` record (keyed like // statsByPath: the second half of this cache's key, and the key the video's // cues were stored under) and of its `meta` (when its last scan began). One // LMDB get per video, and no file I/O, on the unchanged path. const indexMtimes = root.openDB({ name: "mtimes", encoding: "msgpack", }); const indexMeta = root.openDB({ name: "meta", encoding: "msgpack" }); // NEVER CLEAR A CACHE A NEWER BUILD WROTE. An old build running beside a // new one (an editor not yet restarted onto the new code) would otherwise // clear it, refill it the old way, and the next new run would clear it back: // a full pass each time, and old-logic numbers published in between. const storedSchema = meta.get("schema") as number | undefined; if ( typeof storedSchema === "number" && storedSchema > STATS_SCHEMA_VERSION && !allowsStatsDowngrade() ) { await root.close(); throw new Error( `The stats cache was written by a newer build (stats schema ${storedSchema}; this build's is ${STATS_SCHEMA_VERSION}). ` + `Refusing to clear it: rebuild and restart onto the current code. ` + `For a deliberate rollback, set ${STATS_DOWNGRADE_ENV}=1.`, ); } const { entries, channels, held } = await scanSource(paths.channelsDir, log); log(`Scanned ${entries.length} videos across ${channels.size} channels.`); const schemaBumped = storedSchema !== STATS_SCHEMA_VERSION; if (schemaBumped) { // A clear with a channel's media unreachable would drop that channel's // stats for good (it cannot be rescanned), and the pages built from this // run would publish it as empty. Refuse instead. if (held.size > 0) { await root.close(); throw new Error( `The stats cache must be rebuilt (stats schema ${storedSchema ?? ""} -> ${STATS_SCHEMA_VERSION}), ` + `but ${held.size} channel(s) cannot be read: ${describeHeld(held)}. ${HELD_WAYS_OUT}`, ); } log( `Stats schema change (${storedSchema ?? ""} -> ${STATS_SCHEMA_VERSION}); clearing stats cache.`, ); await statsByPath.clearAsync(); await meta.put("schema", STATS_SCHEMA_VERSION); } const liveIds = new Set( entries.map((e) => pathKeyId([e.channelSlug, e.videoDir])), ); const scannedAt = indexMeta.get(INDEX_SCANNED_AT_KEY) as number | undefined; const toProcess: { e: ScanEntry; idx: string; indexKey?: IndexKey }[] = []; let added = 0; let changed = 0; let notIndexedYet = 0; let notIndexable = 0; for (const e of entries) { const pk: PathKey = [e.channelSlug, e.videoDir]; const rec = indexMtimes.get(pk); const idx = indexSignature(rec); if (idx === NOT_INDEXED) { if (typeof scannedAt !== "number" || e.metaMs > scannedAt) notIndexedYet++; else notIndexable++; } const prev = statsByPath.get(pk); if (!prev) { added++; toProcess.push({ e, idx, indexKey: rec?.indexKey }); } else if (prev.metaMs !== e.metaMs || prev.idx !== idx) { // The second half is the fix for stats frozen at first sight: a // transcript that arrives later makes buildIndex re-process the video // (and a video first seen before the index had it moves off // NOT_INDEXED), while the metadata file — the old key's only input — is // never touched. changed++; toProcess.push({ e, idx, indexKey: rec?.indexKey }); } } // Neither is a failure of this build, and both are said, so a published // number that lags the disk has its reason in the log. Only the first kind // resolves itself. if (notIndexedYet > 0) { log( `${notIndexedYet} video(s) were downloaded after the last index build; their transcripts reach the stats on the first run after the next one.`, ); } if (notIndexable > 0) { log( `${notIndexable} video(s) are not in the index although they are older than its last build: it skipped them (no upload_date, or it failed on them: see that build's log), or its channel's media was unreachable during that build (run an index build with every drive mounted). Their stats show no transcript until then.`, ); } const removedKeys: PathKey[] = []; const keptHeld = new Map(); for (const { key } of statsByPath.getRange()) { const k = key as PathKey; if (held.has(k[0])) { keptHeld.set(k[0], (keptHeld.get(k[0]) ?? 0) + 1); continue; } if (!liveIds.has(pathKeyId(k))) removedKeys.push(k); } for (const [slug, why] of held) { log( `Channel ${slug}: ${why}; its ${keptHeld.get(slug) ?? 0} cached stat(s) are kept as they are, not rescanned.`, ); } const removed = removedKeys.length; const BATCH = 200; for (let i = 0; i < toProcess.length; i += BATCH) { signal?.throwIfAborted(); const slice = toProcess.slice(i, i + BATCH); await Promise.all( slice.map(async ({ e, idx, indexKey: recordedKey }) => { try { const metaRaw = await readFile(e.metaPath, "utf8"); const parsedMeta = JSON.parse(metaRaw) as RawMetadata; const base = summarize( e.channelSlug, e.videoDir, parsedMeta, e.configName, ); if (!base.uploadDate) return; // The key buildIndex stored this video's cues under, when it has a // record: its uploadDate comes from transcript.cues.json when that // is fresh, which the metadata's can differ from. The computed key // is only a fallback for a video the index does not have. const indexKey: IndexKey = recordedKey ?? [base.uploadDate, e.channelSlug, base.id]; const cueList = cues.get(indexKey); const cueCount = cueList ? cueList.length : null; const coverage = transcriptCoverage(cueList, base.duration).coverage; // Placeholder. `status` is NOT a cached field any more: it is applied // from buildIndex's `videoState` sub-DB at collection time below. // This record's key does not see availability (metadata mtime and // the index's transcript mtime only), so caching a status here meant // a pure availability flip only reached the chart on the next // metadata touch or schema bump — a video deleted today did not show // as deleted today. const status: VideoStatus = "available"; const hasTranscript = cueCount != null && cueCount > 0; const { downloadedDate, transcribedDate } = await resolveAcquisitionDates( path.dirname(e.metaPath), e.metaMs, hasTranscript, ); const stat = summarizeStats( e.channelSlug, e.videoDir, parsedMeta, e.configName, { hasTranscript, cueCount, status, downloadedDate, transcribedDate, coverage, }, ); await statsByPath.put([e.channelSlug, e.videoDir], { metaMs: e.metaMs, idx, stat, }); } catch (err) { log(`Failed to process ${e.channelSlug}/${e.videoDir}: ${String(err)}`); } }), ); if (toProcess.length > BATCH) log(` processed ${Math.min(i + BATCH, toProcess.length)}/${toProcess.length}`); } for (const k of removedKeys) statsByPath.remove(k); await statsByPath.flushed; // --------------------------------------------------------------------------- // Per-site stats: the dataset is extracted once (incrementally cached above), // then written as a FILTERED bundle per site. Stats are channel-scoped, so a // site exposing a subset of channels must filter — sharing one dataset would // leak other sites' channels when a channel filter is cleared in the charts // UI. Output lands under //stats/, composed into // each site's /stats/ at build:site time. // --------------------------------------------------------------------------- let generation = (meta.get("generation") as number | undefined) ?? 0; if (added > 0 || changed > 0 || removed > 0 || schemaBumped) { generation++; await meta.put("generation", generation); } // Snapshot buildIndex's state map once. Sparse, so this stays small even on // the full corpus. Its hash goes into the per-site fingerprint because an // availability flip touches no metadata mtime: without it, added/changed/ // removed all stay 0, every site skips its rebuild, and the freshly-correct // status would never reach a chart. const stateByPath = new Map(); for (const { key, value } of videoState.getRange()) { stateByPath.set(key as string, value as VideoState); } const stateSignature = createHash("sha1") .update(JSON.stringify([...stateByPath.entries()].sort())) .digest("hex"); const sites = listSites(paths); const siteIds = new Set(sites.map((s) => s.siteId)); const siteFingerprint = (slugs: string[]): string => JSON.stringify({ gen: generation, max: STATS_MAX_PAGE_BYTES, channels: slugs, state: stateSignature, }); // Decide which sites need a rebuild before paying to collect the dataset. const plans = await Promise.all( sites.map(async (site) => { const slugs = site.channels.map((c) => c.slug); const fp = siteFingerprint(slugs); const outDir = siteStatsDir(paths, site.siteId); const siteManifestPath = path.join(outDir, "manifest.json"); const stored = meta.get(`statsFp:${site.siteId}`) as string | undefined; let canSkip = !schemaBumped && stored === fp; if (canSkip) { try { await stat(siteManifestPath); } catch { canSkip = false; } } return { site, slugs, fp, outDir, siteManifestPath, canSkip }; }), ); const needBuild = plans.filter((p) => !p.canSkip); // Collect + sort the full dataset once, when at least one site rebuilds OR the // whole-pool bundle was requested (the hub always wants the full dataset). const collectAll = needBuild.length > 0 || !!wholePoolStatsDir; const all: VideoStat[] = []; const perChannel = new Map(); if (collectAll) { for (const { key, value } of statsByPath.getRange()) { const rec = value as StatsRecord; // Apply visibility here rather than at extraction: this is the one place // that sees the current state map, so a video deleted (or dropped out of // its channel listing) since the last metadata touch shows correctly. const state = stateByPath.get(pathKeyId(key as PathKey)) ?? "available"; all.push( state === rec.stat.status ? rec.stat : { ...rec.stat, status: state }, ); perChannel.set( rec.stat.channelSlug, (perChannel.get(rec.stat.channelSlug) ?? 0) + 1, ); } // Sort newest-first to match the summaries ordering (uploadDate desc, then // channelSlug, then id). all.sort( (a, b) => b.uploadDate.localeCompare(a.uploadDate) || a.channelSlug.localeCompare(b.channelSlug) || a.id.localeCompare(b.id), ); } let sitesBuilt = 0; let sitesSkipped = 0; let aggregatePages = 0; let aggregateTotal = 0; let representativeChannelCount = 0; for (const plan of plans) { if (plan.canSkip) { sitesSkipped++; log(`Stats site ${plan.site.siteId}: up to date; skipping.`); continue; } signal?.throwIfAborted(); const slugSet = new Set(plan.slugs); const filtered = all.filter((s) => slugSet.has(s.channelSlug)); const { pageCount } = await writePages( plan.outDir, filtered, STATS_MAX_PAGE_BYTES, ); const channelEntries: StatsChannelEntry[] = plan.slugs .map((slug) => ({ slug, name: channels.get(slug)?.name ?? slug, count: perChannel.get(slug) ?? 0, })) .sort((a, b) => a.slug.localeCompare(b.slug)); const manifest: StatsManifest = { version: STATS_MANIFEST_VERSION, generatedAt: new Date().toISOString(), totalCount: filtered.length, pageCount, maxPageBytes: STATS_MAX_PAGE_BYTES, channels: channelEntries, }; await writeJsonAtomic(plan.siteManifestPath, manifest); await meta.put(`statsFp:${plan.site.siteId}`, plan.fp); sitesBuilt++; aggregatePages += pageCount; aggregateTotal += filtered.length; representativeChannelCount = Math.max( representativeChannelCount, channelEntries.length, ); log(`Stats site ${plan.site.siteId}: ${filtered.length} videos, ${pageCount} page(s).`); } // Whole-pool bundle for the hub: every non-excluded channel, except a channel // only unlisted sites expose (site.json `listed: false`): the bundle is // published as the homepage's `stats/`, and an unlisted site's content is in // no public total. Its own per-site bundle above is built as before. Built // from the same in-memory dataset so it stays consistent with the per-site // bundles. Always rewritten when requested (stats records are small). if (wholePoolStatsDir) { const unlistedOnly = channelsOnlyOnUnlistedSites(sites); const pooled = unlistedOnly.size > 0 ? all.filter((s) => !unlistedOnly.has(s.channelSlug)) : all; const { pageCount } = await writePages( wholePoolStatsDir, pooled, STATS_MAX_PAGE_BYTES, ); const channelEntries: StatsChannelEntry[] = [...channels.keys()] .filter((slug) => !unlistedOnly.has(slug)) .map((slug) => ({ slug, name: channels.get(slug)?.name ?? slug, count: perChannel.get(slug) ?? 0, })) .sort((a, b) => a.slug.localeCompare(b.slug)); const manifest: StatsManifest = { version: STATS_MANIFEST_VERSION, generatedAt: new Date().toISOString(), totalCount: pooled.length, pageCount, maxPageBytes: STATS_MAX_PAGE_BYTES, channels: channelEntries, }; await writeJsonAtomic(path.join(wholePoolStatsDir, "manifest.json"), manifest); log( `Stats whole-pool: ${pooled.length} videos, ${pageCount} page(s)` + (unlistedOnly.size > 0 ? `; ${unlistedOnly.size} channel(s) only unlisted sites expose left out.` : "."), ); } // Prune fingerprints for sites that no longer exist (their staging dirs are // removed by buildIndex, which owns //). const staleFp: string[] = []; for (const { key } of meta.getRange()) { const k = key as string; if ( typeof k === "string" && k.startsWith("statsFp:") && !siteIds.has(k.slice("statsFp:".length)) ) { staleFp.push(k); } } for (const k of staleFp) meta.remove(k); await meta.flushed; await root.close(); log( `Stats built across ${sites.length} site(s): ${sitesBuilt} built, ${sitesSkipped} up to date (added ${added}, changed ${changed}, removed ${removed}).`, ); return { totalCount: aggregateTotal, pageCount: aggregatePages, channelCount: representativeChannelCount, added, changed, removed, notIndexedYet, notIndexable, heldChannels: [...held.keys()], pagesWritten: aggregatePages, shortCircuited: needBuild.length === 0, durationMs: Date.now() - t0, }; }