import type { Platform } from "./platform"; import { pageFileName } from "./manifest"; import type { VideoState } from "./availability"; // Bumping this invalidates the LMDB `statsByPath` incremental cache and forces // a full re-extraction (e.g. when a new field is added below). It versions the // CACHE, not the published pages: STATS_MANIFEST_VERSION is theirs. // 6 โ€” the cache key gained the index's transcript record, and a transcript // always has a `transcribedDate` (caption videos: the VTT's arrival). // The page shape did not change. export const STATS_SCHEMA_VERSION = 6; export const STATS_MANIFEST_VERSION = 1; // A key in buildIndex's `meta` sub-DB (not this cache's): the ms timestamp the // last COMPLETED index build's scan began at. buildIndex writes it; buildStats // reads it to tell a video downloaded since that scan from one the scan saw and // did not index. Here rather than in buildIndex.ts so buildStats need not load // the index builder. Additive: the index schema did not move. export const INDEX_SCANNED_AT_KEY = "scannedAt"; // Visibility of a video on its source platform. One type with the viewer's, so // the status chart and the search filter can never drift apart; see VideoState // in lib/availability for the taxonomy (available + the five missing leaves). export type VideoStatus = VideoState; // Coarse media kind. YouTube distinguishes shorts; other platforms only carry // video vs. livestream, so they fall back to one of those. export type MediaType = "video" | "livestream" | "short"; // Per-page byte cap for the served stats index. Static hosts limit individual // file sizes (Cloudflare Pages = 25 MB; smaller elsewhere), so pages are // flushed by byte size rather than a fixed record count. Kept well under 25 MB. export const STATS_MAX_PAGE_BYTES = 20 * 1024 * 1024; export const statsPageFileName = pageFileName; // One record per video. Engagement fields are nullable: not every platform / // metadata.info.json carries them, and older archived videos may omit them. export type VideoStat = { slug: string; // `${channelSlug}/${id}` id: string; channelSlug: string; channel: string; // display name title: string; platform: Platform; uploadDate: string; // YYYYMMDD โ€” when the creator published // Acquisition dates (YYYYMMDD, UTC) โ€” when *we* added the content. Null when no // source is resolvable (pre-sidecar video with no usable mtime). Used by the // "content added over time" progress charts; the time X-axis can bin on any of // these date fields. downloadedDate: string | null; // Non-null whenever `hasTranscript` is, since schema 6 (a caption video takes // its captions' arrival). A page written by an older build can still carry a // transcript with a null date: readers count it and only leave it off a time // axis (homepageSummary does exactly that). transcribedDate: string | null; timestamp: number | null; // unix seconds duration: number; // seconds viewCount: number | null; likeCount: number | null; commentCount: number | null; channelFollowerCount: number | null; categories: string[]; tags: string[]; language: string | null; isLivestream: boolean; mediaType: MediaType; status: VideoStatus; hasTranscript: boolean; cueCount: number | null; // Fraction of the video's duration covered by the transcript (last cue end รท // duration). null when there's no transcript or no usable duration. Values // well below 1 flag a truncated download (see lib/transcriptCoverage). coverage: number | null; }; export type StatsChannelEntry = { slug: string; name: string; count: number; }; export type StatsManifest = { version: number; generatedAt: string; totalCount: number; pageCount: number; // Records the byte cap pages were written under, so a changed cap invalidates // stale pages (mirrors the transcript index's ChannelTranscriptsManifest). maxPageBytes: number; channels: StatsChannelEntry[]; };