import type { Platform } from "./platform"; import type { VideoStat } from "./stats"; import type { Site } from "./site"; import { channelsOnlyOnUnlistedSites, isListedSite } from "./siteSchema"; import { accentHex, accentIdOf } from "./accent"; import { wordmarkLeadFor, type AccentId } from "./brand"; import { VIDEO_STATES, type VideoState } from "./availability"; // Pre-computed, lightweight cross-site summary for the hub (homepage) landing. // Built once at compose time (compose-homepage.ts) and embedded into the SSG // HTML, so the landing renders instantly without the browser fetching the // multi-MB whole-pool stats dataset. // // Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl that // are listed — site.json `listed`, lib/siteSchema.ts isListedSite; never a // private or a cited report site), and every // video is attributed to a single PRIMARY public site (the first, by sorted id, // exposing its channel) so the Site and Channel breakdowns partition the same // set and combined totals stay honest. The KPI `totals` (and `availability`) // are instance-wide (count pool-only / URL-less content too) — a deliberate // scope difference — EXCEPT a channel only unlisted sites expose, which no // part of the summary counts (channelsOnlyOnUnlistedSites). An unlisted site is // in no array here; a channel it shares with a listed site is the listed one's. // // Two metrics are pre-binned at two granularities; Cumulative and Share (100%) // are derived client-side from these, so no extra precompute is needed. // v4 adds `availability` (instance-wide counts by source-platform state) and // `status` on each recent item. Both are ADDITIVE and both are declared optional // on the type, because nothing gates on this number — it is a provenance marker, // not a compatibility check — and a summary written by an older build must keep // deserializing. Readers must treat their absence as "unknown", never as zero. // // v5 adds `monthly`, `official` and the per-site `channels` / `recordings` / // `hoursArchived` / `gone` numbers — all additive, all optional on the type for // the same reason: a v4 summary on disk must still render (the numbers hide). // // Still v5: the per-site `accentId` (release 10) is additive and optional in // the same way — a summary without it paints its sites' hex, as before. So is // the per-site `wordmarkLead` (release 14): a summary without it shows each // card's title plain, as before. Nothing reads this number to accept a file. // // v6 (release 14): an unlisted site (site.json `listed: false`) is in no array, // and a channel only unlisted sites expose is in no total — `totals` and // `availability` included. No field was added or removed; the number says the // totals' scope moved. // // Still v6: a cited report site (site.json `search: false`) is left out // exactly as an unlisted one (lib/siteSchema.ts isListedSite) — the same scope // rule, applied to a kind of site no summary had yet counted. export const HOMEPAGE_SUMMARY_VERSION = 6; // Day buckets are capped to this many trailing days so the embedded summary stays // small regardless of archive age (daily detail is only useful recently). const DAY_WINDOW = 182; // channel slug -> ids of the content sites that expose it (same map // compose-homepage builds for the cross-site charts). export type ChannelSitesMap = Record; export type MetricKey = "transcribed" | "downloaded"; // Per-granularity counts. `buckets` are ascending and continuous (gaps are 0): // day buckets are UTC dates ("YYYY-MM-DD"), capped to the last DAY_WINDOW days; // week buckets are Monday-anchored UTC dates; month buckets are "YYYY-MM". // `total`/`bySite[id]`/`byChannel[slug]` are all aligned to `buckets`. export type BucketSeries = { buckets: string[]; total: number[]; bySite: Record; byChannel: Record; }; export type MetricSeries = { day: BucketSeries; week: BucketSeries; month: BucketSeries; }; export type SiteMetricStat = { // For `transcribed`, also counts transcripts with no transcribedDate (a stats // page from before stats schema 6) — they have no month, so they are in // `total` and in neither `thisMonth` nor `last12`. total: number; thisMonth: number; last12: number[]; // last 12 month buckets (zero-padded left), for the sparkline }; export type HomepageSummarySite = { siteId: string; siteTitle: string; siteDescription: string; siteUrl: string; // public sites only — always present transcribed: SiteMetricStat; downloaded: SiteMetricStat; // v5 per-site numbers, over the channels this site is the PRIMARY public site // for (the same attribution as `transcribed`/`downloaded`, so the cards sum to // `official`). Optional: absent from pre-v5 summaries. channels?: number; // channels with at least one transcript recordings?: number; // records with a download date (matches totals.downloads) hoursArchived?: number; // transcribed-video durations, rounded to the hour gone?: number; // records re-checked and found deleted at the source // The site's own brand accent ("#rrggbb"), when it defines one. Absent for a // site that hasn't set one — callers fall back to the chart palette rather // than inventing a brand colour. Added in v4. accent?: string; // The named accent behind `accent`, when the site picked one rather than a // custom hex (lib/accent.ts accentIdOf). `accent` is then that accent's // ON-DARK value only; the id lets the family's own pages paint it on every // base (`var(--swatch-)`, lib/siteColor.ts). Optional, additive // (release 10): absent for a custom hex, no accent, or an older summary. accentId?: AccentId; // The site's wordmark lead ("Jer" of "Jeralyzer"): site.json `wordmarkLead` // resolved against `siteTitle` by lib/brand.ts wordmarkLeadFor, the resolver // the sites' own header and site.json use — a proper prefix of the title, or // absent. The homepage's card sets the title as the site's wordmark with it. // Optional, additive (release 14): absent for a site with no lead, or an // older summary. wordmarkLead?: string; }; export type HomepageChannelMeta = { slug: string; name: string }; export type HomepageRecentItem = { slug: string; // `${channelSlug}/${id}` — the export deep-link `?v=` value title: string; channel: string; // display name platform: Platform; siteId: string; siteTitle: string; siteUrl: string; transcribedDate: string; // YYYYMMDD // Where this recording stands on the platform it came from, as of the last // time we looked. Optional: summaries written before v4 don't carry it, and a // renderer must show nothing rather than guess "available". status?: VideoState; }; // Instance-wide census of archived recordings by their state on the source // platform. This is the archive's whole point stated as a number, so the // honesty rules matter: // // • `available` is a FLOOR, not a fact. A recording counts as available until // someone re-checks it and finds otherwise; nobody re-checks 60,000 videos // continuously. It means "not known to be gone", never "confirmed safe". // • `deleted` is the opposite: every one of those was individually re-checked // and found gone. It is the only count here that is evidence. // Copy rendered from this must not launder the first into the second. export type HomepageAvailability = { // Counts keyed by VideoState. Every state is present, zero included, so a // renderer can iterate a stable set instead of probing for keys. byState: Record; // Total records censused — the denominator for any percentage. counted: number; }; // One calendar month of the archive's BACK CATALOGUE: transcribed recordings in // the public universe, keyed by the month the creator UPLOADED them (not the // month we transcribed them — that history is only as old as the tool, while // upload month shows how far back each archive reaches). Values are keyed by // siteId and attributed to the primary public site, so a month's values sum // to that month's transcripts with no double counting. export type HomepageMonth = { month: string; // "YYYY-MM" bySite: Record; }; // The family's own numbers: the sum over `sites` (public sites with activity), // NOT the instance-wide `totals`, which also count pool-only channels and // sites without a public URL. Copy that says "across the official instances" // reads this. export type HomepageOfficialTotals = { sites: number; channels: number; recordings: number; transcripts: number; hoursArchived: number; gone: number; }; export type HomepageSummary = { version: number; generatedAt: string; // ISO timestamp // Instance-wide headline numbers (count everything, not just public sites — // but never a channel only unlisted sites expose). totals: { transcripts: number; downloads: number; sites: number; // public site count channels: number; hoursArchived: number; // sum of transcribed-video durations, in hours transcribedThisMonth: number; downloadedThisMonth: number; }; // Public-universe channels (for the Channel-breakdown legend labels). channels: HomepageChannelMeta[]; series: Record; // Public sites, sorted by transcribed total desc. sites: HomepageSummarySite[]; // Newest transcriptions (public universe), newest first. Rendered as "the // rail" on the project site's home page. recent: HomepageRecentItem[]; // Instance-wide state census. Optional — absent from pre-v4 summaries. availability?: HomepageAvailability; // v5. Full history of COMPLETE months: one entry per month from the earliest // upload month in the public universe to the month BEFORE the build month, // zero-filled between. The build month itself is never emitted — it is // partial, and charted it reads as a collapse. Every site in `sites` has a // key in every entry (zero included). Empty when there is nothing to place. // Transcripts whose upload date is missing, malformed, in the build month or // later are not placed; `monthlyUnplaced` counts them so nothing is dropped // silently. monthly?: HomepageMonth[]; monthlyUnplaced?: number; official?: HomepageOfficialTotals; }; const RECENT_LIMIT = 24; // ─── date bucketing ─── function monthOf(date: string | null): string | null { if (!date || date.length < 6) return null; return `${date.slice(0, 4)}-${date.slice(4, 6)}`; } // An upload date "YYYYMMDD" -> "YYYY-MM", or null when it is not a plausible // date (missing, short, non-numeric, month outside 1..12). function uploadMonthOf(date: string | null | undefined): string | null { if (!date || !/^\d{8}$/.test(date)) return null; const m = +date.slice(4, 6); if (m < 1 || m > 12 || +date.slice(0, 4) < 1900) return null; return `${date.slice(0, 4)}-${date.slice(4, 6)}`; } // "YYYY-MM" -> the calendar month before it. function previousMonth(ym: string): string { const [y, m] = ym.split("-").map(Number); return m === 1 ? `${String(y - 1).padStart(4, "0")}-12` : `${String(y).padStart(4, "0")}-${String(m - 1).padStart(2, "0")}`; } function monthRange(start: string, end: string): string[] { const out: string[] = []; let [y, m] = start.split("-").map(Number); const [ey, em] = end.split("-").map(Number); while (y < ey || (y === ey && m <= em)) { out.push(`${String(y).padStart(4, "0")}-${String(m).padStart(2, "0")}`); m += 1; if (m > 12) { m = 1; y += 1; } } return out; } function isoDate(dt: Date): string { return `${dt.getUTCFullYear()}-${String(dt.getUTCMonth() + 1).padStart(2, "0")}-${String(dt.getUTCDate()).padStart(2, "0")}`; } // "YYYYMMDD" -> the Monday-anchored UTC week-start "YYYY-MM-DD" (or null). function weekOf(date: string | null): string | null { if (!date || date.length < 8) return null; const y = +date.slice(0, 4); const m = +date.slice(4, 6); const d = +date.slice(6, 8); if (!y || !m || !d) return null; const dt = new Date(Date.UTC(y, m - 1, d)); const diff = (dt.getUTCDay() + 6) % 7; // days since Monday (Sun=0 → 6) dt.setUTCDate(dt.getUTCDate() - diff); return isoDate(dt); } function weekRange(start: string, end: string): string[] { const out: string[] = []; const dt = new Date(`${start}T00:00:00Z`); const endDt = new Date(`${end}T00:00:00Z`); while (dt.getTime() <= endDt.getTime()) { out.push(isoDate(dt)); dt.setUTCDate(dt.getUTCDate() + 7); } return out; } // "YYYYMMDD" -> the calendar UTC day "YYYY-MM-DD" (or null). function dayOf(date: string | null): string | null { if (!date || date.length < 8) return null; return `${date.slice(0, 4)}-${date.slice(4, 6)}-${date.slice(6, 8)}`; } function dayRange(start: string, end: string): string[] { const out: string[] = []; const dt = new Date(`${start}T00:00:00Z`); const endDt = new Date(`${end}T00:00:00Z`); while (dt.getTime() <= endDt.getTime()) { out.push(isoDate(dt)); dt.setUTCDate(dt.getUTCDate() + 1); } return out; } // ─── helpers ─── type Attributed = { date: string; channelSlug: string; siteId: string }; // Bin a metric's attributed videos at one granularity into a BucketSeries. function bucketize( items: Attributed[], keyOf: (date: string) => string | null, rangeOf: (min: string, max: string) => string[], nowKey: string, floor?: string, ): BucketSeries { let min: string | null = null; for (const it of items) { const k = keyOf(it.date); if (k && (min === null || k < min)) min = k; } // Clamp the start to `floor` (used to cap the day series to a recent window). if (floor && (min === null || min < floor)) min = floor; const buckets = min ? rangeOf(min < nowKey ? min : nowKey, nowKey) : []; const index = new Map(buckets.map((b, i) => [b, i])); const total = new Array(buckets.length).fill(0); const bySite: Record = {}; const byChannel: Record = {}; for (const it of items) { const k = keyOf(it.date); const i = k != null ? index.get(k) : undefined; if (i == null) continue; total[i] += 1; (bySite[it.siteId] ??= new Array(buckets.length).fill(0))[i] += 1; (byChannel[it.channelSlug] ??= new Array(buckets.length).fill(0))[i] += 1; } return { buckets, total, bySite, byChannel }; } function siteStatFrom( month: BucketSeries, siteId: string, nowMonth: string, ): SiteMetricStat { const arr = month.bySite[siteId] ?? []; const total = arr.reduce((a, b) => a + b, 0); const idx = month.buckets.indexOf(nowMonth); const thisMonth = idx >= 0 ? (arr[idx] ?? 0) : 0; const last12 = arr.slice(Math.max(0, arr.length - 12)); while (last12.length < 12) last12.unshift(0); return { total, thisMonth, last12 }; } // A site's transcribed card counts its undated transcripts too. They have no // month, so they join `total` only — never `thisMonth` or the sparkline. function withUndated(stat: SiteMetricStat, undated: number): SiteMetricStat { return undated > 0 ? { ...stat, total: stat.total + undated } : stat; } export function buildHomepageSummary( stats: readonly VideoStat[], channelSites: ChannelSitesMap, sites: readonly Site[], now: Date, ): HomepageSummary { const nowMonth = `${now.getUTCFullYear()}-${String(now.getUTCMonth() + 1).padStart(2, "0")}`; const nowDay = isoDate(now); const nowWeek = weekOf(nowDay.replace(/-/g, ""))!; const dayFloor = isoDate(new Date(now.getTime() - (DAY_WINDOW - 1) * 86400000)); // Public sites only (need a link target + a stable place on the chart), and // only the listed ones. const publicSites = sites.filter( (s): s is Site & { siteUrl: string } => !!s.siteUrl && isListedSite(s), ); const siteById = new Map(publicSites.map((s) => [s.siteId, s])); // An unlisted site's own channels: counted nowhere below, totals included. const unlistedOnly = channelsOnlyOnUnlistedSites(sites); const inScope = unlistedOnly.size > 0 ? stats.filter((s) => !unlistedOnly.has(s.channelSlug)) : stats; // channel slug -> primary public site (first by sorted id). Channels with no // public site are out of the chart universe entirely. Over listed sites only, // so a channel an unlisted site shares with a listed one is the listed one's. const primarySiteOf = new Map(); for (const slug of Object.keys(channelSites)) { const primary = [...channelSites[slug]] .filter((id) => siteById.has(id)) .sort()[0]; if (primary) primarySiteOf.set(slug, primary); } // Collect attributed (public-universe) videos per metric, plus channel names. const channelName = new Map(); const transcribedItems: Attributed[] = []; const downloadedItems: Attributed[] = []; // Instance-wide KPI accumulators (count everything in scope, not just public). let transcripts = 0; let downloads = 0; let hoursSeconds = 0; let transcribedThisMonth = 0; let downloadedThisMonth = 0; const channelSet = new Set(); // Per public site: channels with a transcript, recordings, transcripts, // seconds, gone. `transcripts` is counted HERE, in the same pass that places // or leaves unplaced each transcript, so `placed + unplaced = // official.transcripts` holds even for a `transcribedDate` that bucketize // drops (a future or malformed month). `undated` counts the transcripts with // no `transcribedDate` at all (a stats page from before stats schema 6): no // transcribed bucket can hold them, so they join the card's total directly. type SiteAcc = { channels: Set; recordings: number; transcripts: number; undated: number; seconds: number; gone: number; }; const siteAcc = new Map(); const accOf = (id: string): SiteAcc => { let a = siteAcc.get(id); if (!a) siteAcc.set( id, (a = { channels: new Set(), recordings: 0, transcripts: 0, undated: 0, seconds: 0, gone: 0 }), ); return a; }; // Upload-month placement of transcribed public-universe records. const uploadItems: { month: string; siteId: string }[] = []; let monthlyUnplaced = 0; for (const s of inScope) { // A transcript COUNTS whether or not it carries a date: transcripts, // channels, hours, upload-month placement. Only the transcribed time series // (and "this month", and the recent rail) need `transcribedDate`. buildStats // guarantees one since stats schema 6, but a page written before that could // hold a whole channel of transcripts with none — and requiring the date // here made a site serving 1,889 videos show 0 transcripts, 0 channels and // 0 hours. const hasTx = s.hasTranscript; const dated = hasTx && !!s.transcribedDate; if (hasTx) { transcripts += 1; channelSet.add(s.channelSlug); hoursSeconds += s.duration > 0 ? s.duration : 0; if (dated && monthOf(s.transcribedDate) === nowMonth) transcribedThisMonth += 1; } if (s.downloadedDate) { downloads += 1; if (monthOf(s.downloadedDate) === nowMonth) downloadedThisMonth += 1; } const siteId = primarySiteOf.get(s.channelSlug); if (!siteId) continue; // not on any public site channelName.set(s.channelSlug, s.channel); const acc = accOf(siteId); if (s.downloadedDate) acc.recordings += 1; // "Gone at the source, still here": only a record we actually hold. if (s.status === "deleted" && s.downloadedDate) acc.gone += 1; if (hasTx) { acc.transcripts += 1; acc.channels.add(s.channelSlug); acc.seconds += s.duration > 0 ? s.duration : 0; const um = uploadMonthOf(s.uploadDate); if (um && um < nowMonth) uploadItems.push({ month: um, siteId }); else monthlyUnplaced += 1; if (dated) { transcribedItems.push({ date: s.transcribedDate as string, channelSlug: s.channelSlug, siteId }); } else { acc.undated += 1; } } if (s.downloadedDate) { downloadedItems.push({ date: s.downloadedDate, channelSlug: s.channelSlug, siteId }); } } const metricSeries = (items: Attributed[]): MetricSeries => ({ day: bucketize(items, dayOf, dayRange, nowDay, dayFloor), week: bucketize(items, weekOf, weekRange, nowWeek), month: bucketize(items, monthOf, monthRange, nowMonth), }); const series: Record = { transcribed: metricSeries(transcribedItems), downloaded: metricSeries(downloadedItems), }; const summarySites: HomepageSummarySite[] = publicSites .map((s) => ({ siteId: s.siteId, siteTitle: s.siteTitle, siteDescription: s.siteDescription, siteUrl: s.siteUrl, transcribed: withUndated( siteStatFrom(series.transcribed.month, s.siteId, nowMonth), siteAcc.get(s.siteId)?.undated ?? 0, ), downloaded: siteStatFrom(series.downloaded.month, s.siteId, nowMonth), channels: siteAcc.get(s.siteId)?.channels.size ?? 0, recordings: siteAcc.get(s.siteId)?.recordings ?? 0, hoursArchived: Math.round((siteAcc.get(s.siteId)?.seconds ?? 0) / 3600), gone: siteAcc.get(s.siteId)?.gone ?? 0, // A published colour: an accent id becomes its hex (lib/accent.ts), and // travels as its id too. ...(accentHex(s.accent) ? { accent: accentHex(s.accent) } : {}), ...(accentIdOf(s.accent) ? { accentId: accentIdOf(s.accent) } : {}), ...(wordmarkLeadFor(s.siteTitle, s.wordmarkLead) ? { wordmarkLead: wordmarkLeadFor(s.siteTitle, s.wordmarkLead) } : {}), })) // Keep a public site only if it has any activity in either metric. .filter((s) => s.transcribed.total > 0 || s.downloaded.total > 0) .sort( (a, b) => b.transcribed.total - a.transcribed.total || a.siteTitle.localeCompare(b.siteTitle), ); const channels: HomepageChannelMeta[] = [...channelName.entries()] .map(([slug, name]) => ({ slug, name })) .sort((a, b) => a.name.localeCompare(b.name)); // Recent feed (public universe), attributed to the primary site for its link. const recent: HomepageRecentItem[] = inScope .filter((s): s is VideoStat & { transcribedDate: string } => Boolean(s.hasTranscript && s.transcribedDate && primarySiteOf.get(s.channelSlug)), ) .sort( (a, b) => b.transcribedDate.localeCompare(a.transcribedDate) || a.slug.localeCompare(b.slug), ) .slice(0, RECENT_LIMIT) .map((s) => { const siteId = primarySiteOf.get(s.channelSlug)!; const site = siteById.get(siteId)!; return { slug: s.slug, title: s.title, channel: s.channel, platform: s.platform, siteId, siteTitle: site.siteTitle, siteUrl: site.siteUrl, transcribedDate: s.transcribedDate, status: s.status, }; }); // State census over every in-scope record, matching `totals`' instance-wide // scope (pool-only channels included, a channel only unlisted sites expose // not) rather than the charts' public-site universe. // Seeded with every state at zero so the shape is stable across corpora. const byState = Object.fromEntries( VIDEO_STATES.map((s) => [s, 0]), ) as Record; for (const s of inScope) byState[s.status] = (byState[s.status] ?? 0) + 1; const availability: HomepageAvailability = { byState, counted: inScope.length, }; // Monthly back-catalogue series, keyed by the sites that survived the // activity filter (a site with no activity has no transcripts to place). const keptIds = summarySites.map((s) => s.siteId); let firstMonth: string | null = null; for (const it of uploadItems) { if (firstMonth === null || it.month < firstMonth) firstMonth = it.month; } const lastComplete = previousMonth(nowMonth); const monthly: HomepageMonth[] = firstMonth ? monthRange(firstMonth, lastComplete).map((month) => ({ month, bySite: Object.fromEntries(keptIds.map((id) => [id, 0])), })) : []; const monthIndex = new Map(monthly.map((m, i) => [m.month, i])); for (const it of uploadItems) { const i = monthIndex.get(it.month); if (i === undefined) continue; // unreachable: every item's month is in range // A site the activity filter dropped has no key here and gets none: its // transcripts are not placed, and not counted in `official` either. if (!(it.siteId in monthly[i].bySite)) continue; monthly[i].bySite[it.siteId] += 1; } // The family's own totals — hours summed in seconds, then rounded once. let officialSeconds = 0; for (const id of keptIds) officialSeconds += siteAcc.get(id)?.seconds ?? 0; const official: HomepageOfficialTotals = { sites: summarySites.length, channels: summarySites.reduce((a, s) => a + (s.channels ?? 0), 0), recordings: summarySites.reduce((a, s) => a + (s.recordings ?? 0), 0), transcripts: keptIds.reduce( (a, id) => a + (siteAcc.get(id)?.transcripts ?? 0), 0, ), hoursArchived: Math.round(officialSeconds / 3600), gone: summarySites.reduce((a, s) => a + (s.gone ?? 0), 0), }; return { version: HOMEPAGE_SUMMARY_VERSION, generatedAt: now.toISOString(), totals: { transcripts, downloads, sites: summarySites.length, channels: channelSet.size, hoursArchived: Math.round(hoursSeconds / 3600), transcribedThisMonth, downloadedThisMonth, }, channels, series, sites: summarySites, recent, availability, monthly, monthlyUnplaced, official, }; }