commit 286c422999e8d65b57187aa19d6f6b02d6c2c7f2
parent 03cb237c64a43ff2ad687e1074d3cc60b909f106
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 29 Sep 2026 20:46:18 -0400
common: the homepage summary, channel-sites.json and the pooled stats leave out an unlisted site and the channels only it exposes; summary v6
buildHomepageSummary's public universe is listed sites with a siteUrl, so an
unlisted site is in no array (sites, official, monthly, series, recent) and a
channel it shares with a listed site is the listed one's; a channel only
unlisted sites expose is in no total — `totals` and `availability` included.
Pool-only channels still count in `totals`, as before. channelSitesOf maps
listed sites only, and buildStats' whole-pool bundle (the homepage's
`stats/`) omits those channels; the unlisted site's own bundle is unchanged.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
7 files changed, 222 insertions(+), 35 deletions(-)
diff --git a/common/bin/compose-homepage.ts b/common/bin/compose-homepage.ts
@@ -8,6 +8,9 @@
// public/channel-sites.json <- channel slug -> [siteId, ...]
// public/homepage-summary.json <- small cross-site landing summary
//
+// None of the three names an unlisted site (site.json `listed: false`) or holds
+// a channel only unlisted sites expose (lib/siteSchema.ts isListedSite).
+//
// Requires build:index to have populated the cues LMDB first (the homepage
// prebuild chains it), same as the export pipeline.
@@ -15,6 +18,7 @@ import path from "node:path";
import { getPaths, type Paths } from "../lib/paths";
import { writeJsonAtomic as writeJsonAtomicShared } from "../lib/jsonFile-server";
import { buildPoolSummary } from "../controller/poolSummary";
+import { isListedSite } from "../lib/site";
import { runIfEntryPoint } from "./_cli";
// Where the homepage Next.js app serves static assets from. Overridable for e2e
@@ -45,7 +49,7 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
statsDir,
});
- // channel slug -> the ids of the content sites that expose it. Drives
+ // channel slug -> the ids of the listed content sites that expose it. Drives
// `groupBy: "site"` in the hub's dashboard (see channelSites.tsx).
await writeJsonAtomic(
path.join(publicDir, "channel-sites.json"),
@@ -60,8 +64,11 @@ export async function main(opts: { paths?: Paths } = {}): Promise<void> {
summary,
);
+ const unlisted = sites.filter((s) => !isListedSite(s)).length;
console.log(
- `compose-homepage: ${Object.keys(channelSites).length} channel(s) mapped across ${sites.length} site(s); ` +
+ `compose-homepage: ${Object.keys(channelSites).length} channel(s) mapped across ${sites.length - unlisted} listed site(s)` +
+ (unlisted > 0 ? ` (${unlisted} unlisted left out)` : "") +
+ "; " +
`summary covers ${summary.totals.transcripts} transcription(s) / ${summary.totals.downloads} download(s) ` +
`across ${summary.sites.length} public site(s) into ${publicDir}.`,
);
diff --git a/common/controller/buildStats.test.ts b/common/controller/buildStats.test.ts
@@ -65,6 +65,7 @@ const { buildIndex } = await import("./buildIndex");
const { buildStats, STATS_DOWNGRADE_ENV } = await import("./buildStats");
const { normalizeTranscript } = await import("./normalizeTranscript");
const { readStatsPages } = await import("./poolSummary");
+const { siteStatsDir } = await import("../lib/site");
const { STATS_SCHEMA_VERSION } = await import("../lib/stats");
const { open } = await import("lmdb");
@@ -594,6 +595,62 @@ test("(j) the schema guard: an older cache is cleared, a newer one is refused un
}
});
+// Release 14 slice HS: the whole-pool bundle is published as the homepage's
+// `stats/`, and an unlisted site's content is in no public total.
+test("(k) the whole-pool bundle leaves out a channel only an unlisted site exposes; the site's own bundle keeps it", async () => {
+ resetCorpus();
+ const seedChannel = (slug: string, ids: string[]) => {
+ writeJson(path.join(paths.channelsDir, slug, "config.json"), {
+ handling: "youtube",
+ name: slug,
+ url: `https://www.youtube.com/@${slug}/videos`,
+ });
+ for (const id of ids) {
+ seedVideo(id, "2026-07-11T11:00:00Z", {}, slug);
+ addCaptions(id, "2026-07-11T12:00:00Z", slug);
+ }
+ };
+ seedVideo("listed-1");
+ seedChannel("unlisted-channel", ["u1", "u2"]);
+ seedChannel("shared-channel", ["s1"]);
+ seedChannel("pool-channel", ["p1"]);
+ // The listed site also exposes the shared channel; the unlisted site exposes
+ // its own channel and the shared one. The pool channel is on no site.
+ const siteFile = path.join(paths.sitesDir, SITE, "site.json");
+ const listed = JSON.parse(readFileSync(siteFile, "utf8"));
+ listed.channels.push({ slug: "shared-channel", groupId: "default" });
+ writeJson(siteFile, listed);
+ writeJson(path.join(paths.sitesDir, "fixture-unlisted", "site.json"), {
+ ...listed,
+ siteId: "fixture-unlisted",
+ siteTitle: "Unlisted",
+ headerTitle: "Unlisted",
+ siteUrl: "https://unlisted.example",
+ listed: false,
+ channels: [
+ { slug: "unlisted-channel", groupId: "default" },
+ { slug: "shared-channel", groupId: "default" },
+ ],
+ });
+ await runIndex();
+ const log: string[] = [];
+ const { byId } = await runStats(log);
+ assert.deepEqual([...byId.keys()].sort(), ["listed-1", "p1", "s1"]);
+ const manifest = JSON.parse(readFileSync(path.join(POOL, "manifest.json"), "utf8"));
+ assert.equal(manifest.totalCount, 3);
+ assert.deepEqual(
+ manifest.channels.map((c: { slug: string }) => c.slug),
+ ["pool-channel", "shared-channel", CHANNEL],
+ );
+ assert.ok(
+ log.includes("Stats whole-pool: 3 videos, 1 page(s); 1 channel(s) only unlisted sites expose left out."),
+ log.join("\n"),
+ );
+ // The unlisted site still builds as before: its own bundle has its videos.
+ const own = await readStatsPages(siteStatsDir(paths, "fixture-unlisted"));
+ assert.deepEqual(own.map((s) => s.id).sort(), ["s1", "u1", "u2"]);
+});
+
test("(z) no write this file caused landed outside its temp root", () => {
// LMDB writes natively, past the spy: its file must be under the root too.
assert.ok(paths.lmdbPath.startsWith(ROOT + path.sep), paths.lmdbPath);
diff --git a/common/controller/buildStats.ts b/common/controller/buildStats.ts
@@ -70,7 +70,11 @@ import {
import { getSettings } from "../lib/settings";
import { locationLabelOfDataDir } from "../lib/storageLocations";
import type { Paths } from "../lib/paths";
-import { listSites, siteStatsDir } from "../lib/site";
+import {
+ channelsOnlyOnUnlistedSites,
+ listSites,
+ siteStatsDir,
+} from "../lib/site";
import {
INDEX_SCANNED_AT_KEY,
STATS_SCHEMA_VERSION,
@@ -169,11 +173,12 @@ export type BuildStatsOptions = {
paths: Paths;
onLog?: (msg: string) => void;
signal?: AbortSignal;
- // When set, also write an UNFILTERED whole-pool stats bundle (every non-
- // excluded channel) into this dir as {manifest,page-NNNN}.json. Used by the
- // Archilyzer hub (compose-homepage), whose cross-site charts need the full
- // dataset rather than any one site's filtered slice. Forces collection of the
- // full dataset even when no per-site bundle needs a rebuild.
+ // When set, also write a whole-pool stats bundle (every non-excluded
+ // channel, but for those only unlisted sites expose) into this dir as
+ // {manifest,page-NNNN}.json. Used by the Archilyzer hub (compose-homepage),
+ // whose cross-site charts need the full dataset rather than any one site's
+ // filtered slice. Forces collection of the full dataset even when no per-site
+ // bundle needs a rebuild.
wholePoolStatsDir?: string;
};
@@ -723,16 +728,25 @@ export async function buildStats({
log(`Stats site ${plan.site.siteId}: ${filtered.length} videos, ${pageCount} page(s).`);
}
- // Whole-pool bundle for the hub: every non-excluded channel, unfiltered. Built
+ // Whole-pool bundle for the hub: every non-excluded channel, except a channel
+ // only unlisted sites expose (site.json `listed: false`): the bundle is
+ // published as the homepage's `stats/`, and an unlisted site's content is in
+ // no public total. Its own per-site bundle above is built as before. Built
// from the same in-memory dataset so it stays consistent with the per-site
// bundles. Always rewritten when requested (stats records are small).
if (wholePoolStatsDir) {
+ const unlistedOnly = channelsOnlyOnUnlistedSites(sites);
+ const pooled =
+ unlistedOnly.size > 0
+ ? all.filter((s) => !unlistedOnly.has(s.channelSlug))
+ : all;
const { pageCount } = await writePages(
wholePoolStatsDir,
- all,
+ pooled,
STATS_MAX_PAGE_BYTES,
);
const channelEntries: StatsChannelEntry[] = [...channels.keys()]
+ .filter((slug) => !unlistedOnly.has(slug))
.map((slug) => ({
slug,
name: channels.get(slug)?.name ?? slug,
@@ -742,13 +756,18 @@ export async function buildStats({
const manifest: StatsManifest = {
version: STATS_MANIFEST_VERSION,
generatedAt: new Date().toISOString(),
- totalCount: all.length,
+ totalCount: pooled.length,
pageCount,
maxPageBytes: STATS_MAX_PAGE_BYTES,
channels: channelEntries,
};
await writeJsonAtomic(path.join(wholePoolStatsDir, "manifest.json"), manifest);
- log(`Stats whole-pool: ${all.length} videos, ${pageCount} page(s).`);
+ log(
+ `Stats whole-pool: ${pooled.length} videos, ${pageCount} page(s)` +
+ (unlistedOnly.size > 0
+ ? `; ${unlistedOnly.size} channel(s) only unlisted sites expose left out.`
+ : "."),
+ );
}
// Prune fingerprints for sites that no longer exist (their staging dirs are
diff --git a/common/controller/poolSummary.test.ts b/common/controller/poolSummary.test.ts
@@ -0,0 +1,28 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { parseSite } from "../lib/siteSchema";
+import { channelSitesOf } from "./poolSummary";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common test
+//
+// channelSitesOf is the published `channel-sites.json` (compose-homepage) and
+// the map the summary attributes channels by. Release 14 slice HS: an unlisted
+// site (site.json `listed: false`) is in neither.
+
+test("channel-sites.json names listed sites only; a channel only an unlisted site exposes is absent", () => {
+ const sites = [
+ parseSite("fixture-a", { channels: [{ slug: "a1" }, { slug: "shared" }] }),
+ parseSite("fixture-b", { channels: [{ slug: "b1" }, { slug: "shared" }] }),
+ parseSite("fixture-unlisted", {
+ listed: false,
+ channels: [{ slug: "shared" }, { slug: "own" }],
+ }),
+ ];
+ assert.deepEqual(channelSitesOf(sites), {
+ a1: ["fixture-a"],
+ shared: ["fixture-a", "fixture-b"],
+ b1: ["fixture-b"],
+ });
+ assert.ok(!JSON.stringify(channelSitesOf(sites)).includes("fixture-unlisted"));
+});
diff --git a/common/controller/poolSummary.ts b/common/controller/poolSummary.ts
@@ -11,7 +11,7 @@ import path from "node:path";
import { mkdir, readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { buildStats } from "./buildStats";
-import { listSites, type Site } from "../lib/site";
+import { isListedSite, listSites, type Site } from "../lib/site";
import {
statsPageFileName,
type StatsManifest,
@@ -45,11 +45,14 @@ export async function readStatsPages(statsDir: string): Promise<VideoStat[]> {
return out;
}
-// channel slug -> the ids of the content sites that expose it. A channel on
-// multiple sites maps to all of them; a pool-only channel is simply absent.
-export function channelSitesOf(sites: Site[]): ChannelSitesMap {
+// channel slug -> the ids of the LISTED content sites that expose it: the
+// published `channel-sites.json`. A channel on multiple sites maps to all of
+// them; a pool-only channel is simply absent, and so is an unlisted site
+// (site.json `listed: false`) and a channel only unlisted sites expose.
+export function channelSitesOf(sites: readonly Site[]): ChannelSitesMap {
const channelSites: ChannelSitesMap = {};
for (const site of sites) {
+ if (!isListedSite(site)) continue;
for (const c of site.channels) {
(channelSites[c.slug] ??= []).push(site.siteId);
}
@@ -71,8 +74,9 @@ export async function buildPoolSummary(opts: {
}): Promise<PoolSummary> {
const { paths, statsDir } = opts;
await mkdir(statsDir, { recursive: true });
- // Whole-pool stats dataset (every non-excluded channel). buildStats also
- // refreshes the per-site bundles as a side effect, which is harmless.
+ // Whole-pool stats dataset (every non-excluded channel but those only
+ // unlisted sites expose). buildStats also refreshes the per-site bundles as a
+ // side effect, which is harmless.
await buildStats({ paths, wholePoolStatsDir: statsDir });
const sites = listSites(paths);
const channelSites = channelSitesOf(sites);
diff --git a/common/lib/homepageSummary.test.ts b/common/lib/homepageSummary.test.ts
@@ -277,3 +277,54 @@ test("a site's wordmark lead travels when it is a proper prefix of the title; ot
assert.ok(plain.sites.every((x) => !("wordmarkLead" in x)));
assert.equal(s.version, HOMEPAGE_SUMMARY_VERSION);
});
+
+// Release 14 slice HS: site.json `listed: false`. The site still builds and
+// deploys; the family's public pages do not list it, and no public total counts
+// the channels only it exposes.
+test("an unlisted site is in no array and no total; a channel it shares is the listed site's", () => {
+ const unlisted = { ...site("zeta", ["q1", "a1"], "https://zeta.example"), listed: false } as Site;
+ const sites = [...SITES, unlisted];
+ const channelSites = { ...CHANNEL_SITES, a1: ["alpha", "zeta"], q1: ["zeta"] };
+ const own = [
+ stat({ channelSlug: "q1", id: "q-1", uploadDate: "20251101", duration: 7200 }),
+ stat({ channelSlug: "q1", id: "q-2", uploadDate: "20260201", status: "deleted" }),
+ stat({ channelSlug: "q1", id: "q-3", hasTranscript: false, transcribedDate: null }),
+ ];
+ const s = buildHomepageSummary([...STATS, ...own], channelSites, sites, NOW);
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
+ // Byte for byte the summary without the unlisted site: every array (sites,
+ // channels, series, recent, monthly) and every total (totals, official,
+ // availability, monthlyUnplaced).
+ assert.deepEqual(s, base);
+ const text = JSON.stringify(s);
+ for (const needle of ["zeta", "ZETA", "q1", "Q1", "q-1"]) {
+ assert.ok(!text.includes(needle), needle);
+ }
+ // The shared channel stays credited to alpha.
+ assert.equal(s.sites.find((x) => x.siteId === "alpha")!.channels, 2);
+ assert.equal(s.version, 6);
+});
+
+test("an unlisted site with no siteUrl, or with every channel shared, changes nothing either", () => {
+ const base = buildHomepageSummary(STATS, CHANNEL_SITES, SITES, NOW);
+ // No siteUrl: never public, and its own channel is still in no total (unlike
+ // a pool-only channel, which `totals` counts).
+ const urlless = { ...site("zeta", ["r1"]), listed: false } as Site;
+ assert.deepEqual(
+ buildHomepageSummary(
+ [...STATS, stat({ channelSlug: "r1", id: "r-1" })],
+ { ...CHANNEL_SITES, r1: ["zeta"] },
+ [...SITES, urlless],
+ NOW,
+ ),
+ base,
+ );
+ const sharedOnly = { ...site("zeta", ["a1", "b1"], "https://zeta.example"), listed: false } as Site;
+ assert.deepEqual(
+ buildHomepageSummary(STATS, { ...CHANNEL_SITES, a1: ["alpha", "zeta"], b1: ["beta", "zeta"] }, [...SITES, sharedOnly], NOW),
+ base,
+ );
+ // Listed explicitly is the default: the same summary as no key at all.
+ const listedTrue = SITES.map((x) => ({ ...x, listed: true })) as Site[];
+ assert.deepEqual(buildHomepageSummary(STATS, CHANNEL_SITES, listedTrue, NOW), base);
+});
diff --git a/common/lib/homepageSummary.ts b/common/lib/homepageSummary.ts
@@ -1,6 +1,7 @@
import type { Platform } from "./platform";
import type { VideoStat } from "./stats";
import type { Site } from "./site";
+import { channelsOnlyOnUnlistedSites, isListedSite } from "./siteSchema";
import { accentHex, accentIdOf } from "./accent";
import { wordmarkLeadFor, type AccentId } from "./brand";
import { VIDEO_STATES, type VideoState } from "./availability";
@@ -10,11 +11,15 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// HTML, so the landing renders instantly without the browser fetching the
// multi-MB whole-pool stats dataset.
//
-// Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl), and
-// every video is attributed to a single PRIMARY public site (the first, by
-// sorted id, exposing its channel) so the Site and Channel breakdowns partition
-// the same set and combined totals stay honest. The KPI `totals` are instance-
-// wide (count pool-only / URL-less content too) — a deliberate scope difference.
+// Scope: the chart "universe" is PUBLIC sites only (those with a siteUrl that
+// are listed — site.json `listed`, lib/siteSchema.ts isListedSite), and every
+// video is attributed to a single PRIMARY public site (the first, by sorted id,
+// exposing its channel) so the Site and Channel breakdowns partition the same
+// set and combined totals stay honest. The KPI `totals` (and `availability`)
+// are instance-wide (count pool-only / URL-less content too) — a deliberate
+// scope difference — EXCEPT a channel only unlisted sites expose, which no
+// part of the summary counts (channelsOnlyOnUnlistedSites). An unlisted site is
+// in no array here; a channel it shares with a listed site is the listed one's.
//
// Two metrics are pre-binned at two granularities; Cumulative and Share (100%)
// are derived client-side from these, so no extra precompute is needed.
@@ -33,7 +38,12 @@ import { VIDEO_STATES, type VideoState } from "./availability";
// the same way — a summary without it paints its sites' hex, as before. So is
// the per-site `wordmarkLead` (release 14): a summary without it shows each
// card's title plain, as before. Nothing reads this number to accept a file.
-export const HOMEPAGE_SUMMARY_VERSION = 5;
+//
+// v6 (release 14): an unlisted site (site.json `listed: false`) is in no array,
+// and a channel only unlisted sites expose is in no total — `totals` and
+// `availability` included. No field was added or removed; the number says the
+// totals' scope moved.
+export const HOMEPAGE_SUMMARY_VERSION = 6;
// Day buckets are capped to this many trailing days so the embedded summary stays
// small regardless of archive age (daily detail is only useful recently).
@@ -166,7 +176,8 @@ export type HomepageOfficialTotals = {
export type HomepageSummary = {
version: number;
generatedAt: string; // ISO timestamp
- // Instance-wide headline numbers (count everything, not just public sites).
+ // Instance-wide headline numbers (count everything, not just public sites —
+ // but never a channel only unlisted sites expose).
totals: {
transcripts: number;
downloads: number;
@@ -351,12 +362,21 @@ export function buildHomepageSummary(
const nowWeek = weekOf(nowDay.replace(/-/g, ""))!;
const dayFloor = isoDate(new Date(now.getTime() - (DAY_WINDOW - 1) * 86400000));
- // Public sites only (need a link target + a stable place on the chart).
- const publicSites = sites.filter((s): s is Site & { siteUrl: string } => !!s.siteUrl);
+ // Public sites only (need a link target + a stable place on the chart), and
+ // only the listed ones.
+ const publicSites = sites.filter(
+ (s): s is Site & { siteUrl: string } => !!s.siteUrl && isListedSite(s),
+ );
const siteById = new Map(publicSites.map((s) => [s.siteId, s]));
+ // An unlisted site's own channels: counted nowhere below, totals included.
+ const unlistedOnly = channelsOnlyOnUnlistedSites(sites);
+ const inScope = unlistedOnly.size > 0
+ ? stats.filter((s) => !unlistedOnly.has(s.channelSlug))
+ : stats;
// channel slug -> primary public site (first by sorted id). Channels with no
- // public site are out of the chart universe entirely.
+ // public site are out of the chart universe entirely. Over listed sites only,
+ // so a channel an unlisted site shares with a listed one is the listed one's.
const primarySiteOf = new Map<string, string>();
for (const slug of Object.keys(channelSites)) {
const primary = [...channelSites[slug]]
@@ -369,7 +389,7 @@ export function buildHomepageSummary(
const channelName = new Map<string, string>();
const transcribedItems: Attributed[] = [];
const downloadedItems: Attributed[] = [];
- // Instance-wide KPI accumulators (count everything, not just public).
+ // Instance-wide KPI accumulators (count everything in scope, not just public).
let transcripts = 0;
let downloads = 0;
let hoursSeconds = 0;
@@ -405,7 +425,7 @@ export function buildHomepageSummary(
const uploadItems: { month: string; siteId: string }[] = [];
let monthlyUnplaced = 0;
- for (const s of stats) {
+ for (const s of inScope) {
// A transcript COUNTS whether or not it carries a date: transcripts,
// channels, hours, upload-month placement. Only the transcribed time series
// (and "this month", and the recent rail) need `transcribedDate`. buildStats
@@ -496,7 +516,7 @@ export function buildHomepageSummary(
.sort((a, b) => a.name.localeCompare(b.name));
// Recent feed (public universe), attributed to the primary site for its link.
- const recent: HomepageRecentItem[] = stats
+ const recent: HomepageRecentItem[] = inScope
.filter((s): s is VideoStat & { transcribedDate: string } =>
Boolean(s.hasTranscript && s.transcribedDate && primarySiteOf.get(s.channelSlug)),
)
@@ -522,16 +542,17 @@ export function buildHomepageSummary(
};
});
- // State census over every record, matching `totals`' instance-wide scope
- // (pool-only channels included) rather than the charts' public-site universe.
+ // State census over every in-scope record, matching `totals`' instance-wide
+ // scope (pool-only channels included, a channel only unlisted sites expose
+ // not) rather than the charts' public-site universe.
// Seeded with every state at zero so the shape is stable across corpora.
const byState = Object.fromEntries(
VIDEO_STATES.map((s) => [s, 0]),
) as Record<VideoState, number>;
- for (const s of stats) byState[s.status] = (byState[s.status] ?? 0) + 1;
+ for (const s of inScope) byState[s.status] = (byState[s.status] ?? 0) + 1;
const availability: HomepageAvailability = {
byState,
- counted: stats.length,
+ counted: inScope.length,
};
// Monthly back-catalogue series, keyed by the sites that survived the