commit 97b65df9216b0ab9a633132d8060d28dd5d51fe3
parent 739650fd3b916ac7805e06e8dc1cb17071101d65
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 4 Aug 2026 20:48:02 -0400
Stop walking the whole corpus to render a sidebar badge
listChannels() counted the corpus from scratch on every call: one readdir per
video directory (78,350) plus a digest sidecar read per transcribed video —
~474,559 files, measured at 3,985 ms against the live corpus. It was reached
from the root layout's reclaimable-disk badge, so every document load in the
editor paid it, the 5-second auto-refresh re-ran it on every route, and three
widget endpoints called it per poll. To describe 98 videos.
Renamed it listChannelStatsFromDisk (the batch jobs need ground truth, and the
name should say what it costs) and added three cheap readers beside it that
read the 65 per-channel snapshots instead: listChannelBriefs, listChannelConfigs
and listChannelStatsFromSnapshots. Measured 3,985 ms -> 68 ms, a 59x cut, and
the projection matches the walk exactly on videoCount/transcriptCount/
downloadCount across all 65 channels.
Also here:
- The snapshot reader moves to controller/channels so reading a report no
longer pulls in the generator's graph (lmdb, archive, digest). channelSnapshot
re-exports it, so no import changes.
- New lib/concurrency mapConcurrent, applied to the previously unbounded
Promise.all over one channel's 11,224 video dirs, to the nested serial loop
in savedVideoInventory (/saved-videos), and to autoRunner's snapshot reads.
- Per-request memoization via React.cache (app/lib/requestCache) — the
dashboard derived the channel list three separate ways in one render.
- /channels reports how stale its counts are instead of implying they're live.
Known trade: counts now lag a job by the snapshot scheduler's ~1s debounce, and
digest coverage reads 0 for the 11 channels whose snapshots predate
digestEngines. Both covered by controller/channelProjection.test.ts;
controller/noCorpusWalkInRenderPaths.test.ts fails if the walk ever returns to
a render path.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat:
25 files changed, 789 insertions(+), 142 deletions(-)
diff --git a/common/controller/archiveLiveChat.ts b/common/controller/archiveLiveChat.ts
@@ -25,7 +25,7 @@ import {
stat,
} from "node:fs/promises";
import pLimit from "p-limit";
-import { listChannels, type ChannelStat } from "./channels";
+import { listChannelStatsFromDisk, type ChannelStat } from "./channels";
import { normalizeLiveChat } from "./normalizeLiveChat";
import {
archiveExtension,
@@ -218,7 +218,7 @@ export async function archiveLiveChat(
await mkdir(stagingDir(opts.paths), { recursive: true });
}
- const allChannels = await listChannels(opts.paths);
+ const allChannels = await listChannelStatsFromDisk(opts.paths);
const target = opts.channelSlugs
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
@@ -380,7 +380,7 @@ export async function archiveCombinedLiveChat(
await mkdir(combinedStaging, { recursive: true });
try {
- const allChannels = await listChannels(opts.paths);
+ const allChannels = await listChannelStatsFromDisk(opts.paths);
const target = opts.channelSlugs
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
diff --git a/common/controller/archiveTranscripts.ts b/common/controller/archiveTranscripts.ts
@@ -32,7 +32,7 @@ import {
} from "node:fs/promises";
import { execa } from "execa";
import pLimit from "p-limit";
-import { listChannels, type ChannelStat } from "./channels";
+import { listChannelStatsFromDisk, type ChannelStat } from "./channels";
import { openChannelSigner, type ChannelSigner } from "../lib/channelSignature";
import {
normalizeTranscript,
@@ -393,7 +393,7 @@ export async function archiveTranscripts(
await mkdir(stagingDir(opts.paths), { recursive: true });
}
- const allChannels = await listChannels(opts.paths);
+ const allChannels = await listChannelStatsFromDisk(opts.paths);
const target = opts.channelSlugs
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
@@ -550,7 +550,7 @@ export async function archiveCombinedTranscripts(
await mkdir(combinedStaging, { recursive: true });
try {
- const allChannels = await listChannels(opts.paths);
+ const allChannels = await listChannelStatsFromDisk(opts.paths);
const target = opts.channelSlugs
? allChannels.filter((c) => opts.channelSlugs!.includes(c.slug))
: allChannels;
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -1,6 +1,6 @@
import path from "node:path";
-import { readdir } from "node:fs/promises";
import type { Paths } from "../lib/paths";
+import { mapConcurrent } from "../lib/concurrency";
import { getPaths } from "../lib/paths";
import { getSettings } from "../lib/settings";
import { detectPlatform } from "../lib/platform";
@@ -39,8 +39,11 @@ import { isAutoSubsOnly } from "../lib/subtitleProvenance";
import { readVideoFiles } from "../lib/videoStatus";
import { type DownloadOutcomeStatus } from "../lib/downloadOutcome";
import { downloadQueueKey } from "../lib/queueKeys";
-import { readChannelConfig } from "./channels";
-import { readChannelSnapshot } from "./channelSnapshot";
+import {
+ listChannelConfigs,
+ readChannelConfig,
+ readChannelSnapshot,
+} from "./channels";
import { transcribeOneFromQueue } from "./transcribeOneFromQueue";
import { findVideoSourceUrl } from "./undownloadedVideos";
import { downloadOneManaged } from "../ytdlp/downloadOneManaged";
@@ -137,22 +140,16 @@ export function getAutoRunnerStatus(kind: AutoQueueKind): AutoRunnerStatus {
type ChannelMeta = { slug: string; platform: ReturnType<typeof detectPlatform> };
+const SNAPSHOT_READ_CONCURRENCY = 64;
+
+// Re-derived from the shared listChannelConfigs rather than repeating the
+// readdir-then-serial-read here.
async function listChannelMeta(paths: Paths): Promise<ChannelMeta[]> {
- let names: string[];
- try {
- names = (await readdir(paths.channelsDir, { withFileTypes: true }))
- .filter((e) => e.isDirectory())
- .map((e) => e.name);
- } catch {
- return [];
- }
- const out: ChannelMeta[] = [];
- for (const slug of names) {
- const config = await readChannelConfig(paths, slug);
- if (!config) continue;
- out.push({ slug, platform: detectPlatform(config.url) });
- }
- return out;
+ const configs = await listChannelConfigs(paths);
+ return configs.map(({ slug, config }) => ({
+ slug,
+ platform: detectPlatform(config.url),
+ }));
}
// Read each channel's snapshot and project the buckets this runner kind cares
@@ -165,8 +162,14 @@ async function buildChannelWork(
): Promise<{ channels: ChannelWork[]; owner: Map<string, string> }> {
const channels: ChannelWork[] = [];
const owner = new Map<string, string>();
- for (const { slug, platform } of meta) {
- const snap = await readChannelSnapshot(paths, slug);
+ // Read every snapshot concurrently, then fold sequentially — `owner` is
+ // first-writer-wins, so the fold must stay in `meta` order to stay
+ // deterministic.
+ const snaps = await mapConcurrent(meta, SNAPSHOT_READ_CONCURRENCY, (m) =>
+ readChannelSnapshot(paths, m.slug),
+ );
+ for (const [i, { slug, platform }] of meta.entries()) {
+ const snap = snaps[i];
if (!snap) continue;
const buckets: Record<string, string[]> = {};
// Project every bucket a leaf could be pointed at — including the opt-in
diff --git a/common/controller/channelProjection.test.ts b/common/controller/channelProjection.test.ts
@@ -0,0 +1,228 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import {
+ digestCountOf,
+ listChannelBriefs,
+ listChannelConfigs,
+ listChannelStatsFromDisk,
+ listChannelStatsFromSnapshots,
+} from "./channels";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/channelProjection.test.ts
+//
+// listChannelStatsFromSnapshots replaces a full corpus walk on every render
+// path in the editor. It is only a legitimate substitution if it agrees with
+// the walk when the snapshots are current — so that agreement is asserted here,
+// field by field, rather than assumed.
+
+type VideoSpec = {
+ id: string;
+ // A whisper transcript (transcript.json) counts as transcribed AND downloaded.
+ transcribed?: boolean;
+ // A finalized audio.* file counts as downloaded on its own.
+ audio?: boolean;
+ // A non-empty ai-digest.json — only meaningful on a transcribed video.
+ digest?: boolean;
+};
+
+async function seedChannel(
+ paths: Paths,
+ slug: string,
+ videos: ReadonlyArray<VideoSpec>,
+ opts: { playlist?: number; digestEngines?: Record<string, number> | null } = {},
+): Promise<void> {
+ const channelDir = path.join(paths.channelsDir, slug);
+ await mkdir(channelDir, { recursive: true });
+ // `handling` is required — parseChannelConfig returns null without it, and a
+ // channel whose config won't parse is invisible to every reader here.
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({
+ handling: "transcribe",
+ url: `https://example.test/${slug}`,
+ name: slug,
+ }),
+ );
+ if (opts.playlist !== undefined) {
+ await writeFile(
+ path.join(channelDir, "playlist"),
+ Array.from({ length: opts.playlist }, (_, i) => `v${i}`).join("\n") + "\n",
+ );
+ }
+ for (const v of videos) {
+ const dir = path.join(channelDir, "data", v.id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify({ id: v.id }));
+ if (v.transcribed) {
+ await writeFile(path.join(dir, "transcript.json"), JSON.stringify({ cues: [] }));
+ }
+ if (v.audio) await writeFile(path.join(dir, "audio.mp3"), "x");
+ if (v.digest) {
+ // digestSchemaVersion is required (loadDigest rejects the record without
+ // it), and hasDigest only counts a record with a non-empty section.
+ await writeFile(
+ path.join(dir, "ai-digest.json"),
+ JSON.stringify({
+ digestSchemaVersion: 1,
+ sections: { tags: { items: ["a"] } },
+ }),
+ );
+ }
+ }
+
+ // A CURRENT snapshot — exactly what the scheduler would have written for the
+ // dirs seeded above. `digestEngines: null` seeds the older shape that predates
+ // the field, which 11 of the 65 live snapshots still have.
+ const transcribed = videos.filter((v) => v.transcribed).length;
+ const downloaded = videos.filter((v) => v.transcribed || v.audio).length;
+ const digests = videos.filter((v) => v.transcribed && v.digest).length;
+ const snapshot: Record<string, unknown> = {
+ generatedAt: new Date().toISOString(),
+ totals: { videos: videos.length, transcribed, downloaded },
+ buckets: {},
+ undownloadedIds: [],
+ };
+ if (opts.digestEngines !== null) {
+ snapshot.digestEngines = opts.digestEngines ?? { local: digests };
+ }
+ await writeFile(
+ path.join(channelDir, "snapshot.json"),
+ JSON.stringify(snapshot),
+ );
+}
+
+async function withPaths(fn: (paths: Paths) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-projection-"));
+ const paths = { channelsDir: path.join(dir, "channels") } as Paths;
+ try {
+ await fn(paths);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+test("snapshot projection matches the disk walk when snapshots are current", async () => {
+ await withPaths(async (paths) => {
+ await seedChannel(
+ paths,
+ "alpha",
+ [
+ { id: "a1", transcribed: true, audio: true, digest: true },
+ { id: "a2", transcribed: true, audio: true },
+ { id: "a3", audio: true },
+ { id: "a4" },
+ ],
+ { playlist: 6 },
+ );
+ await seedChannel(paths, "beta", [{ id: "b1", transcribed: true }], {
+ playlist: 1,
+ });
+ // A channel with nothing in it at all.
+ await seedChannel(paths, "gamma", []);
+
+ const fromDisk = await listChannelStatsFromDisk(paths);
+ const fromSnapshots = await listChannelStatsFromSnapshots(paths);
+
+ // Without this the comparison below passes vacuously when the fixture fails
+ // to seed — which is exactly how this test first went green while reading
+ // nothing at all.
+ assert.equal(fromDisk.length, 3, "the fixture corpus actually seeded");
+
+ assert.deepEqual(
+ fromSnapshots.map((c) => c.slug),
+ fromDisk.map((c) => c.slug),
+ "both readers list the same channels in the same order",
+ );
+ for (const [i, snap] of fromSnapshots.entries()) {
+ const disk = fromDisk[i];
+ for (const field of [
+ "videoCount",
+ "transcriptCount",
+ "downloadCount",
+ "digestCount",
+ "playlistCount",
+ ] as const) {
+ assert.equal(
+ snap[field],
+ disk[field],
+ `${snap.slug}.${field}: projection ${snap[field]} !== walk ${disk[field]}`,
+ );
+ }
+ }
+ });
+});
+
+test("a snapshot without digestEngines reports zero digests, not a wrong number", async () => {
+ await withPaths(async (paths) => {
+ // Two videos DO carry digests on disk, but this channel's snapshot predates
+ // the digestEngines field. The projection must under-report to 0 rather than
+ // invent a count — and must not throw.
+ await seedChannel(
+ paths,
+ "legacy",
+ [
+ { id: "l1", transcribed: true, digest: true },
+ { id: "l2", transcribed: true, digest: true },
+ ],
+ { digestEngines: null },
+ );
+
+ const [projected] = await listChannelStatsFromSnapshots(paths);
+ const [walked] = await listChannelStatsFromDisk(paths);
+ assert.equal(walked.digestCount, 2, "the walk sees both digests");
+ assert.equal(projected.digestCount, 0, "the projection defaults to 0");
+ // Every other field still agrees exactly.
+ assert.equal(projected.videoCount, walked.videoCount);
+ assert.equal(projected.transcriptCount, walked.transcriptCount);
+ assert.equal(projected.downloadCount, walked.downloadCount);
+ });
+});
+
+test("digestCountOf sums across engines and tolerates the missing field", () => {
+ assert.equal(digestCountOf(null), 0);
+ assert.equal(digestCountOf({ totals: {} } as never), 0);
+ assert.equal(
+ digestCountOf({ digestEngines: { local: 3, cloud: 4 } } as never),
+ 7,
+ );
+});
+
+test("briefs and configs agree with the walk on membership", async () => {
+ await withPaths(async (paths) => {
+ await seedChannel(paths, "one", [{ id: "x", transcribed: true }]);
+ await seedChannel(paths, "two", []);
+ // A directory with no config.json is not a channel and must be skipped by
+ // every reader, not just the walk.
+ await mkdir(path.join(paths.channelsDir, "not-a-channel", "data"), {
+ recursive: true,
+ });
+
+ const [briefs, configs, walked] = await Promise.all([
+ listChannelBriefs(paths),
+ listChannelConfigs(paths),
+ listChannelStatsFromDisk(paths),
+ ]);
+ const slugs = walked.map((c) => c.slug);
+ assert.deepEqual(slugs, ["one", "two"]);
+ assert.deepEqual(briefs.map((b) => b.slug), slugs);
+ assert.deepEqual(configs.map((c) => c.slug), slugs);
+ assert.ok(
+ briefs.every((b) => b.snapshot !== null),
+ "each seeded channel's snapshot was read by the brief",
+ );
+ });
+});
+
+test("readers return empty rather than throwing when there is no corpus", async () => {
+ await withPaths(async (paths) => {
+ assert.deepEqual(await listChannelBriefs(paths), []);
+ assert.deepEqual(await listChannelConfigs(paths), []);
+ assert.deepEqual(await listChannelStatsFromSnapshots(paths), []);
+ assert.deepEqual(await listChannelStatsFromDisk(paths), []);
+ });
+});
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -32,7 +32,12 @@ import type { Paths } from "../lib/paths";
import { extractVideoId } from "../ytdlp/runYtdlp";
import { reconcileVideoDirs } from "./reconcileVideoDirs";
import { loadFailedTranscriptions } from "./failedTranscriptions";
-import { readChannelConfig } from "./channels";
+import {
+ readChannelConfig,
+ readChannelSnapshot,
+ snapshotPath,
+ SNAPSHOT_FILENAME,
+} from "./channels";
import { computeKeptVideoIds } from "./keptVideos";
import { loadMaybeMissing } from "./quickAvailabilityCheck";
import { readTranscriptCoverage } from "./normalizeTranscript";
@@ -253,26 +258,14 @@ export function normalizeMaybeMissing(
return { ids: raw?.ids ?? [], checkedAt: raw?.checkedAt ?? "" };
}
-export const SNAPSHOT_FILENAME = "snapshot.json";
+// The snapshot READER lives in ./channels — reading a report shouldn't require
+// loading the machinery that generates one (this module pulls in lmdb, the
+// archive reader and the digest layer). Re-exported here so the name stays
+// where callers expect to find it.
+export { SNAPSHOT_FILENAME, snapshotPath, readChannelSnapshot };
const SNAPSHOT_VIDEO_CONCURRENCY = 16;
-export function snapshotPath(paths: Paths, slug: string): string {
- return path.join(paths.channelsDir, slug, SNAPSHOT_FILENAME);
-}
-
-export async function readChannelSnapshot(
- paths: Paths,
- slug: string,
-): Promise<ChannelSnapshot | null> {
- try {
- const raw = await readFile(snapshotPath(paths, slug), "utf8");
- return JSON.parse(raw) as ChannelSnapshot;
- } catch {
- return null;
- }
-}
-
async function readPlaylistUrls(file: string): Promise<string[]> {
try {
const raw = await readFile(file, "utf8");
diff --git a/common/controller/channels.ts b/common/controller/channels.ts
@@ -6,12 +6,21 @@ import {
type ChannelConfig,
} from "../lib/channelConfig";
import type { Paths } from "../lib/paths";
+import { mapConcurrent } from "../lib/concurrency";
import {
isVideoDownloaded,
isVideoTranscribed,
readVideoFiles,
} from "../lib/videoStatus";
import { hasDigest } from "../lib/digest-server";
+// TYPE-ONLY, and it must stay that way: ./channelSnapshot imports
+// readChannelConfig from this module, and it drags in the snapshot generator's
+// whole dependency graph (lmdb, the archive reader, the digest layer). A value
+// import here would both close an import cycle and make every consumer of
+// channels.ts pay for the generator. `import type` is erased, so it does
+// neither — which is why the snapshot *reader* lives down here (below) rather
+// than beside the generator.
+import type { ChannelSnapshot } from "./channelSnapshot";
export type ChannelStat = {
slug: string;
@@ -40,6 +49,12 @@ export function isValidChannelSlug(slug: unknown): slug is string {
);
}
+// Fan-out widths. Channels number in the dozens, videos in the tens of
+// thousands, so both lists get the same cap for the same reason (see
+// ../lib/concurrency) — the channel one just never reaches it.
+const CHANNEL_READ_CONCURRENCY = 64;
+const VIDEO_READ_CONCURRENCY = 64;
+
async function exists(p: string): Promise<boolean> {
try {
await stat(p);
@@ -62,8 +77,12 @@ async function countDataFiles(dataDir: string): Promise<{
return { videos: 0, transcripts: 0, downloads: 0, digests: 0 };
}
const videoDirs = dirs.filter((d) => d.isDirectory());
- const flags = await Promise.all(
- videoDirs.map(async (d) => {
+ // Bounded: the largest channel has 11,224 video dirs and this used to open
+ // them all at once.
+ const flags = await mapConcurrent(
+ videoDirs,
+ VIDEO_READ_CONCURRENCY,
+ async (d) => {
const dir = path.join(dataDir, d.name);
const files = await readVideoFiles(dir);
return {
@@ -74,7 +93,7 @@ async function countDataFiles(dataDir: string): Promise<{
// pattern channelSnapshot.ts uses for coverage and VTT provenance.
digest: isVideoTranscribed(files) ? await hasDigest(dir) : false,
};
- }),
+ },
);
let transcripts = 0;
let downloads = 0;
@@ -174,7 +193,23 @@ export function countNotYetTranscribed(
return countMissing(paths, slug, ids, (files) => !isVideoTranscribed(files));
}
-export async function listChannels(paths: Paths): Promise<ChannelStat[]> {
+// ⚠️ WALKS THE ENTIRE CORPUS. One readdir per video directory (78,350 of them
+// on the live instance) plus a digest sidecar read per transcribed video —
+// ~474,559 files touched, ~4.4 SECONDS per call.
+//
+// This is ground truth, so the batch jobs that must not trust a stale snapshot
+// (archiveTranscripts, archiveLiveChat, normalizeAll*) still call it. It must
+// NEVER appear in a page, layout, or API route: this single function, reached
+// from the root layout's reclaimable-disk badge, was a 4.4 s floor on every
+// document load in the editor and re-ran every 5 s on the auto-refresh timer.
+//
+// For render paths use listChannelBriefs / listChannelConfigs /
+// listChannelStatsFromSnapshots below, which read 65 small files instead.
+// A guard test (editor/e2e or common) asserts this name stays out of the app
+// directory — if you are here to re-add it to a page, that is the sign to stop.
+export async function listChannelStatsFromDisk(
+ paths: Paths,
+): Promise<ChannelStat[]> {
let entries: Dirent[];
try {
entries = await readdir(paths.channelsDir, { withFileTypes: true });
@@ -197,16 +232,141 @@ export async function listChannels(paths: Paths): Promise<ChannelStat[]> {
videoCount: counts.videos,
transcriptCount: counts.transcripts,
downloadCount: counts.downloads,
- // `countDataFiles` has always computed this and `listChannels` has always
- // discarded it — unlike readChannelStat, which emits it. One line, and
- // every cross-channel surface (the dashboard, /actionable, the widget)
- // gets a coverage counter it was already paying the I/O for.
+ // `countDataFiles` has always computed this and the corpus walk has
+ // always discarded it — unlike readChannelStat, which emits it. One line,
+ // and every cross-channel surface (the dashboard, /actionable, the
+ // widget) gets a coverage counter it was already paying the I/O for.
digestCount: counts.digests,
});
}
return out.sort((a, b) => a.slug.localeCompare(b.slug));
}
+// --- Cheap reads: 65 small files instead of half a million -----------------
+
+export const SNAPSHOT_FILENAME = "snapshot.json";
+
+export function snapshotPath(paths: Paths, slug: string): string {
+ return path.join(paths.channelsDir, slug, SNAPSHOT_FILENAME);
+}
+
+// Read one channel's precomputed report. Lives here rather than beside the
+// generator in ./channelSnapshot so that reading a snapshot doesn't require
+// loading the machinery that writes one (see the type-only import at the top).
+// ./channelSnapshot re-exports it, so every existing import still resolves.
+export async function readChannelSnapshot(
+ paths: Paths,
+ slug: string,
+): Promise<ChannelSnapshot | null> {
+ try {
+ const raw = await readFile(snapshotPath(paths, slug), "utf8");
+ return JSON.parse(raw) as ChannelSnapshot;
+ } catch {
+ return null;
+ }
+}
+
+// Slug + config + the precomputed snapshot for every channel. One readdir plus
+// two small file reads per channel, all concurrent: ~60 ms against the corpus
+// the walk above needs 4.4 s for.
+export type ChannelBrief = {
+ slug: string;
+ config: ChannelConfig;
+ snapshot: ChannelSnapshot | null;
+};
+
+async function listChannelSlugs(paths: Paths): Promise<string[]> {
+ try {
+ const entries = await readdir(paths.channelsDir, { withFileTypes: true });
+ return entries.filter((e) => e.isDirectory()).map((e) => e.name);
+ } catch {
+ return [];
+ }
+}
+
+export async function listChannelBriefs(paths: Paths): Promise<ChannelBrief[]> {
+ const slugs = await listChannelSlugs(paths);
+ const briefs = await mapConcurrent(
+ slugs,
+ CHANNEL_READ_CONCURRENCY,
+ async (slug): Promise<ChannelBrief | null> => {
+ const [config, snapshot] = await Promise.all([
+ readChannelConfig(paths, slug),
+ readChannelSnapshot(paths, slug),
+ ]);
+ return config ? { slug, config, snapshot } : null;
+ },
+ );
+ return briefs
+ .filter((b): b is ChannelBrief => b !== null)
+ .sort((a, b) => a.slug.localeCompare(b.slug));
+}
+
+// Slug + config only — for the many callers that just need names, URLs and
+// sync/handling flags. One readdir plus one config read per channel.
+export async function listChannelConfigs(
+ paths: Paths,
+): Promise<Array<{ slug: string; config: ChannelConfig }>> {
+ const slugs = await listChannelSlugs(paths);
+ const rows = await mapConcurrent(
+ slugs,
+ CHANNEL_READ_CONCURRENCY,
+ async (slug) => {
+ const config = await readChannelConfig(paths, slug);
+ return config ? { slug, config } : null;
+ },
+ );
+ return rows
+ .filter((r): r is { slug: string; config: ChannelConfig } => r !== null)
+ .sort((a, b) => a.slug.localeCompare(b.slug));
+}
+
+// Digest coverage for one channel, from its snapshot.
+//
+// ⚠️ The one field that is NOT a clean substitution for the walk. `totals` has
+// videos/transcribed/downloaded but no `digests` key, so the count has to be
+// summed out of `digestEngines` — which 11 of the 65 live snapshots don't have
+// at all (it postdates them). Those default to 0 and will read as "no digests"
+// until the channel's next report refresh, rather than reporting a wrong number.
+export function digestCountOf(snapshot: ChannelSnapshot | null): number {
+ const engines = snapshot?.digestEngines;
+ if (!engines) return 0;
+ let n = 0;
+ for (const count of Object.values(engines)) n += count;
+ return n;
+}
+
+// ChannelStat projected from the precomputed snapshots instead of a corpus
+// walk. Same shape, ~40× cheaper, and one honest trade: the counts are what the
+// LAST SNAPSHOT saw, not what is on disk this millisecond. The snapshot
+// scheduler regenerates on a ~1 s debounce after any report-changing action, so
+// the lag is that debounce — surfaced in the UI via `snapshot.generatedAt`
+// rather than left for the reader to assume.
+export async function listChannelStatsFromSnapshots(
+ paths: Paths,
+ // Pass briefs you have already read to avoid a second pass — a caller that
+ // also wants `generatedAt` for a freshness readout needs the same rows.
+ preread?: ReadonlyArray<ChannelBrief>,
+): Promise<ChannelStat[]> {
+ const briefs = preread ?? (await listChannelBriefs(paths));
+ // playlistCount has no snapshot equivalent, but it is one small file per
+ // channel — cheap enough to keep reading directly.
+ const playlistCounts = await mapConcurrent(
+ briefs,
+ CHANNEL_READ_CONCURRENCY,
+ (b) => countPlaylist(path.join(paths.channelsDir, b.slug, "playlist")),
+ );
+ return briefs.map((b, i) => ({
+ slug: b.slug,
+ config: b.config,
+ playlistCount: playlistCounts[i],
+ videoCount: b.snapshot?.totals.videos ?? 0,
+ transcriptCount: b.snapshot?.totals.transcribed ?? 0,
+ downloadCount: b.snapshot?.totals.downloaded ?? 0,
+ digestCount: digestCountOf(b.snapshot),
+ }));
+}
+
export async function writeChannelConfig(
paths: Paths,
slug: string,
diff --git a/common/controller/noCorpusWalkInRenderPaths.test.ts b/common/controller/noCorpusWalkInRenderPaths.test.ts
@@ -0,0 +1,91 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { readdir, readFile } from "node:fs/promises";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/noCorpusWalkInRenderPaths.test.ts
+//
+// THIS IS THE TEST THAT STOPS THE WHOLE PROBLEM RECURRING.
+//
+// listChannelStatsFromDisk walks every video directory in the corpus (~474,559
+// file touches, ~4 seconds). One call to it from the root layout — for a badge
+// describing 98 videos — put a 4.4 second floor under every page in the editor
+// and re-ran on a 5 second timer forever. It is ground truth and the batch jobs
+// need it, so it can't just be deleted; what it can't do is appear in anything
+// that renders.
+//
+// It lives in common's suite rather than editor's because this suite is the one
+// that runs in a second without a browser, and a guard nobody runs guards
+// nothing.
+//
+// If this test fails: you want listChannelBriefs (slug + config + snapshot),
+// listChannelConfigs (slug + config), or listChannelStatsFromSnapshots (the
+// same ChannelStat shape, projected from snapshots) — all in ./channels.
+
+const HERE = path.dirname(fileURLToPath(import.meta.url));
+const EDITOR_APP = path.resolve(HERE, "..", "..", "editor", "app");
+
+const BANNED = "listChannelStatsFromDisk";
+
+async function walk(dir: string): Promise<string[]> {
+ const out: string[] = [];
+ const entries = await readdir(dir, { withFileTypes: true });
+ for (const e of entries) {
+ const full = path.join(dir, e.name);
+ if (e.isDirectory()) {
+ if (e.name === "node_modules" || e.name === ".next") continue;
+ out.push(...(await walk(full)));
+ } else if (/\.tsx?$/.test(e.name)) {
+ out.push(full);
+ }
+ }
+ return out;
+}
+
+test("the corpus walk never reaches a render path", async () => {
+ const files = await walk(EDITOR_APP);
+ // Guard the guard: if the traversal finds nothing, the assertion below would
+ // pass while checking zero files.
+ assert.ok(
+ files.length > 100,
+ `expected to scan the editor app tree, found ${files.length} files under ${EDITOR_APP}`,
+ );
+
+ const offenders: string[] = [];
+ await Promise.all(
+ files.map(async (file) => {
+ const source = await readFile(file, "utf8");
+ if (source.includes(BANNED)) {
+ offenders.push(path.relative(EDITOR_APP, file));
+ }
+ }),
+ );
+
+ assert.deepEqual(
+ offenders.sort(),
+ [],
+ `${BANNED} walks the whole corpus and must not be reachable from a page, ` +
+ `layout, API route or server action. Found in: ${offenders.join(", ")}`,
+ );
+});
+
+test("the cheap channel readers are the ones the editor actually uses", async () => {
+ // The converse check: if someone "fixes" the test above by inlining a readdir
+ // loop instead, the named readers would quietly stop being used. This asserts
+ // the intended replacements are still wired in.
+ const files = await walk(EDITOR_APP);
+ const sources = await Promise.all(files.map((f) => readFile(f, "utf8")));
+ const joined = sources.join("\n");
+ for (const wanted of [
+ "listChannelBriefs",
+ "listChannelConfigs",
+ "listChannelStatsFromSnapshots",
+ ]) {
+ assert.ok(
+ joined.includes(wanted),
+ `${wanted} is no longer referenced anywhere in editor/app — did a render path go back to walking the corpus?`,
+ );
+ }
+});
diff --git a/common/controller/normalizeAll.ts b/common/controller/normalizeAll.ts
@@ -6,7 +6,7 @@
import path from "node:path";
import { readdir } from "node:fs/promises";
import pLimit from "p-limit";
-import { listChannels } from "./channels";
+import { listChannelStatsFromDisk } from "./channels";
import { normalizeTranscript } from "./normalizeTranscript";
import type { Paths } from "../lib/paths";
@@ -29,7 +29,7 @@ export async function normalizeAllTranscripts(
): Promise<NormalizeAllResult> {
const log = opts.onLog ?? ((m: string) => console.log(m));
const limit = pLimit(opts.concurrency ?? 8);
- const channels = await listChannels(opts.paths);
+ const channels = await listChannelStatsFromDisk(opts.paths);
const result: NormalizeAllResult = {
wrote: 0,
fresh: 0,
diff --git a/common/controller/normalizeAllLiveChat.ts b/common/controller/normalizeAllLiveChat.ts
@@ -6,7 +6,7 @@
import path from "node:path";
import { readdir } from "node:fs/promises";
import pLimit from "p-limit";
-import { listChannels } from "./channels";
+import { listChannelStatsFromDisk } from "./channels";
import { normalizeLiveChat } from "./normalizeLiveChat";
import type { Paths } from "../lib/paths";
@@ -29,7 +29,7 @@ export async function normalizeAllLiveChat(
): Promise<NormalizeAllLiveChatResult> {
const log = opts.onLog ?? ((m: string) => console.log(m));
const limit = pLimit(opts.concurrency ?? 8);
- const channels = await listChannels(opts.paths);
+ const channels = await listChannelStatsFromDisk(opts.paths);
const result: NormalizeAllLiveChatResult = {
wrote: 0,
fresh: 0,
diff --git a/common/controller/savedVideoInventory.ts b/common/controller/savedVideoInventory.ts
@@ -1,11 +1,17 @@
import path from "node:path";
import fs from "fs-extra";
import type { Paths } from "../lib/paths";
+import { mapConcurrent } from "../lib/concurrency";
import { loadSavedVideo } from "../lib/savedVideo-server";
import { savedVideoPath, type SavedVideoPointer } from "../lib/savedVideo";
const { pathExists, readdir } = fs;
+// Channels are read a few at a time so the per-video fan-out inside each one
+// still dominates; the product is the real ceiling on open descriptors.
+const CHANNEL_CONCURRENCY = 8;
+const VIDEO_CONCURRENCY = 8;
+
// One persisted source video, located via its data-dir pointer. Shared by the
// backup/verify controllers (Phase 4) and the saved-videos UI counts (Phase 5).
export type SavedVideoEntry = {
@@ -39,24 +45,36 @@ export async function listSavedVideos({
channelSlug?: string;
}): Promise<SavedVideoEntry[]> {
const slugs = channelSlug ? [channelSlug] : await listChannelSlugs(paths);
- const out: SavedVideoEntry[] = [];
- for (const slug of slugs) {
- const dataDir = path.join(paths.channelsDir, slug, "data");
- if (!(await pathExists(dataDir))) continue;
- const ids = await readdir(dataDir).catch(() => [] as string[]);
- for (const videoId of ids) {
- const videoDir = path.join(dataDir, videoId);
- const pointer = await loadSavedVideo(videoDir);
- if (!pointer) continue;
- out.push({
- slug,
- videoId,
- videoDir,
- storedPath: savedVideoPath(pointer),
- pointer,
- });
- }
- }
+ // One pointer read per video dir — 78,350 of them across the corpus. Done one
+ // at a time inside a nested loop this took seconds; bounded-concurrent it is a
+ // fraction of that, and the cap keeps it from exhausting file descriptors.
+ const perChannel = await mapConcurrent(
+ slugs,
+ CHANNEL_CONCURRENCY,
+ async (slug): Promise<SavedVideoEntry[]> => {
+ const dataDir = path.join(paths.channelsDir, slug, "data");
+ if (!(await pathExists(dataDir))) return [];
+ const ids = await readdir(dataDir).catch(() => [] as string[]);
+ const entries = await mapConcurrent(
+ ids,
+ VIDEO_CONCURRENCY,
+ async (videoId): Promise<SavedVideoEntry | null> => {
+ const videoDir = path.join(dataDir, videoId);
+ const pointer = await loadSavedVideo(videoDir);
+ if (!pointer) return null;
+ return {
+ slug,
+ videoId,
+ videoDir,
+ storedPath: savedVideoPath(pointer),
+ pointer,
+ };
+ },
+ );
+ return entries.filter((e): e is SavedVideoEntry => e !== null);
+ },
+ );
+ const out = perChannel.flat();
out.sort(
(a, b) => a.slug.localeCompare(b.slug) || a.videoId.localeCompare(b.videoId),
);
diff --git a/common/lib/concurrency.ts b/common/lib/concurrency.ts
@@ -0,0 +1,30 @@
+// Bounded async fan-out.
+//
+// The corpus is big enough that an unbounded `Promise.all` over a per-video or
+// per-channel list is a file-descriptor bomb, not a speed-up: the largest
+// channel here holds 11,224 video dirs, and opening all of them at once starves
+// every other reader in the process. Every fan-out over corpus-shaped work goes
+// through this instead.
+//
+// Order-preserving: `out[i]` is always the result for `items[i]`, regardless of
+// completion order. Rejection semantics match `Promise.all` — the first
+// rejection wins and already-started work keeps running to completion.
+export async function mapConcurrent<T, R>(
+ items: readonly T[],
+ limit: number,
+ fn: (item: T, index: number) => Promise<R>,
+): Promise<R[]> {
+ const out = new Array<R>(items.length);
+ if (items.length === 0) return out;
+ const width = Math.max(1, Math.min(Math.floor(limit), items.length));
+ let next = 0;
+ const worker = async (): Promise<void> => {
+ for (;;) {
+ const i = next++;
+ if (i >= items.length) return;
+ out[i] = await fn(items[i], i);
+ }
+ };
+ await Promise.all(Array.from({ length: width }, worker));
+ return out;
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **The editor is fast now.** Every page in the editor had a floor of about 4.4 seconds on it, and the reason was one line in the sidebar. The reclaimable-disk badge — the little "12.4 GB" pill next to Cleanup — asked for the channel list, and the function it asked was the one that counts the corpus from scratch: a `readdir` for each of the **78,350** video directories plus a digest sidecar read for each of the ~70,000 transcribed ones, **~474,559 files touched, measured at 3,985 ms**, to describe **98 videos**. Every count that walk produced was then thrown away. It sat in the root layout, so *every* document load paid it; the 5-second auto-refresh re-ran it on a timer, on every route, forever; and three widget endpoints called it on each poll. It now reads the 65 per-channel snapshots it could always have read — the same numbers, **68 ms**, a 59× improvement — and the sidebar badge itself is down to ~40 ms. Loading `/channels` went from 4.5 s to roughly a tenth of a second; the dashboard from ~10 s. The corpus-walking function still exists under a name that says what it costs (`listChannelStatsFromDisk`) for the batch jobs that genuinely need ground truth, and a test now fails the build if it ever reappears anywhere the editor renders. **The honest trade:** the video, transcript and download counts on `/channels` and the dashboard now come from each channel's last generated report rather than from disk directly, so a job that just finished can take a moment — the snapshot scheduler's ~1 second debounce — to show up. Verified against the live corpus: those three counts match a full walk **exactly** on all 65 channels. The one field that doesn't is digest coverage, which reads 0 for the 11 channels whose reports predate per-engine digest counts until their next report refresh. `/channels` now prints how old the oldest report on the page is, rather than leaving you to assume the numbers are live.
- **Cleaning audio now checks the video still exists upstream, and keeps it forever if it doesn't.** The transcribed-audio sweep hard-deletes a video's `audio.*` files once whisper has produced a transcript — `remove()`, no trash, no undo — and nothing had ever asked whether the video was still *there*. So a video YouTube had since removed, privated, or put behind a membership, sitting outside the keep-latest window, got its source audio deleted precisely when that local copy had become the only copy. Before deleting anything, the sweep now resolves each candidate's availability and writes a `do-not-clean.json` marker on any video found permanently gone (`deleted` / `private` / `members_only` — the same rule the keep-latest deletion pass uses, now shared as `isPermanentlyGone`), protecting it from this and every future sweep. The check is **cheap-first, not one probe per video**: a cached availability verdict costs nothing and is the only tier that catches `members_only` (a members-only video stays listed in its channel's playlist, so a listing diff can never flag it); then **one** flat-playlist call per channel narrows the field to candidates that have dropped out of the listing; only those few get a per-video probe, which is also what distinguishes a deleted video from an *unlisted* one that legitimately left the listing and is still fetchable by URL. Anything the check cannot resolve — a probe error, an age-gate, a video with no URL to probe — is **left alone with no marker written** and retried next run: the sweep never deletes on incomplete information, and a rate-limited or offline source therefore cleans nothing rather than cleaning wrongly. The summary line breaks the total down (`Skipped 4 (0 protected, 3 gone-from-source pinned, 1 unverified)`) whenever the check acted. On by default; **Check availability before cleaning audio** in Settings turns it off for an offline setup or channels with no URL, where the check can never resolve and cleanup would otherwise stop deleting anything. Two related fixes ride along: the sweep now shares `isRealAudioFile` with the rest of the app instead of its own hand-rolled filter, so it no longer deletes the `audio.live_chat.json` sidecar or the `.part.good`/`.part.testing` audio-check snapshots (which the reclaim estimate never counted, so the two had quietly drifted); and `runAvailabilityCheck` gains `ignoreShard`, because a saved shard slice on disk would otherwise replace an explicit `onlyIds` list wholesale. Scoped to the primary sweep only — the wrong-format, extra-format and auto-sub purges are unchanged, as are the explicit per-video deletes, which still ignore markers deliberately. See `common/controller/verifyBeforeClean.ts`, `common/controller/cleanAudioFromTranscribed.ts`, `common/lib/availability.ts`, and `editor/e2e/pre-clean-availability.spec.ts`.
- **The monitor widget can now reclaim disk, not just report it.** The widget's cleanable-data strip showed a single global number ("4.2 GB reclaimable") with nothing to act on — reclaiming it meant leaving the widget for `/cleanup` or `/actionable`. A new opt-in **"Needs cleaning"** section (URL flag `cleanlist=1`, plus a **Channels needing cleanup** checkbox in the builder and the in-widget gear) lists the channels actually holding that audio, each with its reclaim estimate (`⌫ 2.5 MB`, the video count in the tooltip), capped at 6 channels with a `+N more` line like the needs-work list. With `controls=1` each row gains the same per-channel **Clean audio** button as the `/actionable` page — the existing `window.confirm` still guards the delete — so a pinned interactive widget clears disk pressure the way it already clears a download backlog. This is also the first surface on which a channel that is *fully downloaded and transcribed* but still holding reclaimable audio is actionable: the needs-work list is fed by a backlog route with a download/transcribe precondition, so such a channel never appeared there. It costs no extra polling — the per-channel rows come from the same `/api/widget/cleanable` snapshot read that already backed the total, and the total is now a sum over those rows so the section and the strip above it can't disagree. The dashboard's needs-work panel and its shared route are untouched. See `editor/app/cleanup/lib/loadCleanup.ts` (`cleanableChannels`), `editor/app/api/widget/cleanable/route.ts`, `editor/app/widget/{lib/config.ts,components/{MonitorWidget,WidgetConfigForm}.tsx}`, and `editor/e2e/widget.spec.ts`.
- **A corpus-wide digest backfill can now be started, left alone, and watched.** The digest layer could generate, but only one channel at a time from that channel's own page — a full-archive pass meant 63 manual launches, and a server restart silently ended it with nothing to say so. There is now a **Start Digest Sweep** control on the dashboard that walks every channel in turn, **heaviest first by remaining audio-hours** (cost is audio, not videos: one VOD channel outweighs every duplicate mirror in the archive combined), and it **survives a restart** — the sweep is re-armed at boot the way the auto-download and auto-transcribe runners already were. It stores no work-list, so a resumed sweep re-does nothing: what still needs generating is re-derived from disk every time, which also means a transcript that finishes mid-sweep, or a duplicate cluster you confirm, is simply picked up on the next pass. A separate **Pause Digests** control holds a running sweep at zero without ending it (the pause flag has existed since the digest layer shipped and nothing could set it). **The digest lane now yields the GPU to transcription**: the two were deliberately on separate queues so they wouldn't serialise, whose unintended consequence was that the model and whisper competed for the same card — measured at 90 seconds per audio-hour against the 27 an idle machine managed. While transcription is working the digest lane steps aside and resumes when the card is free; it can be turned off in Settings. Coverage is now visible — a digest instrument on the dashboard with the corpus percentage (to two decimals, because rounding 0.13% up to 1% flatters an 80-day job), a **No digest** column on the channels table, and a per-channel count on the needs-work rows. Progress bars also **work during a regeneration** for the first time: they re-counted digest files from disk, and a regenerated digest is rewritten in place, so a job that was working sat at 0% for its whole run. Time-remaining estimates for digest work are now computed in **seconds per audio-hour** rather than by averaging videos, which for this archive is wrong by more than an order of magnitude between a VOD channel and a shorts channel. New `common/bin/digest-plan.ts` prices the whole backfill in audio-hours before you commit hardware to it.
diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts
@@ -2,7 +2,7 @@
import { revalidatePath } from "next/cache";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import {
excludedDownloadIdSet,
generateChannelSnapshot,
@@ -36,7 +36,7 @@ export type RunDuplicateDetectionResult =
export async function refreshAllChannelSnapshotsAction(): Promise<RefreshAllResult> {
const paths = getPaths();
- const channels = await listChannels(paths);
+ const channels = await listChannelConfigs(paths);
const active = new Set(
getRegistry()
.list()
diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts
@@ -1,16 +1,16 @@
import type { Paths } from "yt-dlp-transcript-common/lib/paths";
+import { cache } from "react";
+import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels";
import {
- listChannels,
- type ChannelStat,
-} from "yt-dlp-transcript-common/controller/channels";
+ getChannelBriefs,
+ getDuplicateReport,
+} from "../../lib/requestCache";
import {
excludedDownloadIdSet,
- readChannelSnapshot,
type ChannelSnapshot,
} from "yt-dlp-transcript-common/controller/channelSnapshot";
import {
readDuplicateOverrides,
- readDuplicateReport,
} from "yt-dlp-transcript-common/controller/duplicateShorts";
import type {
DuplicateOverrides,
@@ -18,7 +18,9 @@ import type {
} from "yt-dlp-transcript-common/lib/duplicates";
export type ActionableRow = {
- channel: ChannelStat;
+ channel: ChannelBrief;
+ // Always `channel.snapshot` — kept as a sibling field because every count
+ // helper and every consumer already reads `row.snapshot`.
snapshot: ChannelSnapshot | null;
};
@@ -124,17 +126,17 @@ export function actionableCleanExtraFormatsBytes(row: ActionableRow): number {
export async function loadActionableSummary(
paths: Paths,
): Promise<ActionableSummary> {
- const channels = await listChannels(paths);
- const [rows, duplicates, duplicateOverrides] = await Promise.all([
- Promise.all(
- channels.map(async (channel) => ({
- channel,
- snapshot: await readChannelSnapshot(paths, channel.slug),
- })),
- ),
- readDuplicateReport(paths),
+ const [channels, duplicates, duplicateOverrides] = await Promise.all([
+ getChannelBriefs(paths),
+ getDuplicateReport(paths),
readDuplicateOverrides(paths),
]);
+ // The brief already read the snapshot; this used to read each one a second
+ // time on top of a full corpus walk.
+ const rows: ActionableRow[] = channels.map((channel) => ({
+ channel,
+ snapshot: channel.snapshot,
+ }));
const undownloaded = rows
.filter((r) => actionableUndownloadedCount(r) > 0)
@@ -200,3 +202,10 @@ export async function loadActionableSummary(
duplicateOverrides,
};
}
+
+// Per-request memoized. The dashboard renders this summary and several
+// components derived from it in one pass; see ../../lib/requestCache for why a
+// request is the only cache lifetime this data can safely have.
+export const getActionableSummary = cache((paths: Paths) =>
+ loadActionableSummary(paths),
+);
diff --git a/editor/app/api/widget/sync/route.ts b/editor/app/api/widget/sync/route.ts
@@ -1,6 +1,7 @@
import { NextResponse } from "next/server";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { digestCountOf } from "yt-dlp-transcript-common/controller/channels";
+import { getChannelBriefs } from "../../../lib/requestCache";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { buildScheduleView } from "yt-dlp-transcript-common/jobs/syncScheduler";
import { readSchedulerState } from "yt-dlp-transcript-common/jobs/syncSchedulerState";
@@ -46,7 +47,7 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> {
const settings = getSettings();
const state = await readSchedulerState(paths);
const now = Date.now();
- const channels = await listChannels(paths);
+ const channels = await getChannelBriefs(paths);
let lastIndividualSyncAt: number | null = null;
for (const c of channels) {
@@ -75,14 +76,16 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> {
}
}
- // Free: listChannels already counted these, it just used to throw the digest
- // half away (see controller/channels.ts).
+ // Free: the snapshot each brief already carries has both halves. This used to
+ // come off a full corpus walk — a 4.4 s cost on an endpoint the widget polls.
+ // `digestEngines` postdates 11 of the live snapshots, so those read as 0 until
+ // their next report refresh (see digestCountOf).
let digested = 0;
let videos = 0;
let channelsWithAny = 0;
for (const c of channels) {
- videos += c.videoCount;
- const n = c.digestCount ?? 0;
+ videos += c.snapshot?.totals.videos ?? 0;
+ const n = digestCountOf(c.snapshot);
digested += n;
if (n > 0) channelsWithAny++;
}
diff --git a/editor/app/auto-queue/page.tsx b/editor/app/auto-queue/page.tsx
@@ -1,7 +1,7 @@
import type { Metadata } from "next";
import Link from "next/link";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import { PLATFORM_VALUES } from "yt-dlp-transcript-common/lib/platform";
import { selectableBucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy";
import { buildAutoQueueStatusPayload } from "./status";
@@ -23,7 +23,7 @@ const BUCKETS_BY_KIND = {
export default async function AutoQueuePage() {
const [initial, channels] = await Promise.all([
buildAutoQueueStatusPayload(),
- listChannels(getPaths()),
+ listChannelConfigs(getPaths()),
]);
const channelOptions = channels.map((c) => ({
slug: c.slug,
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -16,7 +16,7 @@ import {
createChannel,
deleteChannel,
isValidChannelSlug,
- listChannels,
+ listChannelConfigs,
readChannelConfig,
writeChannelConfig,
} from "yt-dlp-transcript-common/controller/channels";
@@ -364,7 +364,7 @@ export type SyncAllResult = {
export async function syncAllChannelsAction(): Promise<SyncAllResult> {
const paths = getPaths();
- const channels = await listChannels(paths);
+ const channels = await listChannelConfigs(paths);
const active = activeSyncSlugs();
const queued: string[] = [];
const skipped: { slug: string; reason: string }[] = [];
diff --git a/editor/app/channels/page.tsx b/editor/app/channels/page.tsx
@@ -1,6 +1,10 @@
import type { Metadata } from "next";
import Link from "next/link";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import {
+ listChannelBriefs,
+ listChannelStatsFromSnapshots,
+ type ChannelBrief,
+} from "yt-dlp-transcript-common/controller/channels";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
getSite,
@@ -15,6 +19,33 @@ export const dynamic = "force-dynamic";
export const metadata: Metadata = { title: "Channels" };
+// The video/transcript/download counts in this table are projected from each
+// channel's last generated snapshot rather than counted off disk, which is what
+// makes the page load in milliseconds instead of seconds. That trade is only
+// honest if the page says how old the numbers are — so report the OLDEST
+// snapshot on screen, and name the channels that have never had one.
+function summariseFreshness(
+ briefs: ReadonlyArray<ChannelBrief>,
+): { oldest: string | null; missing: string[] } {
+ let oldest: number | null = null;
+ let oldestIso: string | null = null;
+ const missing: string[] = [];
+ for (const b of briefs) {
+ const at = b.snapshot?.generatedAt;
+ if (!at) {
+ missing.push(b.slug);
+ continue;
+ }
+ const ms = Date.parse(at);
+ if (Number.isNaN(ms)) continue;
+ if (oldest === null || ms < oldest) {
+ oldest = ms;
+ oldestIso = at;
+ }
+ }
+ return { oldest: oldestIso, missing };
+}
+
export default async function ChannelsPage({
searchParams,
}: {
@@ -23,7 +54,10 @@ export default async function ChannelsPage({
const paths = getPaths();
const { site } = await searchParams;
const active = resolveActiveSite(site, listSiteIds(paths));
- const all = await listChannels(paths);
+ // Counts come from each channel's last snapshot, not a corpus walk. One read
+ // serves both the table and the freshness footer below.
+ const briefs = await listChannelBriefs(paths);
+ const all = await listChannelStatsFromSnapshots(paths, briefs);
// Scope to the active site's membership; "all sites" shows the full pool.
const channels =
active.isAll || !active.siteId
@@ -32,6 +66,10 @@ export default async function ChannelsPage({
const slugs = siteChannelSlugs(getSite(active.siteId, paths));
return all.filter((c) => slugs.has(c.slug));
})();
+ const shown = new Set(channels.map((c) => c.slug));
+ const freshness = summariseFreshness(
+ briefs.filter((b) => shown.has(b.slug)),
+ );
return (
<div className="flex flex-col gap-4">
<div className="flex items-center justify-between">
@@ -53,7 +91,37 @@ export default async function ChannelsPage({
: "No channels yet."}
</p>
) : (
- <ChannelsTable channels={channels} />
+ <>
+ <ChannelsTable channels={channels} />
+ <p
+ className="text-xs text-muted-foreground"
+ data-testid="channels-freshness"
+ >
+ {freshness.oldest ? (
+ <>
+ Video, transcript and download counts come from each channel’s
+ last report; the oldest on this page was generated{" "}
+ <time dateTime={freshness.oldest}>
+ {new Date(freshness.oldest).toLocaleString()}
+ </time>
+ . A just-finished job can take a moment to show up here.
+ </>
+ ) : (
+ <>
+ No channel on this page has a generated report yet, so every count
+ reads zero. Run <em>Refresh report</em> from a channel to populate
+ them.
+ </>
+ )}
+ {freshness.missing.length > 0 && freshness.oldest && (
+ <>
+ {" "}
+ No report yet for {freshness.missing.join(", ")} — those rows
+ read zero.
+ </>
+ )}
+ </p>
+ </>
)}
</div>
);
diff --git a/editor/app/cleanup/lib/loadCleanup.ts b/editor/app/cleanup/lib/loadCleanup.ts
@@ -1,12 +1,7 @@
import type { Paths } from "yt-dlp-transcript-common/lib/paths";
-import {
- listChannels,
- type ChannelStat,
-} from "yt-dlp-transcript-common/controller/channels";
-import {
- readChannelSnapshot,
- type ChannelSnapshot,
-} from "yt-dlp-transcript-common/controller/channelSnapshot";
+import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels";
+import { getChannelBriefs } from "../../lib/requestCache";
+import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot";
import { loadFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions";
import { loadFailedTranscodings } from "yt-dlp-transcript-common/controller/failedTranscodings";
@@ -16,7 +11,7 @@ import { loadFailedTranscodings } from "yt-dlp-transcript-common/controller/fail
// sweeps, not additive — so the headline total uses `transcribedBytes` (the
// primary, safe "Clean audio for transcribed videos" sweep) only.
export type CleanupRow = {
- channel: ChannelStat;
+ channel: ChannelBrief;
snapshot: ChannelSnapshot | null;
included: boolean;
transcribedBytes: number;
@@ -50,7 +45,7 @@ function transcribedBytesOf(snap: ChannelSnapshot | null): number {
}
function rowOf(
- channel: ChannelStat,
+ channel: ChannelBrief,
snapshot: ChannelSnapshot | null,
failedTranscriptions: number,
failedTranscodings: number,
@@ -71,15 +66,16 @@ function rowOf(
export async function loadCleanupSummary(
paths: Paths,
): Promise<CleanupSummary> {
- const channels = await listChannels(paths);
+ const channels = await getChannelBriefs(paths);
const rows = await Promise.all(
channels.map(async (channel) => {
- const [snapshot, failedT, failedX] = await Promise.all([
- readChannelSnapshot(paths, channel.slug),
+ // The brief already carries the snapshot; only the two failed-lists are
+ // still per-channel reads, and they run concurrently.
+ const [failedT, failedX] = await Promise.all([
loadFailedTranscriptions(paths, channel.slug),
loadFailedTranscodings(paths, channel.slug),
]);
- return rowOf(channel, snapshot, failedT.length, failedX.length);
+ return rowOf(channel, channel.snapshot, failedT.length, failedX.length);
}),
);
@@ -132,22 +128,25 @@ export type CleanableChannelRow = {
// total. Reads only the small snapshot JSONs (unlike loadCleanupSummary, which
// also reads both failed-lists per channel), so it's cheap on the AutoRefresh
// cadence / per widget poll. Sorted by reclaim, descending.
+//
+// The comment above was true of the snapshot reads and false of the line that
+// fetched the channel list: it used the corpus walk, so describing ~98 videos
+// cost ~474,559 file touches — on every render of the root layout, every
+// widget poll, and every auto-refresh tick. It reads the 65 briefs now.
export async function cleanableChannels(
paths: Paths,
): Promise<CleanableChannelRow[]> {
- const channels = await listChannels(paths);
- const rows = await Promise.all(
- channels.map(async (channel) => {
- if (channel.config.excludeFromCleanup) return null;
- const snap = await readChannelSnapshot(paths, channel.slug);
- const bytes = transcribedBytesOf(snap);
- const count = snap?.buckets.transcribedWithAudio?.length ?? 0;
- // Nothing to reclaim and nothing to sweep — drop the row. Zero-byte rows
- // contribute nothing to the total, so dropping them keeps it unchanged.
- if (bytes <= 0 && count <= 0) return null;
- return { slug: channel.slug, count, bytes };
- }),
- );
+ const channels = await getChannelBriefs(paths);
+ const rows = channels.map((channel) => {
+ if (channel.config.excludeFromCleanup) return null;
+ const snap = channel.snapshot;
+ const bytes = transcribedBytesOf(snap);
+ const count = snap?.buckets.transcribedWithAudio?.length ?? 0;
+ // Nothing to reclaim and nothing to sweep — drop the row. Zero-byte rows
+ // contribute nothing to the total, so dropping them keeps it unchanged.
+ if (bytes <= 0 && count <= 0) return null;
+ return { slug: channel.slug, count, bytes };
+ });
return rows
.filter((r): r is CleanableChannelRow => r !== null)
.sort((a, b) => b.bytes - a.bytes);
diff --git a/editor/app/lib/requestCache.ts b/editor/app/lib/requestCache.ts
@@ -0,0 +1,42 @@
+import { cache } from "react";
+import type { Paths } from "yt-dlp-transcript-common/lib/paths";
+import { listChannelBriefs } from "yt-dlp-transcript-common/controller/channels";
+import { readDuplicateReport } from "yt-dlp-transcript-common/controller/duplicateShorts";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+
+// Per-REQUEST memoization for loaders that more than one component on the same
+// page reaches for. The dashboard derives the channel list three separate ways
+// in a single render (the layout's reclaimable-disk badge, loadActionableSummary
+// and buildWidgetSyncPayload); without this it reads the same 65 files each
+// time.
+//
+// `React.cache` is scoped to one request, so there is no staleness risk here by
+// construction — the request boundary is already the consistency boundary, and
+// two components rendering the same page should see the same numbers anyway.
+//
+// This lives in editor/app rather than common/ deliberately, on both counts:
+// common/ is shared with the export, MCP and CLI packages and must not import
+// react. And no cache with a LONGER life than a request belongs on this data —
+// no `unstable_cache`, no `use cache`, no module-level map. The snapshot
+// scheduler rewrites these files on a ~1 s debounce from job runners inside
+// common/, which cannot import next/cache to invalidate anything. A cache
+// nothing can invalidate is just a stale number with extra steps.
+// Keyed on the `paths` argument by identity, which works because getPaths()
+// memoizes its result at module scope and hands back the same object every
+// call. Pass it straight through; don't spread or rebuild it.
+//
+// (loadActionableSummary gets the same treatment, but its cached wrapper lives
+// beside it in ../actionable/lib/loadActionable — this module deliberately
+// imports nothing from app/ so those loaders can import IT.)
+export const getChannelBriefs = cache((paths: Paths) =>
+ listChannelBriefs(paths),
+);
+
+// The duplicates report is a 6.7 MB JSON parse. It measures at ~65 ms, so this
+// is tidiness rather than a headline win — but /actionable and the dashboard
+// both want it in one render.
+export const getDuplicateReport = cache((paths: Paths) =>
+ readDuplicateReport(paths),
+);
+
+export const getSettingsCached = cache(() => getSettings());
diff --git a/editor/app/page.tsx b/editor/app/page.tsx
@@ -11,7 +11,7 @@ import {
actionableNoDigestCount,
actionableUndownloadedCount,
actionableUntranscribedCount,
- loadActionableSummary,
+ getActionableSummary,
} from "./actionable/lib/loadActionable";
import { resolveActiveSite } from "./lib/activeSite";
import { buildActiveJobsPayload } from "./jobs/active/buildActiveJobs";
@@ -54,7 +54,7 @@ export default async function Dashboard({
const paths = getPaths();
const { site } = await searchParams;
const active = resolveActiveSite(site, listSiteIds(paths));
- const summary = await loadActionableSummary(paths);
+ const summary = await getActionableSummary(paths);
// Scope the channel-derived data to the active site's membership. Under "all
// sites" the full pool is shown. Job/worker state stays global (pool state).
@@ -70,7 +70,9 @@ export default async function Dashboard({
const channels: DashboardChannel[] = rows.map((r) => ({
slug: r.channel.slug,
handling: r.channel.config.handling,
- videoCount: r.channel.videoCount,
+ // From the snapshot rather than a corpus walk, so it lags a sync by the
+ // snapshot scheduler's debounce.
+ videoCount: r.snapshot?.totals.videos ?? 0,
lastSyncedAt: toMs(r.channel.config.lastSyncedAt),
hasUrl: Boolean(r.channel.config.url),
undownloaded: actionableUndownloadedCount(r),
diff --git a/editor/app/scheduler/runTick.ts b/editor/app/scheduler/runTick.ts
@@ -1,5 +1,5 @@
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { activeSyncJobs } from "yt-dlp-transcript-common/jobs/syncJobs";
@@ -101,7 +101,7 @@ export async function runSchedulerTick(): Promise<SchedulerTickResult> {
};
}
- const channels: ChannelEntry[] = (await listChannels(paths)).map((c) => ({
+ const channels: ChannelEntry[] = (await listChannelConfigs(paths)).map((c) => ({
slug: c.slug,
config: c.config,
}));
diff --git a/editor/app/scheduler/status.ts b/editor/app/scheduler/status.ts
@@ -1,5 +1,5 @@
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import {
getSettings,
type SyncSchedulerSettings,
@@ -32,7 +32,7 @@ export async function buildSchedulerStatusPayload(): Promise<SchedulerStatusPayl
const settings = getSettings();
const state = await readSchedulerState(paths);
const now = Date.now();
- const channels = (await listChannels(paths)).map((c) => ({
+ const channels = (await listChannelConfigs(paths)).map((c) => ({
slug: c.slug,
config: c.config,
}));
diff --git a/editor/app/sites/[siteId]/page.tsx b/editor/app/sites/[siteId]/page.tsx
@@ -2,7 +2,7 @@ import type { Metadata } from "next";
import Link from "next/link";
import { notFound } from "next/navigation";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import {
getSite,
isValidSiteId,
@@ -37,7 +37,7 @@ export default async function EditSitePage({
notFound();
}
const site = getSite(siteId, paths);
- const channels: ChannelOption[] = (await listChannels(paths)).map((c) => ({
+ const channels: ChannelOption[] = (await listChannelConfigs(paths)).map((c) => ({
slug: c.slug,
name: c.config.name ?? c.slug,
}));
diff --git a/editor/app/sites/new/page.tsx b/editor/app/sites/new/page.tsx
@@ -1,7 +1,7 @@
import type { Metadata } from "next";
import Link from "next/link";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { listChannelConfigs } from "yt-dlp-transcript-common/controller/channels";
import {
DEFAULT_GROUP_FALLBACK_ID,
FALLBACK_GROUP,
@@ -18,7 +18,7 @@ export const metadata: Metadata = { title: "New site — Sites" };
export default async function NewSitePage() {
const paths = getPaths();
- const channels: ChannelOption[] = (await listChannels(paths)).map((c) => ({
+ const channels: ChannelOption[] = (await listChannelConfigs(paths)).map((c) => ({
slug: c.slug,
name: c.config.name ?? c.slug,
}));