import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import { digestCountOf, listChannelBriefs, listChannelConfigs, listChannelStatsFromDisk, listChannelStatsFromSnapshots, } from "./channels"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/channelProjection.test.ts // // listChannelStatsFromSnapshots replaces a full corpus walk on every render // path in the editor. It is only a legitimate substitution if it agrees with // the walk when the snapshots are current — so that agreement is asserted here, // field by field, rather than assumed. type VideoSpec = { id: string; // A whisper transcript (transcript.json) counts as transcribed AND downloaded. transcribed?: boolean; // A finalized audio.* file counts as downloaded on its own. audio?: boolean; // A non-empty ai-digest.json — only meaningful on a transcribed video. digest?: boolean; }; async function seedChannel( paths: Paths, slug: string, videos: ReadonlyArray, opts: { playlist?: number; digestEngines?: Record | null } = {}, ): Promise { const channelDir = path.join(paths.channelsDir, slug); await mkdir(channelDir, { recursive: true }); // `handling` is required — parseChannelConfig returns null without it, and a // channel whose config won't parse is invisible to every reader here. await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ handling: "transcribe", url: `https://example.test/${slug}`, name: slug, }), ); if (opts.playlist !== undefined) { await writeFile( path.join(channelDir, "playlist"), Array.from({ length: opts.playlist }, (_, i) => `v${i}`).join("\n") + "\n", ); } for (const v of videos) { const dir = path.join(channelDir, "data", v.id); await mkdir(dir, { recursive: true }); await writeFile(path.join(dir, "metadata.info.json"), JSON.stringify({ id: v.id })); if (v.transcribed) { await writeFile(path.join(dir, "transcript.json"), JSON.stringify({ cues: [] })); } if (v.audio) await writeFile(path.join(dir, "audio.mp3"), "x"); if (v.digest) { // digestSchemaVersion is required (loadDigest rejects the record without // it), and the channel count only counts a record with a non-empty // section. await writeFile( path.join(dir, "ai-digest.json"), JSON.stringify({ digestSchemaVersion: 1, sections: { tags: { items: ["a"] } }, }), ); } } // A CURRENT snapshot — exactly what the scheduler would have written for the // dirs seeded above. `digestEngines: null` seeds the older shape that predates // the field, which 11 of the 65 live snapshots still have. const transcribed = videos.filter((v) => v.transcribed).length; const downloaded = videos.filter((v) => v.transcribed || v.audio).length; const digests = videos.filter((v) => v.transcribed && v.digest).length; const snapshot: Record = { generatedAt: new Date().toISOString(), totals: { videos: videos.length, transcribed, downloaded }, buckets: {}, undownloadedIds: [], }; if (opts.digestEngines !== null) { snapshot.digestEngines = opts.digestEngines ?? { local: digests }; } await writeFile( path.join(channelDir, "snapshot.json"), JSON.stringify(snapshot), ); } async function withPaths(fn: (paths: Paths) => Promise): Promise { const dir = await mkdtemp(path.join(tmpdir(), "ttb-projection-")); const paths = { channelsDir: path.join(dir, "channels") } as Paths; try { await fn(paths); } finally { await rm(dir, { recursive: true, force: true }); } } test("snapshot projection matches the disk walk when snapshots are current", async () => { await withPaths(async (paths) => { await seedChannel( paths, "alpha", [ { id: "a1", transcribed: true, audio: true, digest: true }, { id: "a2", transcribed: true, audio: true }, { id: "a3", audio: true }, { id: "a4" }, ], { playlist: 6 }, ); await seedChannel(paths, "beta", [{ id: "b1", transcribed: true }], { playlist: 1, }); // A channel with nothing in it at all. await seedChannel(paths, "gamma", []); const fromDisk = await listChannelStatsFromDisk(paths); const fromSnapshots = await listChannelStatsFromSnapshots(paths); // Without this the comparison below passes vacuously when the fixture fails // to seed — which is exactly how this test first went green while reading // nothing at all. assert.equal(fromDisk.length, 3, "the fixture corpus actually seeded"); assert.deepEqual( fromSnapshots.map((c) => c.slug), fromDisk.map((c) => c.slug), "both readers list the same channels in the same order", ); for (const [i, snap] of fromSnapshots.entries()) { const disk = fromDisk[i]; for (const field of [ "videoCount", "transcriptCount", "downloadCount", "digestCount", "playlistCount", ] as const) { assert.equal( snap[field], disk[field], `${snap.slug}.${field}: projection ${snap[field]} !== walk ${disk[field]}`, ); } } }); }); test("a snapshot without digestEngines reports zero digests, not a wrong number", async () => { await withPaths(async (paths) => { // Two videos DO carry digests on disk, but this channel's snapshot predates // the digestEngines field. The projection must under-report to 0 rather than // invent a count — and must not throw. await seedChannel( paths, "legacy", [ { id: "l1", transcribed: true, digest: true }, { id: "l2", transcribed: true, digest: true }, ], { digestEngines: null }, ); const [projected] = await listChannelStatsFromSnapshots(paths); const [walked] = await listChannelStatsFromDisk(paths); assert.equal(walked.digestCount, 2, "the walk sees both digests"); assert.equal(projected.digestCount, 0, "the projection defaults to 0"); // Every other field still agrees exactly. assert.equal(projected.videoCount, walked.videoCount); assert.equal(projected.transcriptCount, walked.transcriptCount); assert.equal(projected.downloadCount, walked.downloadCount); }); }); test("digestCountOf sums across engines and tolerates the missing field", () => { assert.equal(digestCountOf(null), 0); assert.equal(digestCountOf({ totals: {} } as never), 0); assert.equal( digestCountOf({ digestEngines: { local: 3, cloud: 4 } } as never), 7, ); }); test("briefs and configs agree with the walk on membership", async () => { await withPaths(async (paths) => { await seedChannel(paths, "one", [{ id: "x", transcribed: true }]); await seedChannel(paths, "two", []); // A directory with no config.json is not a channel and must be skipped by // every reader, not just the walk. await mkdir(path.join(paths.channelsDir, "not-a-channel", "data"), { recursive: true, }); const [briefs, configs, walked] = await Promise.all([ listChannelBriefs(paths), listChannelConfigs(paths), listChannelStatsFromDisk(paths), ]); const slugs = walked.map((c) => c.slug); assert.deepEqual(slugs, ["one", "two"]); assert.deepEqual(briefs.map((b) => b.slug), slugs); assert.deepEqual(configs.map((c) => c.slug), slugs); assert.ok( briefs.every((b) => b.snapshot !== null), "each seeded channel's snapshot was read by the brief", ); }); }); test("readers return empty rather than throwing when there is no corpus", async () => { await withPaths(async (paths) => { assert.deepEqual(await listChannelBriefs(paths), []); assert.deepEqual(await listChannelConfigs(paths), []); assert.deepEqual(await listChannelStatsFromSnapshots(paths), []); assert.deepEqual(await listChannelStatsFromDisk(paths), []); }); });