import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, readFile, rm, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import type { Paths } from "../lib/paths"; import { emptyRoster, loadRoster, mergeRoster, observationsFromUrls, recordSweep, rosterPath, seedRosterIfAbsent, writeRoster, } from "./rosterStore"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/rosterStore.test.ts const T1 = "2026-08-01T00:00:00.000Z"; const T2 = "2026-08-02T00:00:00.000Z"; function yt(id: string): string { return `https://www.youtube.com/watch?v=${id}`; } async function withPaths(fn: (paths: Paths) => Promise): Promise { const dir = await mkdtemp(path.join(tmpdir(), "ttb-roster-")); const paths = { channelsDir: path.join(dir, "channels") } as Paths; try { await fn(paths); } finally { await rm(dir, { recursive: true, force: true }); } } // --- mergeRoster: additive, and only additive ------------------------------- test("mergeRoster never drops an entry, however the listing shrinks", () => { const full = mergeRoster( emptyRoster(), ["a", "b", "c"].map((id) => ({ id, url: yt(id) })), T1, "listing", ); // The failure this file exists to prevent: an enumeration that returns one of // three entries. const after = mergeRoster(full, [{ id: "a", url: yt("a") }], T2, "listing"); assert.deepEqual(Object.keys(after.entries).sort(), ["a", "b", "c"]); assert.equal(after.entries.b.url, yt("b")); }); test("firstSeenAt and source are stable across merges while lastListedAt advances", () => { const first = mergeRoster( emptyRoster(), [{ id: "a", url: yt("a") }], T1, "listing", ); assert.equal(first.entries.a.firstSeenAt, T1); assert.equal(first.entries.a.lastListedAt, T1); const second = mergeRoster(first, [{ id: "a", url: yt("a") }], T2, "listing"); assert.equal(second.entries.a.firstSeenAt, T1, "firstSeenAt must not move"); assert.equal(second.entries.a.lastListedAt, T2); assert.equal(second.entries.a.source, "listing"); }); test("a non-listing observation does not advance lastListedAt", () => { const first = mergeRoster( emptyRoster(), [{ id: "a", url: yt("a") }], T1, "listing", ); const imported = mergeRoster(first, [{ id: "a", url: yt("a") }], T2, "import"); assert.equal( imported.entries.a.lastListedAt, T1, "an import is not an enumeration", ); }); test("a listing URL overwrites a stored one; an import only fills a gap", () => { const seeded = mergeRoster(emptyRoster(), [{ id: "a", url: "" }], T1, "disk"); assert.equal(seeded.entries.a.url, ""); assert.equal(seeded.entries.a.source, "disk"); const filled = mergeRoster(seeded, [{ id: "a", url: yt("a") }], T1, "import"); assert.equal(filled.entries.a.url, yt("a"), "an empty URL is a gap to fill"); assert.equal(filled.entries.a.source, "disk", "source records first sighting"); const stale = mergeRoster( filled, [{ id: "a", url: "https://example.com/other" }], T2, "import", ); assert.equal(stale.entries.a.url, yt("a"), "an import must not overwrite"); const relisted = mergeRoster( filled, [{ id: "a", url: "https://www.youtube.com/watch?v=a&t=1" }], T2, "listing", ); assert.equal(relisted.entries.a.url, "https://www.youtube.com/watch?v=a&t=1"); }); test("mergeRoster is pure and returns the same object when nothing changed", () => { const before = mergeRoster( emptyRoster(), [{ id: "a", url: yt("a") }], T1, "listing", ); const snapshot = JSON.stringify(before); const same = mergeRoster(before, [{ id: "a", url: yt("a") }], T1, "listing"); assert.equal(same, before, "no change => same reference, so callers can skip"); assert.equal(JSON.stringify(before), snapshot, "input must not be mutated"); }); test("observationsFromUrls canonicalizes and drops what it cannot", () => { const observed = observationsFromUrls([ yt("abc"), "https://rumble.com/v123abc-some-title.html", "not a url", ]); assert.deepEqual( observed.map((o) => o.id), ["abc", "v123abc"], ); }); // --- the file --------------------------------------------------------------- test("a corrupt roster file reads as empty rather than throwing", async () => { await withPaths(async (paths) => { await mkdir(path.join(paths.channelsDir, "ch"), { recursive: true }); await writeFile(rosterPath(paths, "ch"), "{ not json"); assert.deepEqual(await loadRoster(paths, "ch"), emptyRoster()); }); }); test("a missing roster file reads as empty", async () => { await withPaths(async (paths) => { assert.deepEqual(await loadRoster(paths, "nope"), emptyRoster()); }); }); test("ill-typed entries are dropped and a half-written sweep is discarded", async () => { await withPaths(async (paths) => { await mkdir(path.join(paths.channelsDir, "ch"), { recursive: true }); await writeFile( rosterPath(paths, "ch"), JSON.stringify({ version: 1, entries: { good: { url: yt("good"), firstSeenAt: T1, lastListedAt: T1, source: "listing" }, bad: 42, partial: { firstSeenAt: T1 }, }, lastSweep: { at: T1, listedCount: "lots", verdict: "ok" }, }), ); const roster = await loadRoster(paths, "ch"); assert.deepEqual(Object.keys(roster.entries).sort(), ["good", "partial"]); assert.equal(roster.entries.partial.url, ""); assert.equal(roster.entries.partial.lastListedAt, T1); assert.equal(roster.lastSweep, null); }); }); test("a written roster round-trips, sweep verdict and all", async () => { await withPaths(async (paths) => { const roster = recordSweep( mergeRoster(emptyRoster(), [{ id: "a", url: yt("a") }], T1, "listing"), { at: T1, listedCount: 1, verdict: "shrink-suspect" }, ); await writeRoster(paths, "ch", roster); assert.deepEqual(await loadRoster(paths, "ch"), roster); // No tmp file left behind by the atomic write. const raw = await readFile(rosterPath(paths, "ch"), "utf8"); assert.match(raw, /\n$/); }); }); // --- seeding ---------------------------------------------------------------- async function seedFixture( paths: Paths, slug: string, opts: { playlist?: ReadonlyArray; dirs?: ReadonlyArray<{ id: string; availabilityUrl?: string; metadataUrl?: string }>; }, ): Promise { const channelDir = path.join(paths.channelsDir, slug); await mkdir(channelDir, { recursive: true }); if (opts.playlist) { await writeFile( path.join(channelDir, "playlist"), opts.playlist.join("\n") + "\n", ); } for (const d of opts.dirs ?? []) { const dir = path.join(channelDir, "data", d.id); await mkdir(dir, { recursive: true }); if (d.availabilityUrl) { await writeFile( path.join(dir, "availability.json"), JSON.stringify({ checkedAt: T1, availability: "public", webpageUrl: d.availabilityUrl, }), ); } if (d.metadataUrl) { await writeFile( path.join(dir, "metadata.info.json"), JSON.stringify({ id: d.id, formats: [], webpage_url: d.metadataUrl }), ); } } } test("seeding a channel with a playlist but no data/ preserves every URL", async () => { await withPaths(async (paths) => { // The exposure the roster closes: listed, never downloaded, so no dir and // no metadata.info.json — `playlist` is the ONLY place these URLs live. await seedFixture(paths, "ch", { playlist: [yt("a"), yt("b")] }); const roster = await seedRosterIfAbsent(paths, "ch", { now: T1 }); assert.deepEqual(Object.keys(roster.entries).sort(), ["a", "b"]); assert.equal(roster.entries.a.url, yt("a")); assert.equal(roster.entries.b.source, "listing"); // And it is on disk, so the next sweep cannot erase it. assert.deepEqual(await loadRoster(paths, "ch"), roster); }); }); test("seeding unions data/ dirs the playlist does not mention", async () => { await withPaths(async (paths) => { // The live corpus has a channel with 617 videos and no playlist file at // all; this is that case. await seedFixture(paths, "ch", { dirs: [ { id: "a", availabilityUrl: yt("a") }, { id: "b", metadataUrl: yt("b") }, { id: "c" }, ], }); const roster = await seedRosterIfAbsent(paths, "ch", { now: T1 }); assert.deepEqual(Object.keys(roster.entries).sort(), ["a", "b", "c"]); assert.equal(roster.entries.a.url, yt("a"), "from availability.json"); assert.equal(roster.entries.b.url, yt("b"), "from metadata.info.json"); assert.equal(roster.entries.c.url, "", "no URL recoverable, but recorded"); for (const id of ["a", "b", "c"]) { assert.equal(roster.entries[id].source, "disk"); } }); }); test("a listed video keeps its listing URL even when it also has a dir", async () => { await withPaths(async (paths) => { await seedFixture(paths, "ch", { playlist: [yt("a")], dirs: [{ id: "a", availabilityUrl: "https://example.com/stale" }], }); const roster = await seedRosterIfAbsent(paths, "ch", { now: T1 }); assert.equal(roster.entries.a.url, yt("a")); assert.equal(roster.entries.a.source, "listing"); }); }); test("seeding is a one-off: an existing roster is returned untouched", async () => { await withPaths(async (paths) => { await seedFixture(paths, "ch", { playlist: [yt("a")] }); const first = await seedRosterIfAbsent(paths, "ch", { now: T1 }); // Something left the listing after the seed. Re-seeding must not drop it. await writeFile( path.join(paths.channelsDir, "ch", "playlist"), yt("z") + "\n", ); const second = await seedRosterIfAbsent(paths, "ch", { now: T2 }); assert.deepEqual(second, first); }); }); test("seeding a channel with nothing at all writes no file", async () => { await withPaths(async (paths) => { await mkdir(path.join(paths.channelsDir, "ch"), { recursive: true }); const roster = await seedRosterIfAbsent(paths, "ch", { now: T1 }); assert.deepEqual(roster, emptyRoster()); await assert.rejects(() => readFile(rosterPath(paths, "ch"), "utf8")); }); });