import { readFile, rm, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { channelStage, generateReport, readJson, resetData, resolvePath, writeSettings, } from "./helpers"; // The sync FULL SWEEP: one full listing enumeration that refreshes the stored // `playlist`, flags videos that have left the listing into maybe-missing.json, // and (under a cap) resolves those suspects with the per-video availability // probe — all inside the ordinary `sync` job. // // The sweep is OFF in the e2e default settings (fullSweepIntervalMinutes: 0), // the same way verifyAvailabilityBeforeClean is, so every pre-existing sync // spec keeps exercising the cheap paged walk. Specs that want the sweep opt in, // as this one does. const CHANNEL = "availability-test"; const FIXTURE = "availability-baseline"; const channelRoot = `test-transcripts/channels/${CHANNEL}`; // The baseline fixture's on-disk video dirs (canonical ids). const ALL_IDS = [ "vidpublic1", "vidpublic2", "viddeleted1", "vidprivate1", "vidmembers1", "vidneedsauth1", ]; type MaybeMissing = { checkedAt: string; freshPlaylistCount: number; ids: string[]; }; type ChannelConfigFile = { lastSyncedAt?: string; lastFullSweepAt?: string; }; type RosterFile = { version: number; lastSweep: { at: string; listedCount: number; verdict: string } | null; entries: Record< string, { url: string; firstSeenAt: string; lastListedAt: string; source: string } >; }; type SnapshotFile = { missingNeverFetched?: { id: string; url: string; firstSeenAt: string }[]; }; // The shrink guard has an absolute floor of 25 entries below which no drop is // suspicious, so exercising it needs a channel bigger than the 6-video fixture. // These ids never get directories — the archive is seeded with them so the // sweep's download walk treats them as already fetched. const BULK_IDS = Array.from( { length: 60 }, (_, i) => `bulk${String(i + 1).padStart(8, "0")}`, ); function ytUrl(id: string): string { return `https://www.youtube.com/watch?v=${id}`; } async function readRoster(): Promise { return readJson(`${channelRoot}/roster.json`); } async function writePlaylist(ids: string[]): Promise { await writeFile( resolvePath(`${channelRoot}/playlist`), ids.map(ytUrl).join("\n") + "\n", ); } // generateReport short-circuits when a snapshot already exists, so a spec that // needs the report REDONE after a mutation has to clear it first. async function regenerateReport( page: import("@playwright/test").Page, ): Promise { await rm(resolvePath(`${channelRoot}/snapshot.json`), { force: true }); await generateReport(page, CHANNEL); } // Force a sweep regardless of the cadence — needed once lastFullSweepAt has // been stamped by an accepted sweep. async function runFullSweep( page: import("@playwright/test").Page, previousLastSyncedAt: string | undefined, ): Promise { await page.goto(channelStage(CHANNEL, "playlist")); for (let attempt = 0; attempt < 5; attempt++) { await page .getByRole("button", { name: "Full sweep", exact: true }) .click({ timeout: 5_000 }) .catch(() => {}); for (let i = 0; i < 60; i++) { const cfg = await readConfig().catch(() => null); if (cfg?.lastSyncedAt && cfg.lastSyncedAt !== previousLastSyncedAt) return; await new Promise((r) => setTimeout(r, 250)); } } throw new Error("runFullSweep: lastSyncedAt never advanced"); } // Controls which ids the fake yt-dlp's flat-playlist branch emits. Its cwd // sidecar serves the sweep's single unranged call exactly as it serves the // quick check's. async function setFreshPlaylist(ids: string[]): Promise { await writeFile( resolvePath(`${channelRoot}/.fake-ytdlp-flat-playlist.json`), JSON.stringify({ ids }), ); } // Archive every known video so the sweep's download walk finds nothing new — // the download set is not what this spec is about, and an archived first window // is also the paged walk's stopping condition. async function seedArchive(ids: string[]): Promise { await writeFile( resolvePath(`${channelRoot}/archive`), ids.map((id) => `youtube ${id}`).join("\n") + "\n", ); } // A stale stored playlist, standing in for the real complaint: before the // sweep, only an explicit "store playlist" ever rewrote this file, so it could // be months out of date while "download missing" kept reading it. async function seedStalePlaylist(): Promise { await writeFile( resolvePath(`${channelRoot}/playlist`), "https://www.youtube.com/watch?v=stale00000001\n", ); } async function enableFullSweep( confirmMaxSuspects: number, shrinkGuardPercent = 10, ): Promise { await writeSettings({ adminTitle: "Test Admin", maxTranscriptPageBytes: 8388608, sleepBetweenDownloadsSeconds: 0, minFreeDiskGB: 0, verifyAvailabilityBeforeClean: false, syncScheduler: { // The cron scheduler stays off; the sweep cadence is independent of it. enabled: false, fullSweepIntervalMinutes: 1440, fullSweepConfirmMaxSuspects: confirmMaxSuspects, fullSweepShrinkGuardPercent: shrinkGuardPercent, }, }); } async function readConfig(): Promise { return readJson(`${channelRoot}/config.json`); } // The fake yt-dlp logs each spawn it serves here, so a spec can prove a // per-video probe did or didn't happen. async function readInvocations(): Promise { try { return await readFile( resolvePath(`${channelRoot}/fake-ytdlp.invocations`), "utf8", ); } catch { return ""; } } async function readPlaylist(): Promise { const raw = await readFile(resolvePath(`${channelRoot}/playlist`), "utf8"); return raw.split("\n").filter(Boolean); } // Click Sync and wait for the run to land, detected by config.json's // lastSyncedAt advancing past what it was before the click. async function runSync( page: import("@playwright/test").Page, previousLastSyncedAt: string | undefined, ): Promise { await page.goto(channelStage(CHANNEL, "playlist")); // Retried: a click landing before React hydrates fires nothing at all — the // long-standing flake pattern in this suite. Sync is idempotent here. for (let attempt = 0; attempt < 5; attempt++) { await page .getByRole("button", { name: "Sync", exact: true }) .click({ timeout: 5_000 }) .catch(() => {}); for (let i = 0; i < 60; i++) { const cfg = await readConfig().catch(() => null); if (cfg?.lastSyncedAt && cfg.lastSyncedAt !== previousLastSyncedAt) return; await new Promise((r) => setTimeout(r, 250)); } } throw new Error("runSync: lastSyncedAt never advanced"); } test("a due full sweep refreshes the playlist and flags videos that left the listing", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(25); await seedArchive(ALL_IDS); await seedStalePlaylist(); // The listing no longer carries viddeleted1 — it has gone from the channel. await setFreshPlaylist(ALL_IDS.filter((id) => id !== "viddeleted1")); await generateReport(page, CHANNEL); await runSync(page, undefined); // The sweep's own log line, so a failure says which pass ran. await expect(page.getByLabel("Sync output")).toContainText("Full sweep"); // 1. The deletion signal, written by the sync itself — no Quick check click. const record = await readJson( `${channelRoot}/maybe-missing.json`, ); expect(record.ids).toEqual(["viddeleted1"]); expect(record.freshPlaylistCount).toBe(5); // 2. The stale stored playlist is replaced by the fresh listing. const playlist = await readPlaylist(); expect(playlist).toHaveLength(5); expect(playlist).not.toContain( "https://www.youtube.com/watch?v=stale00000001", ); expect(playlist).toContain("https://www.youtube.com/watch?v=vidpublic1"); // 3. Under the confirm cap, the same job resolved the suspect upstream. const availability = await readJson<{ availability: string }>( `${channelRoot}/data/viddeleted1/availability.json`, ); expect(availability.availability).toBe("deleted"); // One job, of kind `sync` — not a second job kind bolted alongside it. // Located by data-kind: the cell renders the label "Sync", so matching row // text on "sync" is a case-sensitive miss. await page.goto("/jobs"); const rows = page.getByRole("row").filter({ hasText: CHANNEL }); await expect(rows).toHaveCount(1); await expect(rows.first()).toHaveAttribute("data-kind", "sync"); await expect(rows.first()).toContainText("done"); }); test("over the confirm cap, suspects are flagged but not probed", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(1); await seedArchive(ALL_IDS); // Two videos gone, one over the cap of 1. await setFreshPlaylist( ALL_IDS.filter((id) => id !== "viddeleted1" && id !== "vidprivate1"), ); await generateReport(page, CHANNEL); await runSync(page, undefined); const record = await readJson( `${channelRoot}/maybe-missing.json`, ); expect(record.ids.sort()).toEqual(["viddeleted1", "vidprivate1"].sort()); // Flagged, but no per-video probe ran — that decision is left to a human. await expect(page.getByLabel("Sync output")).toContainText( "over the auto-confirm cap", ); const invocations = await readInvocations(); expect(invocations).not.toContain( "dump-json:https://www.youtube.com/watch?v=viddeleted1", ); expect(invocations).not.toContain( "dump-json:https://www.youtube.com/watch?v=vidprivate1", ); // So the suspects stay unresolved: every sync's availability backfill still // writes a record from metadata, but nothing classified these as gone. const deleted = await readJson<{ availability: string }>( `${channelRoot}/data/viddeleted1/availability.json`, ); expect(deleted.availability).not.toBe("deleted"); }); // --- Part C: the roster, the shrink guard, and never-fetched losses --------- test("a suspicious shrink is refused, then accepted when a second enumeration agrees", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(0); const full = [...ALL_IDS, ...BULK_IDS]; await seedArchive(full); // The stored listing a previous sweep left behind: 66 entries, of which 60 // were never downloaded and so have no dir and no metadata.info.json. This // file is the ONLY place their URLs live. await writePlaylist(full); // ...and the fetch comes back with 6 of them. Acting on it would truncate the // stored playlist and flag 60 videos as gone. await setFreshPlaylist(ALL_IDS); await generateReport(page, CHANNEL); await runSync(page, undefined); // 1. Refused. The playlist is untouched, and no missing-video flags were // written at all. await expect(page.getByLabel("Sync output")).toContainText("down 60 from 66"); expect(await readPlaylist()).toHaveLength(full.length); await expect( readJson(`${channelRoot}/maybe-missing.json`), ).rejects.toThrow(); // 2. But the roster was still seeded and merged — additive, so a bad fetch // can only ever add. Every URL survived the refusal. const afterShrink = await readRoster(); expect(Object.keys(afterShrink.entries)).toHaveLength(full.length); expect(afterShrink.entries.bulk00000001.url).toBe(ytUrl("bulk00000001")); expect(afterShrink.lastSweep?.verdict).toBe("shrink-suspect"); expect(afterShrink.lastSweep?.listedCount).toBe(ALL_IDS.length); // Not counted as a sweep, which is what makes the next ordinary sync retry // the enumeration instead of waiting out the daily cadence. expect((await readConfig()).lastFullSweepAt).toBeUndefined(); // 3. The next ORDINARY sync re-enumerates (not the cheap paged walk) and gets // the same answer. A real mass deletion repeats; a transient blip does not. await runSync(page, (await readConfig()).lastSyncedAt); await expect(page.getByLabel("Sync output")).toContainText( "confirmed by a second enumeration", ); expect(typeof (await readConfig()).lastFullSweepAt).toBe("string"); expect(await readPlaylist()).toHaveLength(ALL_IDS.length); const confirmed = await readRoster(); expect(confirmed.lastSweep?.verdict).toBe("shrink-confirmed"); // Nothing was ever dropped from the roster, so the 60 departed videos are // still recoverable by URL. expect(Object.keys(confirmed.entries)).toHaveLength(full.length); expect(confirmed.entries.bulk00000001.url).toBe(ytUrl("bulk00000001")); }); test("a never-downloaded video that leaves the listing is surfaced with its URL", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(0); // ghost is listed but never fetched: no data/ dir, so nothing on disk has // ever recorded its URL. const listed = [...ALL_IDS, "ghost0000001"]; await seedArchive(listed); await setFreshPlaylist(listed); await generateReport(page, CHANNEL); await runSync(page, undefined); expect((await readRoster()).entries.ghost0000001.url).toBe( ytUrl("ghost0000001"), ); // It leaves the listing. A drop of one is far under the guard's floor, so // this is an ordinary accepted sweep. await setFreshPlaylist(ALL_IDS); await runFullSweep(page, (await readConfig()).lastSyncedAt); await expect(page.getByLabel("Full sweep output")).toContainText( "listed but never downloaded", ); // It is NOT maybe-missing: that bucket means "on disk and no longer listed", // and its meaning is unchanged. const record = await readJson( `${channelRoot}/maybe-missing.json`, ); expect(record.ids).toEqual([]); await regenerateReport(page); const snapshot = await readJson( `${channelRoot}/snapshot.json`, ); expect(snapshot.missingNeverFetched).toEqual([ expect.objectContaining({ id: "ghost0000001", url: ytUrl("ghost0000001"), }), ]); // And it is rendered, with a usable link and a recovery action. await page.goto(channelStage(CHANNEL, "diagnostics")); const row = page.getByLabel("never fetched ghost0000001"); await expect(row).toBeVisible(); await expect(page.getByLabel("never fetched url ghost0000001")).toHaveAttribute( "href", ytUrl("ghost0000001"), ); await expect( row.getByRole("button", { name: "Try downloading anyway" }), ).toBeVisible(); }); test("seeding captures an undownloaded video's URL before the first sweep can erase it", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(0); await seedArchive(ALL_IDS); // The pre-existing state every channel is in today: a stored playlist, no // roster, and an entry in it that was never downloaded. `playlist` is the // ONLY place ghost's URL lives. await writePlaylist([...ALL_IDS, "ghost0000001"]); // ...and it has already gone from the channel, so the very first sweep would // have overwritten the playlist and taken the URL with it. await setFreshPlaylist(ALL_IDS); await generateReport(page, CHANNEL); await runSync(page, undefined); const roster = await readRoster(); expect(roster.entries.ghost0000001).toBeTruthy(); expect(roster.entries.ghost0000001.url).toBe(ytUrl("ghost0000001")); expect(roster.entries.ghost0000001.source).toBe("listing"); // Every on-disk video is in the roster too, so nothing is left as an orphan. for (const id of ALL_IDS) expect(roster.entries[id]).toBeTruthy(); // The playlist was legitimately refreshed — the migration ran first. expect(await readPlaylist()).toHaveLength(ALL_IDS.length); await regenerateReport(page); const snapshot = await readJson( `${channelRoot}/snapshot.json`, ); expect(snapshot.missingNeverFetched?.map((v) => v.id)).toEqual([ "ghost0000001", ]); }); test("the next sync stays on the cheap paged walk until the cadence elapses", async ({ page, }) => { await resetData(FIXTURE); await enableFullSweep(25); await seedArchive(ALL_IDS); await setFreshPlaylist(ALL_IDS.filter((id) => id !== "viddeleted1")); await generateReport(page, CHANNEL); await runSync(page, undefined); const afterSweep = await readConfig(); const sweptAt = afterSweep.lastFullSweepAt; expect(typeof sweptAt).toBe("string"); const firstRecord = await readJson( `${channelRoot}/maybe-missing.json`, ); // Second sync, immediately: the daily cadence has not elapsed, so it must not // re-enumerate — no new listing read, no rewritten flags. await runSync(page, afterSweep.lastSyncedAt); const secondRecord = await readJson( `${channelRoot}/maybe-missing.json`, ); expect(secondRecord.checkedAt).toBe(firstRecord.checkedAt); expect((await readConfig()).lastFullSweepAt).toBe(sweptAt); });