import { mkdir, readFile, writeFile } from "node:fs/promises"; import { test, expect, type Page } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, } from "./helpers"; import { baseUrl } from "./baseUrl"; // The per-channel download filter and the metadata scan that feeds it. // // The fake yt-dlp titles every video "Synthetic " (fixtures/bin/fake-ytdlp.mjs), // so an include of /guest/ matches guestvid0001 and guestvid0002 and misses // plainvid0001 and plainvid0002 — no fixture-specific metadata needed. // `needsauthvid01` is the sentinel the fake refuses without cookies, so the scan // has one video it cannot read. const CHANNEL = "test-filter"; const ROOT = `test-transcripts/channels/${CHANNEL}`; const GUEST = ["guestvid0001", "guestvid0002"]; const PLAIN = ["plainvid0001", "plainvid0002"]; const NEEDS_AUTH = "needsauthvid01"; // A FINISHED livestream VOD. Its title contains neither "guest" nor "plain", so // the only thing that can let it through is `includeLivestreams`. const LIVE = "livevid000001"; const SETTLED_BY_GUEST = [LIVE, ...PLAIN].sort(); type MetadataScan = { version: number; entries: Record< string, { title: string; description: string; uploadDate: string; scannedAt: string } >; errors: Record; lastRun: { scanned: number; errors: number; stopped?: string } | null; }; type Snapshot = { generatedAt: string; totals: { videos: number; transcribed: number; downloaded: number }; buckets: { skippedByTitleFilter?: string[]; skippedByFilter?: string[] }; undownloadedIds: string[]; metadataScan?: { scanned: number; errors: number; unscanned: number }; }; async function readInvocations(): Promise { return readFile(resolvePath(`${ROOT}/fake-ytdlp.invocations`), "utf8").catch( () => "", ); } async function readArchive(): Promise { return readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(() => ""); } async function writeFilterConfig(filter: Record | null) { await writeFile( resolvePath(`${ROOT}/config.json`), JSON.stringify( { handling: "youtube", name: "Test Title Filter", url: "https://www.youtube.com/@example/videos", ...(filter ? { downloadFilter: filter } : {}), }, null, 2, ), ); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } // Regenerate the report and wait for the snapshot to actually advance. The click // is retried for the usual reason (a pre-hydration click fires nothing); // regenerating is idempotent. // // `until` exists because "newer" is not the same as "reflects my change". The // debounced snapshot scheduler arms itself whenever a job finishes, so a regen // triggered by the SCAN can land after a config edit and carry the config from // before it — a newer generatedAt with the old verdict. Callers that just // changed the filter pass a predicate and get a report that has actually seen // it; they still assert the exact contents afterwards. async function refreshReport( page: Page, after: string, until?: (snapshot: Snapshot) => boolean, ): Promise { await page.goto("/channels"); const refresh = page.getByRole("button", { name: `refresh report ${CHANNEL}`, }); await refresh.waitFor({ state: "visible" }); await expect .poll( async () => { await refresh.click({ timeout: 5_000 }).catch(() => {}); for (let i = 0; i < 20; i++) { const cur = await readJson(`${ROOT}/snapshot.json`).catch( () => null, ); if (cur && cur.generatedAt > after && (!until || until(cur))) { return true; } await new Promise((r) => setTimeout(r, 250)); } return false; }, { timeout: 60_000, intervals: [1000] }, ) .toBe(true); return readJson(`${ROOT}/snapshot.json`); } async function runScan(page: Page): Promise { await page.goto(channelStage(CHANNEL, "playlist")); await page.getByRole("button", { name: "Scan metadata" }).click(); await expect(page.getByLabel("Scan metadata output")).toContainText( "Metadata scan complete", { timeout: 60_000 }, ); } async function download(page: Page): Promise { await page.goto(channelStage(CHANNEL, "download")); await page.getByRole("button", { name: "Download videos" }).click(); await expect(page.getByLabel("Download videos output")).toContainText( "Managed download complete", { timeout: 60_000 }, ); } test("the scan reads titles without creating a single video directory", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await runScan(page); const scan = await readJson(`${ROOT}/metadata-scan.json`); expect(Object.keys(scan.entries).sort()).toEqual( [...GUEST, ...PLAIN, LIVE].sort(), ); expect(scan.entries.guestvid0001.title).toBe("Synthetic guestvid0001"); // The one video the fake refuses without cookies. No cookie spec is // configured, so there is no auth retry to make and it stays an error. expect(Object.keys(scan.errors)).toEqual([NEEDS_AUTH]); expect(scan.errors[NEEDS_AUTH].class).toBe("needs_auth"); expect(scan.lastRun?.scanned).toBe(5); // THE LOAD-BEARING ASSERTION. A video directory holding a metadata.info.json // is admitted to the LMDB index and the exported site by buildIndex, and its // dir name is what deriveChannelSets reads as "ever fetched". The scan must // create none of them. for (const id of [...GUEST, ...PLAIN, LIVE, NEEDS_AUTH]) { expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); } expect(await readArchive()).toBe(""); // The report then settles the non-matches, with nothing written per video. const before = await readJson(`${ROOT}/snapshot.json`); const snapshot = await refreshReport(page, before.generatedAt); expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(SETTLED_BY_GUEST); expect([...snapshot.undownloadedIds].sort()).toEqual( [...GUEST, NEEDS_AUTH].sort(), ); expect(snapshot.metadataScan?.scanned).toBe(5); expect(snapshot.metadataScan?.errors).toBe(1); // The unreadable video is not re-queued for a scan for a day — otherwise a // members-only video would keep the backlog above zero forever. It is still // ordinary download work, asserted above. expect(snapshot.metadataScan?.unscanned).toBe(0); }); test("a settled video is never downloaded", async ({ page }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await runScan(page); const afterScan = await readJson(`${ROOT}/snapshot.json`); await refreshReport(page, afterScan.generatedAt); await download(page); for (const id of GUEST) { expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); expect(await readArchive()).toContain(id); } // Settled: no directory, no archive line, and — the part that matters on a // channel with ~1,800 of them — yt-dlp was never invoked for them at all. const invocations = await readInvocations(); for (const id of SETTLED_BY_GUEST) { expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); expect(await readArchive()).not.toContain(id); expect(invocations).not.toContain(`prefetch:https://www.youtube.com/watch?v=${id}`); } await expect(page.getByLabel("Download videos output")).toContainText( "Settled by this channel's download filter: 3 skipped", ); }); test("a title-filter rejection leaves no video directory behind", async ({ page, }) => { test.setTimeout(180_000); // NO SCAN FIRST, deliberately: that is the only way a rejection reaches the // downloader at all. Once the scan has read a video the filter settles it and // yt-dlp is never invoked for it (the test above). The leftover this pins // belongs to the other case — a newly listed video the download lane reaches // before any scan does — where the metadata PREFETCH has already written // data//metadata.info.json by the time the filter gets to say no. // // A directory holding a metadata.info.json is admitted to the LMDB index and // the published site by buildIndex, and its name is what deriveChannelSets // reads as "ever fetched". The scan creates none of them; a rejection must // not either, or the same channel gets two different answers depending on // which path reached the video first. await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await download(page); const invocations = await readInvocations(); for (const id of GUEST) { expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); } for (const id of SETTLED_BY_GUEST) { // It WAS prefetched — that is the whole difference from the settled case — // and the directory that prefetch made is gone again. expect(invocations).toContain( `prefetch:https://www.youtube.com/watch?v=${id}`, ); expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); expect(await readArchive()).not.toContain(id); } // AND THE COUNT DOES NOT MOVE. The bucket is derived from the metadata-scan // store — the rejection recorded its entry there before discarding the // directory — so removing the directory costs it nothing. const before = await readJson(`${ROOT}/snapshot.json`); const snapshot = await refreshReport( page, before.generatedAt, (s) => (s.buckets.skippedByTitleFilter ?? []).length > 0, ); // FOUR, not three, and the extra one is the point of this path rather than a // fixture quirk: `needsauthvid01` is the video the METADATA SCAN cannot read // (the scan's fake refuses it without cookies, so the scan-first test above // records an ERROR for it and settles nothing). A download-time prefetch // reads it fine, so the filter decides it here — which is exactly the // difference this test exists to exercise. const settledByDownload = [...SETTLED_BY_GUEST, NEEDS_AUTH].sort(); expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(settledByDownload); const scan = await readJson(`${ROOT}/metadata-scan.json`); expect(Object.keys(scan.entries).sort()).toEqual(settledByDownload); }); test("a rejection never deletes a directory that holds somebody else's data", async ({ page, }) => { test.setTimeout(180_000); // The discard refuses anything but its own prefetch. A CLIP WINDOW is the // case worth pinning: `data//clips/` is media another tool asked this // editor for (umtool's POST /api/media/fetch-window), it is not a // "destination" so the run still attempts the video, and it is invisible to // every video-dir enumerator — so a discard that walked past it would delete // bytes nothing else would ever mention again. // // The second video carries DOWNLOADED MEDIA. On a youtube-handling channel // the destination is transcript.en.vtt, so an audio.mp3 does not prefilter // the video away — the run attempts it, the filter rejects it, and the bytes // have to survive. A video downloaded before the filter was written is // downloaded: that is a fact, not a preference. await resetData("title-filter-channel"); const [withClips, withAudio] = PLAIN; await mkdir(resolvePath(`${ROOT}/data/${withClips}/clips`), { recursive: true, }); await writeFile( resolvePath(`${ROOT}/data/${withClips}/clips/0.00-30.00.mp3`), "fake clip audio\n", ); await mkdir(resolvePath(`${ROOT}/data/${withAudio}`), { recursive: true }); await writeFile( resolvePath(`${ROOT}/data/${withAudio}/audio.mp3`), "fake audio bytes\n", ); await generateReport(page, CHANNEL); await download(page); for (const [id, file] of [ [withClips, "clips/0.00-30.00.mp3"], [withAudio, "audio.mp3"], ] as const) { expect(await pathExists(`${ROOT}/data/${id}/${file}`)).toBe(true); // The directory stayed, so the retryable outcome sidecar is written for it // exactly as it was before this rule existed. const outcome = await readJson<{ status: string }>( `${ROOT}/data/${id}/download-outcome.json`, ); expect(outcome.status).toBe("skipped-filtered"); } // And the rejection that had nothing of its own is still gone. expect(await pathExists(`${ROOT}/data/${LIVE}`)).toBe(false); }); test("changing the filter re-decides the channel with no rescan", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await runScan(page); const afterScan = await readJson(`${ROOT}/snapshot.json`); const settled = await refreshReport(page, afterScan.generatedAt); expect(settled.buckets.skippedByTitleFilter ?? []).toEqual(SETTLED_BY_GUEST); const invocationsBefore = await readInvocations(); // Flip the filter on disk. Nothing stored per video has to change, because // nothing about the verdict was ever stored per video. const settledBefore = JSON.stringify(settled.buckets.skippedByTitleFilter ?? []); await writeFilterConfig({ include: "plain" }); const flipped = await refreshReport( page, settled.generatedAt, (s) => JSON.stringify(s.buckets.skippedByTitleFilter ?? []) !== settledBefore, ); expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual( [...GUEST, LIVE].sort(), ); for (const id of PLAIN) expect(flipped.undownloadedIds).toContain(id); for (const id of GUEST) expect(flipped.undownloadedIds).not.toContain(id); // No rescan: re-deciding is pure computation over what the scan already read. expect(await readInvocations()).toBe(invocationsBefore); // And the download now fetches the other half. await download(page); for (const id of PLAIN) { expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); } for (const id of GUEST) { expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); } }); test("the Playlist stage reports the scan and what the filter makes of it", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "playlist")); const summary = page.getByLabel("metadata scan summary"); await expect(summary).toContainText("Scanned 0 of 6 listed"); await page.getByRole("button", { name: "Scan metadata" }).click(); await expect(page.getByLabel("Scan metadata output")).toContainText( "Metadata scan complete", { timeout: 60_000 }, ); await page.goto(channelStage(CHANNEL, "playlist")); await expect(summary).toContainText("Scanned 5 of 6 listed"); await expect(summary).toContainText("1 could not be read"); const verdict = page.getByLabel("download filter verdict"); await expect(verdict).toContainText("2 match"); await expect(verdict).toContainText("3 filtered out"); await expect(page.getByLabel("download filter match split")).toContainText( "2 by title/description · 0 livestreams", ); await page.getByText("Matched titles").click(); const matched = page.getByLabel("matched titles"); await expect(matched).toContainText("Synthetic guestvid0001"); await expect(matched).not.toContainText("Synthetic plainvid0001"); }); test("the form round-trips both patterns and refuses an invalid regex", async ({ page, }) => { test.setTimeout(120_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "configure")); await page.locator("summary").filter({ hasText: "Advanced" }).click(); await expect(page.getByLabel(/^download filter: include/i)).toHaveValue( "guest", ); await expect(page.getByLabel(/^download filter: exclude/i)).toHaveValue(""); // An unparseable pattern is refused by the form, because a bad pattern on disk // makes the filter silently inert at download time. await page.getByLabel(/^download filter: exclude/i).fill("elf("); await page.getByRole("button", { name: "Save changes" }).click(); await expect( page.getByText(/Download filter exclude is not a valid regular expression/), ).toBeVisible({ timeout: 30_000 }); expect( ( await readJson<{ downloadFilter?: Record }>( `${ROOT}/config.json`, ) ).downloadFilter, ).toEqual({ include: "guest" }); // A valid pair round-trips. await page.goto(channelStage(CHANNEL, "configure")); await page.locator("summary").filter({ hasText: "Advanced" }).click(); await page.getByLabel(/^download filter: include/i).fill("guest|special"); await page.getByLabel(/^download filter: exclude/i).fill("rerun"); await page.getByRole("button", { name: "Save changes" }).click(); await expect .poll( async () => ( await readJson<{ downloadFilter?: Record }>( `${ROOT}/config.json`, ).catch(() => ({}) as { downloadFilter?: Record }) ).downloadFilter ?? null, { timeout: 30_000, intervals: [250, 500] }, ) .toEqual({ include: "guest|special", exclude: "rerun" }); await page.goto(channelStage(CHANNEL, "configure")); await page.locator("summary").filter({ hasText: "Advanced" }).click(); await expect(page.getByLabel(/^download filter: include/i)).toHaveValue( "guest|special", ); await expect(page.getByLabel(/^download filter: exclude/i)).toHaveValue( "rerun", ); // Clearing both clears the stored key (CHANNEL_FORM_FIELDS), which is what // makes "no filter" reachable from the form at all. await page.getByLabel(/^download filter: include/i).fill(""); await page.getByLabel(/^download filter: exclude/i).fill(""); await page.getByRole("button", { name: "Save changes" }).click(); await expect .poll( async () => { const cfg = await readJson>( `${ROOT}/config.json`, ).catch(() => null); return cfg ? "downloadFilter" in cfg : null; }, { timeout: 30_000, intervals: [250, 500] }, ) .toBe(false); }); test("/operations/metadata-scan lists the channel and can run it", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); // The backlog is a snapshot field, so the report has to have seen the // unscanned playlist before the operation page can offer the work. const first = await readJson(`${ROOT}/snapshot.json`); const seeded = await refreshReport(page, first.generatedAt); expect(seeded.metadataScan?.unscanned).toBe(6); await page.goto("/operations/metadata-scan"); await expect( page.getByRole("heading", { name: "Metadata scan" }), ).toBeVisible(); await expect(page.getByRole("link", { name: CHANNEL })).toBeVisible(); await page.getByRole("button", { name: "Scan metadata" }).first().click(); await expect .poll( async () => Object.keys( ( await readJson(`${ROOT}/metadata-scan.json`).catch( () => null, ) )?.entries ?? {}, ).length, { timeout: 60_000, intervals: [500, 1000] }, ) .toBe(5); }); test("Include every livestream lets the VOD through, with no rescan", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); await runScan(page); const afterScan = await readJson(`${ROOT}/snapshot.json`); const settled = await refreshReport(page, afterScan.generatedAt); // The livestream is settled by /guest/ like any other non-matching title. expect(settled.buckets.skippedByTitleFilter ?? []).toContain(LIVE); const invocationsBefore = await readInvocations(); await page.goto(channelStage(CHANNEL, "configure")); await page.locator("summary").filter({ hasText: "Advanced" }).click(); await page.getByLabel(/^include every livestream/i).check(); await page.getByRole("button", { name: "Save changes" }).click(); await expect .poll( async () => ( await readJson<{ downloadFilter?: Record }>( `${ROOT}/config.json`, ).catch(() => ({}) as { downloadFilter?: Record }) ).downloadFilter ?? null, { timeout: 30_000, intervals: [250, 500] }, ) .toEqual({ include: "guest", includeLivestreams: true }); const flipped = await refreshReport( page, settled.generatedAt, (s) => !(s.buckets.skippedByTitleFilter ?? []).includes(LIVE), ); expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN); expect(flipped.undownloadedIds).toContain(LIVE); // A second positive selector, evaluated over what the scan already read. expect(await readInvocations()).toBe(invocationsBefore); // The stage now says WHY it is matched, which is the whole reason the count // is split: turning this on would otherwise look like widening the regex. await page.goto(channelStage(CHANNEL, "playlist")); await expect(page.getByLabel("download filter match split")).toContainText( "2 by title/description · 1 livestream", ); await download(page); expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.en.vtt`)).toBe(true); });