import { readFile, writeFile } from "node:fs/promises"; import { fileURLToPath } from "node:url"; import path from "node:path"; import { test, expect } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, writeSettings, } from "./helpers"; import { baseUrl } from "./baseUrl"; const CHANNEL = "test-live"; const ROOT = `test-transcripts/channels/${CHANNEL}`; const here = path.dirname(fileURLToPath(import.meta.url)); type DownloadOutcome = { status: string; attempts: Array<{ kind: string; handling: string }>; filter?: { name: string; reason: string }; }; type Snapshot = { buckets: { skippedByFilter?: string[] }; }; async function defaultSettings(): Promise> { const raw = await readFile( path.join(here, "fixtures", "test-settings.default.json"), "utf8", ); return JSON.parse(raw) as Record; } async function readInvocations(): Promise { return readFile( resolvePath(`${ROOT}/fake-ytdlp.invocations`), "utf8", ).catch(() => ""); } async function readArchive(): Promise { return readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(() => ""); } test("skip-live on by default: live + upcoming skipped, VOD + normal download", async ({ page, }) => { test.setTimeout(120_000); await resetData("skip-live-channel"); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "download")); await page.getByRole("button", { name: "Download videos" }).click(); const log = page.getByLabel("Download videos output"); await expect(log).toContainText( "Skipping islive000001: video is currently live [filter=skipLive]", { timeout: 60_000 }, ); await expect(log).toContainText( "Skipping isupcoming01: video is an upcoming/scheduled livestream [filter=skipLive]", { timeout: 60_000 }, ); await expect(log).toContainText("2 skipped by filter", { timeout: 60_000 }); await expect(log).toContainText("Managed download complete", { timeout: 60_000, }); // Live video: recorded as skipped, no transcript, not archived. const liveOutcome = await readJson( `${ROOT}/data/islive000001/download-outcome.json`, ); expect(liveOutcome.status).toBe("skipped-filtered"); expect(liveOutcome.filter?.name).toBe("skipLive"); expect( liveOutcome.attempts.some((a) => a.kind === "metadata-prefetch"), ).toBe(true); expect( await pathExists(`${ROOT}/data/islive000001/transcript.en.vtt`), ).toBe(false); expect(await readArchive()).not.toContain("islive000001"); // Finished livestream VOD must download normally (regression guard). expect( await pathExists(`${ROOT}/data/waslive00001/transcript.en.vtt`), ).toBe(true); const vodOutcome = await readJson( `${ROOT}/data/waslive00001/download-outcome.json`, ); expect(vodOutcome.status).toBe("ok"); expect(await readArchive()).toContain("waslive00001"); // Normal video downloads, and the split actually happened (prefetch + reuse). expect( await pathExists(`${ROOT}/data/normalvid001/transcript.en.vtt`), ).toBe(true); const invocations = await readInvocations(); expect(invocations).toMatch(/prefetch:.*normalvid001/); expect(invocations).toMatch( /load-info-json:.*data\/normalvid001\/metadata\.info\.json/, ); // Snapshot surfaces the skipped-live videos in their own bucket. The report // is regenerated automatically after the download job finishes, via the // global debounced scheduler, so poll until it lands rather than reading the // file the instant the job completes. await expect .poll( async () => { const snap = await readJson(`${ROOT}/snapshot.json`).catch( () => null, ); return snap?.buckets.skippedByFilter ?? null; }, { timeout: 15_000, intervals: [200, 300, 500] }, ) .toEqual(expect.arrayContaining(["islive000001", "isupcoming01"])); const snapshot = await readJson(`${ROOT}/snapshot.json`); expect(snapshot.buckets.skippedByFilter ?? []).not.toContain("waslive00001"); }); test("global skipLiveDownloads=false downloads live videos (split still runs)", async ({ page, }) => { test.setTimeout(120_000); await resetData("skip-live-channel"); await writeSettings({ ...(await defaultSettings()), skipLiveDownloads: false, }); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "download")); await page.getByRole("button", { name: "Download videos" }).click(); const log = page.getByLabel("Download videos output"); await expect(log).toContainText("Managed download complete", { timeout: 60_000, }); await expect(log).not.toContainText("Skipping islive000001"); // With the filter off, the live video downloads normally... expect( await pathExists(`${ROOT}/data/islive000001/transcript.en.vtt`), ).toBe(true); // ...but the metadata/download split still runs (always-on primitive). expect(await readInvocations()).toMatch(/prefetch:.*islive000001/); }); test("per-channel skipLiveDownloads=false overrides the global default", async ({ page, }) => { test.setTimeout(120_000); await resetData("skip-live-channel"); // Global default stays on; the channel opts out. await writeFile( resolvePath(`${ROOT}/config.json`), JSON.stringify( { handling: "youtube", name: "Test Skip Live", url: "https://www.youtube.com/@example/videos", skipLiveDownloads: false, }, null, 2, ), ); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "download")); await page.getByRole("button", { name: "Download videos" }).click(); const log = page.getByLabel("Download videos output"); await expect(log).toContainText("Managed download complete", { timeout: 60_000, }); await expect(log).not.toContainText("Skipping islive000001"); expect( await pathExists(`${ROOT}/data/islive000001/transcript.en.vtt`), ).toBe(true); });