import { stat, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, } from "./helpers"; import { baseUrl } from "./baseUrl"; // Mirror common/lib/format.ts formatBytes so the e2e assertion matches the UI // without depending on the package path alias resolving in the test runner. function formatBytes(bytes: number): string { if (bytes < 1024) return `${bytes} B`; if (bytes < 1024 * 1024) return `${(bytes / 1024).toFixed(1)} KB`; if (bytes < 1024 * 1024 * 1024) return `${(bytes / (1024 * 1024)).toFixed(1)} MB`; return `${(bytes / (1024 * 1024 * 1024)).toFixed(2)} GB`; } const SLUG = "test-transcribe"; const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; function dataRel(videoId: string, file: string): string { return `test-transcripts/channels/${SLUG}/data/${videoId}/${file}`; } // Give vidA and vidB a whisper transcript so both qualify for the // transcribed-audio cleanup. vidA gets protected, vidB does not. async function seedTranscript(videoId: string): Promise { await writeFile( resolvePath(dataRel(videoId, "transcript.json")), '{"transcription":[]}\n', ); } test("'do not clean' protects a video's audio from cleanup; toggling off restores it", async ({ page, }) => { await resetData("one-transcribe-channel-with-audio"); await seedTranscript("vidA"); await seedTranscript("vidB"); // The fixture seeds audio.m4a for each video; config audioFormat is m4a. await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); // Mark vidA "do not clean" on its video page. await page.goto(`/channels/${SLUG}/videos/vidA`); await page .getByRole("button", { name: "mark video vidA do not clean" }) .click(); await expect(page.getByLabel("media archived")).toBeVisible(); // Regenerating the snapshot (visiting the channel page) must omit the // protected id from transcribedWithAudio while keeping the unprotected one. await generateReport(page, SLUG); await page.goto(`/channels/${SLUG}`); const snapshot = await readJson<{ buckets: { transcribedWithAudio?: string[] }; }>(SNAPSHOT_REL); expect(snapshot.buckets.transcribedWithAudio).toContain("vidB"); expect(snapshot.buckets.transcribedWithAudio).not.toContain("vidA"); // Run the transcribed-audio cleanup from the channel's Cleanup stage. await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); const log = page.getByLabel("Clean audio output"); await expect(log).toContainText("Skipped 1 (do not clean)", { timeout: 30_000, }); // vidB's audio is gone; vidA's is preserved. expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(false); expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(true); // Toggle vidA back to cleanable, then re-run cleanup. await page.goto(`/channels/${SLUG}/videos/vidA`); await page .getByRole("button", { name: "allow cleanup for video vidA" }) .click(); await expect(page.getByLabel("media archived")).toHaveCount(0); await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "Cleaned 1 audio file", { timeout: 30_000 }, ); expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); }); test("snapshot records reclaimable cleanup bytes and the Cleanup stage shows the estimate", async ({ page, }) => { await resetData("one-transcribe-channel-with-audio"); await seedTranscript("vidA"); await seedTranscript("vidB"); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); // Sum the audio that the transcribed-audio cleanup would delete (both videos // have a whisper transcript and an audio.m4a on disk). const sizeA = (await stat(resolvePath(dataRel("vidA", "audio.m4a")))).size; const sizeB = (await stat(resolvePath(dataRel("vidB", "audio.m4a")))).size; const expectedBytes = sizeA + sizeB; // Visiting the channel page regenerates the snapshot from disk. await generateReport(page, SLUG); await page.goto(`/channels/${SLUG}`); const snapshot = await readJson<{ cleanupBytes?: { transcribedWithAudio?: number }; }>(SNAPSHOT_REL); expect(snapshot.cleanupBytes?.transcribedWithAudio).toBe(expectedBytes); await page.goto(channelStage(SLUG, "cleanup")); await expect( page.getByText(`Estimated space to reclaim: ~${formatBytes(expectedBytes)}`), ).toBeVisible(); });