import { writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, writeSettings, } from "./helpers"; import { baseUrl } from "./baseUrl"; // Speaker-diarization capture. The point of this lane is TIMING, not quality: // audio is deleted once a video is transcribed, so the speaker turns get // captured while the audio is still on disk or never. These specs pin the three // behaviours that make that safe. // // The engine is faked via DIARIZE_BIN (e2e/fixtures/bin/fake-diarize.mjs), wired // in the dev:test / start:test scripts the way every other fake binary is. const SLUG = "test-transcribe"; const SNAPSHOT_REL = `test-transcripts/channels/${SLUG}/snapshot.json`; function dataRel(videoId: string, file: string): string { return `test-transcripts/channels/${SLUG}/data/${videoId}/${file}`; } async function seedTranscript(videoId: string): Promise { await writeFile( resolvePath(dataRel(videoId, "transcript.json")), '{"transcription":[]}\n', ); } async function seedDiarization(videoId: string): Promise { await writeFile( resolvePath(dataRel(videoId, "diarization.json")), JSON.stringify({ videoId, generatedAt: "2026-08-07T00:00:00.000Z", speakers: 1, turns: [{ start: 0, end: 10, speaker: 0 }], engine: { engine: "fake-diarize", threshold: 0.5 }, }) + "\n", ); } const BASE_SETTINGS = { adminTitle: "Test Admin", maxTranscriptPageBytes: 8388608, sleepBetweenDownloadsSeconds: 0, minFreeDiskGB: 0, verifyAvailabilityBeforeClean: false, syncScheduler: { fullSweepIntervalMinutes: 0 }, }; function diarizationSettings(over: Record = {}) { return { ...BASE_SETTINGS, diarization: { enabled: true, inlineAfterTranscribe: true, threshold: 0.5, threads: 1, python: "python3", // Any non-empty pair: the fake engine never reads them, but an empty pair // is reported as "not configured" and the lane never runs. segModel: "/dev/null", embModel: "/dev/null", concurrency: 1, ...over, }, }; } // These drive a real transcribe batch over three videos and then a cleanup // sweep, which does not fit the 30s default on a loaded box. const SLOW = 120_000; // (a) A transcription produces diarization.json. test("transcribing a video captures its speaker turns", async ({ page }) => { test.setTimeout(SLOW); await resetData("one-transcribe-channel-with-audio"); await writeSettings(diarizationSettings()); await page.goto(channelStage(SLUG, "transcribe")); await page .getByRole("button", { name: "Transcribe missing", exact: true }) .click(); await expect .poll(async () => pathExists(dataRel("vidA", "diarization.json")), { timeout: 60_000, }) .toBe(true); const record = await readJson<{ videoId: string; speakers: number; turns: { start: number; end: number; speaker: number }[]; engine: { engine: string; threshold: number }; }>(dataRel("vidA", "diarization.json")); expect(record.videoId).toBe("vidA"); expect(record.turns.length).toBeGreaterThan(0); expect(record.speakers).toBe(2); // Provenance is the whole reason this is a record and not a bare array: a // later attribution pass has to be able to tell what produced a given file. expect(record.engine.engine).toBe("fake-diarize"); expect(record.engine.threshold).toBe(0.5); // Let the batch finish before the test ends. Without this it leaks a running // transcribe job into the NEXT spec, whose resetData then races it — the // runner re-creates files in the tree the walk is deleting. (resetData's own // maxRetries comment documents this hazard; the fix belongs here, at the // source, rather than relying on the retry.) vidC is the fake diarizer's // designated failure, so it gets a transcript but never a sidecar. await expect .poll(async () => pathExists(dataRel("vidB", "diarization.json")), { timeout: 60_000, }) .toBe(true); await expect .poll(async () => pathExists(dataRel("vidC", "transcript.json")), { timeout: 60_000, }) .toBe(true); }); // (b) A failing diarizer must not fail the transcription. test("a failing diarizer does not fail the transcription", async ({ page }) => { test.setTimeout(SLOW); // The fake engine exits non-zero for vidC (see fake-diarize.mjs). vidA and // vidB succeed in the same run, so this also pins that one video's failure // does not take the batch down with it. await resetData("one-transcribe-channel-with-audio"); await writeSettings(diarizationSettings()); await page.goto(channelStage(SLUG, "transcribe")); await page .getByRole("button", { name: "Transcribe missing", exact: true }) .click(); // The transcript still lands for the video whose diarization failed — // diarization is strictly best-effort and must never fail a transcription // that already succeeded. await expect .poll(async () => pathExists(dataRel("vidC", "transcript.json")), { timeout: 60_000, }) .toBe(true); // Its neighbours were diarized normally. Waiting for BOTH also settles the // batch, so this spec does not leak a running job into the next one and the // cleanup below sees a stable tree. await expect .poll(async () => pathExists(dataRel("vidA", "diarization.json")), { timeout: 60_000, }) .toBe(true); await expect .poll(async () => pathExists(dataRel("vidB", "diarization.json")), { timeout: 60_000, }) .toBe(true); // No sidecar for vidC, so its audio stays eligible for a later pass rather // than being silently treated as done. expect(await pathExists(dataRel("vidC", "diarization.json"))).toBe(false); // And the guard holds vidC's audio while releasing the diarized ones. await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "awaiting diarization", { timeout: 30_000 }, ); expect(await pathExists(dataRel("vidC", "audio.m4a"))).toBe(true); }); // (c) THE REGRESSION THAT WOULD SILENTLY DESTROY DATA. // Cleanup must refuse to delete audio for a transcribed-but-not-yet-diarized // video. Without this guard an async diarize pass races the sweep and loses the // only copy of the audio, permanently and with no trace. test("cleanup refuses to delete audio for a transcribed-but-undiarized video", async ({ page, }) => { test.setTimeout(SLOW); await resetData("one-transcribe-channel-with-audio"); // Capture ON, inline OFF — the recommended batch shape, and precisely the // window in which the race exists. await writeSettings(diarizationSettings({ inlineAfterTranscribe: false })); await seedTranscript("vidA"); await seedTranscript("vidB"); // vidA has been diarized; vidB has not. await seedDiarization("vidA"); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); // The snapshot's cleanup bucket must agree with the sweep, or "Est. reclaim" // promises space the sweep is going to refuse to take. // // Poll-with-reload rather than read once: the channel page serves a PERSISTED // snapshot and regenerates on a debounce, and the settings cache this reads // `diarization.enabled` from is invalidated by a fire-and-forget fetch. A // single read can therefore catch a snapshot built from the pre-writeSettings // defaults (diarization off), in which case the guard exclusion has not been // applied yet and vidB is still in the bucket. await generateReport(page, SLUG); await expect .poll( async () => { await page.goto(`/channels/${SLUG}`); const snap = await readJson<{ buckets: { transcribedWithAudio?: string[] }; }>(SNAPSHOT_REL); return snap.buckets.transcribedWithAudio ?? []; }, { timeout: 60_000 }, ) .toEqual(["vidA"]); await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "awaiting diarization", { timeout: 30_000 }, ); // vidA's audio is reclaimed; vidB's is held. expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(true); }); // The backfill loop end to end — this is the workflow a large batch actually // uses: transcribe at full GPU speed with the hook off, let the guard hold the // audio, then catch up. It is also the only path that exercises diarizeAll. test("the Diarize speakers backfill clears the hold and releases the audio", async ({ page, }) => { test.setTimeout(SLOW); await resetData("one-transcribe-channel-with-audio"); await writeSettings(diarizationSettings({ inlineAfterTranscribe: false })); await seedTranscript("vidA"); await seedTranscript("vidB"); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); // Nothing is diarized yet, so the sweep must take nothing. await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "awaiting diarization", { timeout: 30_000 }, ); expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(true); expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(true); // Backfill. await page .getByRole("button", { name: "Diarize speakers", exact: true }) .click(); await expect .poll(async () => pathExists(dataRel("vidA", "diarization.json")), { timeout: 60_000, }) .toBe(true); await expect .poll(async () => pathExists(dataRel("vidB", "diarization.json")), { timeout: 60_000, }) .toBe(true); // With the sidecars in place the hold clears and the audio is reclaimable — // the whole point of the guard being transient rather than a protection. await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "Cleaned 2 audio file", { timeout: 30_000 }, ); expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(false); }); // The guard is opt-in, and turning it off must restore the old behaviour // exactly — otherwise enabling diarization would be a one-way door for disk. test("with diarization disabled, cleanup deletes undiarized audio as before", async ({ page, }) => { test.setTimeout(SLOW); await resetData("one-transcribe-channel-with-audio"); await writeSettings(diarizationSettings({ enabled: false })); await seedTranscript("vidA"); await seedTranscript("vidB"); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "cleanup")); await page.getByRole("button", { name: "Clean audio", exact: true }).click(); await expect(page.getByLabel("Clean audio output")).toContainText( "Cleaned 2 audio file", { timeout: 30_000 }, ); expect(await pathExists(dataRel("vidA", "audio.m4a"))).toBe(false); expect(await pathExists(dataRel("vidB", "audio.m4a"))).toBe(false); });