// The opt-in "replace YouTube auto-captions" lane, end to end. // // A video whose only transcript is YouTube's speech recognition is invisible to // every transcribe path (isVideoTranscribed counts any English VTT). These tests // prove the three new snapshot buckets classify such videos correctly — and only // such videos — and that the manual controls walk one through the whole lane: // autoSubsOnly → (fetch audio) → downloadedAutoSubsOnly → (transcribe) // → supersededAutoSubs → (purge) → nothing. // // Provenance comes from a 4 KB sniff of the VTT itself (see // common/lib/subtitleProvenance.ts), so the fixtures below use realistically // shaped cues: ASR tracks carry `align:start position:N%` plus inline word // timings, manual tracks carry neither. // // Run in default dev mode — E2E_MODE=start serves a stale build. import { mkdir, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, writeSettings, } from "./helpers"; import { baseUrl } from "./baseUrl"; const SLUG = "test-auto-subs"; const ROOT = `test-transcripts/channels/${SLUG}`; const SNAPSHOT_REL = `${ROOT}/snapshot.json`; // Shaped after real yt-dlp --write-auto-subs output. const ASR_VTT = `WEBVTT Kind: captions Language: en 00:00:00.030 --> 00:00:03.919 align:start position:0% so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to 00:00:03.919 --> 00:00:03.929 align:start position:0% so today we're going to 00:00:03.929 --> 00:00:07.070 align:start position:0% so today we're going to talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing `; // Shaped after a human-uploaded caption track: no cue settings, no word timings. const MANUAL_VTT = `WEBVTT Kind: captions Language: en 00:00:01.000 --> 00:00:04.000 So today we're going to talk about the whole thing. 00:00:04.000 --> 00:00:08.500 It's a long story, but bear with me. 00:00:08.500 --> 00:00:12.000 Here we go. `; const WHISPER_JSON = JSON.stringify({ transcription: [{ text: "a real transcript" }], }); type SeedVideo = { id: string; // "asr" writes an auto-caption-shaped VTT and metadata listing the track under // automatic_captions; "manual" writes a human-shaped VTT under subtitles. captions: "asr" | "manual"; whisper?: boolean; audio?: boolean; doNotClean?: boolean; }; function dataRel(videoId: string, file: string): string { return `${ROOT}/data/${videoId}/${file}`; } async function seedChannel(videos: SeedVideo[]): Promise { await mkdir(resolvePath(`${ROOT}/data`), { recursive: true }); // A youtube-handling channel — the case the lane exists for. audioFormat is // pinned so the forced transcribe-handling download lands on audio.mp3. await writeFile( resolvePath(`${ROOT}/config.json`), JSON.stringify({ handling: "youtube", name: "Auto-subs test channel", url: "https://www.youtube.com/@autosubs/videos", audioFormat: "mp3", }), ); // retry-bucket resolves each id's source URL from the stored playlist. await writeFile( resolvePath(`${ROOT}/playlist`), videos.map((v) => `https://www.youtube.com/watch?v=${v.id}\n`).join(""), ); for (const v of videos) { const dir = resolvePath(`${ROOT}/data/${v.id}`); await mkdir(dir, { recursive: true }); await writeFile( `${dir}/transcript.en.vtt`, v.captions === "asr" ? ASR_VTT : MANUAL_VTT, ); const track = { en: [{ ext: "vtt", url: "fake://subs" }] }; await writeFile( `${dir}/metadata.info.json`, JSON.stringify({ id: v.id, title: `Synthetic ${v.id}`, upload_date: "20240101", duration: 60, extractor_key: "Youtube", webpage_url: `https://www.youtube.com/watch?v=${v.id}`, subtitles: v.captions === "manual" ? track : {}, automatic_captions: v.captions === "asr" ? track : {}, }), ); if (v.whisper) await writeFile(`${dir}/transcript.json`, WHISPER_JSON); if (v.audio) await writeFile(`${dir}/audio.mp3`, `fake audio ${v.id}\n`); if (v.doNotClean) { await writeFile( `${dir}/do-not-clean.json`, JSON.stringify({ setAt: new Date(0).toISOString() }), ); } } await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } type Buckets = { autoSubsOnly?: string[]; downloadedAutoSubsOnly?: string[]; supersededAutoSubs?: string[]; }; // Snapshot regeneration is debounced (~1s) after a page visit / job finish, so // every assertion on it polls rather than reading once. async function expectBuckets(expected: Buckets): Promise { await expect .poll( async () => { const snap = await readJson<{ buckets: Buckets }>(SNAPSHOT_REL).catch( () => null, ); if (!snap) return null; return { autoSubsOnly: snap.buckets.autoSubsOnly ?? [], downloadedAutoSubsOnly: snap.buckets.downloadedAutoSubsOnly ?? [], supersededAutoSubs: snap.buckets.supersededAutoSubs ?? [], }; }, { timeout: 20_000 }, ) .toEqual({ autoSubsOnly: expected.autoSubsOnly ?? [], downloadedAutoSubsOnly: expected.downloadedAutoSubsOnly ?? [], supersededAutoSubs: expected.supersededAutoSubs ?? [], }); } test("walks an auto-caption video through fetch → transcribe → purge", async ({ page, }) => { test.setTimeout(120_000); await resetData(); await seedChannel([ { id: "asrvid0001", captions: "asr" }, // Negative control: a human-captioned video must never enter the lane. { id: "manvid0001", captions: "manual" }, ]); // --- Step 0: classification ----------------------------------------------- await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "transcribe")); await expectBuckets({ autoSubsOnly: ["asrvid0001"] }); await expect( page.getByRole("heading", { name: "YouTube auto-captions only (1)" }), ).toBeVisible(); await expect( page .getByLabel("auto captions needing audio list") .getByLabel("auto captions needing audio asrvid0001"), ).toBeVisible(); // --- Step 1: fetch the audio our engine transcribes from ------------------ await page .getByLabel("retry auto-captions audio bucket") .getByRole("button", { name: /^Fetch audio \(1\)$/ }) .click(); const downloadLog = page.getByLabel("Retry auto-captions audio output"); // The VTT must NOT prefilter the video away as "already downloaded". await expect(downloadLog).toContainText("Prefilter: 1 missing destination", { timeout: 60_000, }); await expect(downloadLog).toContainText("Managed download complete", { timeout: 60_000, }); expect(await pathExists(dataRel("asrvid0001", "audio.mp3"))).toBe(true); // --- Step 2: transcribe over the auto-captions ---------------------------- await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "transcribe")); await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0001"] }); await page.reload(); await page .getByRole("button", { name: /^Replace auto-captions \(1\)$/ }) .click(); await expect(page.getByLabel("Replace auto-captions output")).toContainText( "1 succeeded", { timeout: 60_000 }, ); expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true); // normalizeTranscript regenerates the derived cues from the new transcript. expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe( true, ); // The superseded VTT is KEPT as a backup — nothing deletes it automatically. expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe( true, ); // --- Step 3: the backup shows up as purgeable inventory ------------------- await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "transcribe")); await expectBuckets({ supersededAutoSubs: ["asrvid0001"] }); await page.reload(); await page.goto(channelStage(SLUG, "cleanup")); const section = page.getByLabel("superseded auto captions section"); await expect( section.getByRole("heading", { name: "Superseded auto-captions (1)" }), ).toBeVisible(); // --- Step 4: purge, and only then does the VTT go ------------------------ await section .getByLabel("confirm purge superseded auto captions") .fill("purge"); await section .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ }) .click(); await expect( page.getByLabel("Purge superseded auto-captions output"), ).toContainText("Purged 1 superseded auto-caption file", { timeout: 60_000, }); expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe( false, ); expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true); expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe( true, ); // The human-captioned video was never touched at any step. expect(await pathExists(dataRel("manvid0001", "transcript.en.vtt"))).toBe( true, ); await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "transcribe")); await expectBuckets({}); }); test("never buckets or purges captions it can't prove are auto-generated", async ({ page, }) => { test.setTimeout(90_000); await resetData(); await seedChannel([ // Superseded ASR backup: the one thing the purge may remove. { id: "asrdone0001", captions: "asr", whisper: true }, // Manual captions alongside our transcript: not a backup, never purged. { id: "mandone0001", captions: "manual", whisper: true }, // Manual captions and no transcript of ours: not lane work either. { id: "manonly0001", captions: "manual" }, // Archived media: shielded from the purge exactly like the Clean-audio sweep. { id: "asrkeep0001", captions: "asr", whisper: true, doNotClean: true }, ]); await generateReport(page, SLUG); await page.goto(channelStage(SLUG, "transcribe")); // Only the unprotected ASR-plus-whisper video is listed as a backup, and no // manual-caption video appears in the work lane at all. await expectBuckets({ supersededAutoSubs: ["asrdone0001"] }); await page.reload(); await page.goto(channelStage(SLUG, "cleanup")); const section = page.getByLabel("superseded auto captions section"); await section .getByLabel("confirm purge superseded auto captions") .fill("purge"); await section .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ }) .click(); const log = page.getByLabel("Purge superseded auto-captions output"); await expect(log).toContainText("Purged 1 superseded auto-caption file", { timeout: 60_000, }); await expect(log).toContainText("Skipped asrkeep0001 (marked do not clean)"); await expect(log).toContainText( "Kept mandone0001/transcript.en.vtt (manual captions", ); expect(await pathExists(dataRel("asrdone0001", "transcript.en.vtt"))).toBe( false, ); expect(await pathExists(dataRel("mandone0001", "transcript.en.vtt"))).toBe( true, ); expect(await pathExists(dataRel("manonly0001", "transcript.en.vtt"))).toBe( true, ); expect(await pathExists(dataRel("asrkeep0001", "transcript.en.vtt"))).toBe( true, ); }); // One enabled local worker using the fake whisper engine (mirrors auto-queue.spec). const ONE_WORKER = [ { id: "w1", name: "W1", kind: "local", enabled: true, priority: 0, appId: "whisper-cpp", config: {}, }, ]; test("the auto-transcribe runner transcribes over auto-captions when opted in", async ({ page, request, }) => { test.setTimeout(120_000); await resetData(); await seedChannel([ // Already has its audio, so the transcribe runner can take it directly. { id: "asrvid0003", captions: "asr", audio: true }, // Manual captions + audio: the runner must never pick this one up. { id: "manvid0003", captions: "manual", audio: true }, ]); await generateReport(page, SLUG); await page.goto(`/channels/${SLUG}`); await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0003"] }); await writeSettings({ adminTitle: "Test Admin", maxTranscriptPageBytes: 8388608, sleepBetweenDownloadsSeconds: 0, minFreeDiskGB: 0, workers: ONE_WORKER, autoQueue: { transcription: { enabled: true, maxWorkers: 1, // The switch under test: without it the runner's default union never // reaches the auto-caption bucket. replaceAutoSubs: true, root: { id: "root", mode: "strict", children: [{ id: "leaf-all", match: { type: "all" } }] }, }, download: {}, }, }); const started = await request.post(`${baseUrl}/api/auto-queue/control`, { // Behind the ops token since release 19 (A3): it starts and stops lanes. headers: { authorization: "Bearer test-worker-token" }, data: { kind: "transcription", action: "start" }, }); expect(started.ok()).toBeTruthy(); try { await expect .poll( () => pathExists(dataRel("asrvid0003", "transcript.json")), { timeout: 60_000 }, ) .toBe(true); // The manual-caption video stays untouched no matter how long the runner idles. expect(await pathExists(dataRel("manvid0003", "transcript.json"))).toBe( false, ); } finally { await request.post(`${baseUrl}/api/auto-queue/control`, { // Behind the ops token since release 19 (A3): it starts and stops lanes. headers: { authorization: "Bearer test-worker-token" }, data: { kind: "transcription", action: "stop" }, }); } }); test("the auto-queue opt-in is off by default and persists when enabled", async ({ page, }) => { await resetData(); await seedChannel([{ id: "asrvid0002", captions: "asr", audio: true }]); await page.goto("/operations/transcription"); const optIn = page.getByLabel( "replace YouTube auto-captions for auto-transcription", ); await expect(optIn).not.toBeChecked(); // The opt-in bucket is also offered per-leaf, so a single channel can join the // lane without flipping the runner-wide switch. await page.getByRole("button", { name: "+ Channel rule" }).first().click(); await expect( page.getByRole("option", { name: "downloadedAutoSubsOnly" }), ).toHaveCount(1); await optIn.check(); await page.getByRole("button", { name: "Save policy" }).first().click(); await expect(page.getByRole("status").first()).toHaveText("Saved."); await expect .poll( async () => { const settings = await readJson<{ autoQueue?: { transcription?: { replaceAutoSubs?: boolean } }; }>("test-settings.json").catch(() => null); return settings?.autoQueue?.transcription?.replaceAutoSubs ?? null; }, { timeout: 10_000 }, ) .toBe(true); await page.reload(); await expect( page.getByLabel("replace YouTube auto-captions for auto-transcription"), ).toBeChecked(); });