// A transcript that covers only a small fraction of the video's duration means // the audio download silently truncated (yt-dlp exited "ok" but only a few // minutes landed, so whisper transcribed only those). The editor flags any // non-livestream video >=10min whose transcript covers <50% of its runtime: // an "Incomplete transcript" list filter + amber glyph, a warning banner on the // video page with a re-download button, and a transcription-page section. // Detection // reads each video's transcript.cues.json (duration + cues). See // common/lib/transcriptCoverage.ts. import { mkdir, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { baseUrl } from "./baseUrl"; import { channelVideos, generateReport, pathExists, readJson, resetData, resolvePath, } from "./helpers"; const CHANNEL = "test-transcribe"; const DATA = `test-transcripts/channels/${CHANNEL}/data`; // Non-empty whisper transcript so the video reads as transcribed (and isn't // flagged untranscribable for being empty). const WHISPER = JSON.stringify({ transcription: [{ text: "hello" }] }); // Write a transcribed video with a transcript.cues.json of a given duration and // last-cue end. version:1 is required for readNormalizedTranscript to parse it. async function writeVideo( id: string, opts: { duration: number; lastCueEnd: number; isLivestream?: boolean; empty?: boolean; }, ) { const dir = resolvePath(`${DATA}/${id}`); await mkdir(dir, { recursive: true }); await writeFile( `${dir}/transcript.json`, opts.empty ? JSON.stringify({ transcription: [] }) : WHISPER, ); await writeFile( `${dir}/audio.m4a`, // Tiny placeholder; the truncated-audio detail doesn't matter to the test. "fake-audio", ); await writeFile( `${dir}/transcript.cues.json`, JSON.stringify({ version: 1, source: "whisper", duration: opts.duration, isLivestream: Boolean(opts.isLivestream), cues: opts.empty ? [] : [{ start: 0, end: opts.lastCueEnd, text: "hello" }], }), ); } async function seed() { await resetData("one-transcribe-channel"); // Flagged: 2h22m video, transcript stops at ~6:52 → 4.8% coverage. await writeVideo("vidTrunc", { duration: 8541, lastCueEnd: 412 }); // Not flagged: full coverage of a >=10min video. await writeVideo("vidFull", { duration: 1200, lastCueEnd: 1180 }); // Not flagged: low coverage but under the 10min minimum. await writeVideo("vidShort", { duration: 300, lastCueEnd: 10 }); // Not flagged: low coverage but a livestream (duration unreliable). await writeVideo("vidLive", { duration: 8541, lastCueEnd: 100, isLivestream: true, }); } test("incomplete-transcript filter, glyph, panel banner, and the transcription page", async ({ page, }) => { await seed(); // --- Channel list: the chip filters to exactly the flagged video --- await generateReport(page, CHANNEL); await page.goto(channelVideos(CHANNEL)); const list = page.getByLabel("videos", { exact: true }); await expect(list.getByLabel("open vidTrunc")).toBeVisible(); await page .getByRole("button", { name: "Incomplete transcript", exact: true }) .click(); await expect(list.getByLabel("open vidTrunc")).toBeVisible(); await expect(list.getByLabel("open vidFull")).toBeHidden(); await expect(list.getByLabel("open vidShort")).toBeHidden(); await expect(list.getByLabel("open vidLive")).toBeHidden(); // URL carries the filter. await expect .poll(() => new URL(page.url()).searchParams.get("filter")) .toBe("incomplete_transcript"); // Composes with Transcribed (the flagged video is still transcribed). await page.getByRole("button", { name: "Transcribed", exact: true }).click(); await expect(list.getByLabel("open vidTrunc")).toBeVisible(); await expect(list.getByLabel(/^open vid/)).toHaveCount(1); // The amber "truncated" glyph is shown for the flagged row. await expect(page.getByTitle(/transcript truncated/)).toBeVisible(); // --- Video page: warning banner + re-download button on the flagged video --- await page.goto(`/channels/${CHANNEL}/videos/vidTrunc`); const banner = page.getByLabel("incomplete transcript"); await expect(banner).toBeVisible(); // Coverage line: covers 6:52 of 2:22:21 (4.8%). await expect(banner).toContainText("2:22:21"); await expect(banner).toContainText("4.8%"); await expect( banner.getByRole("button", { name: /Re-download & re-transcribe/ }), ).toBeVisible(); // No banner on a fully-covered video. await page.goto(`/channels/${CHANNEL}/videos/vidFull`); await expect(page.getByLabel("incomplete transcript")).toBeHidden(); // --- Actionable view lists the channel under incomplete transcripts --- await page.goto(`/operations/transcription`); const section = page.getByRole("region", { name: "incomplete-transcripts", exact: true, }); await expect(section).toBeVisible(); await expect( section.getByLabel(`incomplete-transcripts row ${CHANNEL}`), ).toBeVisible(); }); // The fix run re-downloads the full audio and re-transcribes, so the transcript // is no longer truncated and the banner's own condition is gone by the time the // run ends. StreamActionLog fires that refresh itself, so a parent that stops // rendering the banner on it destroys the run's log — the
// included. The warning text disappearing is the proof the refresh landed; the // log is read only after that. See plans/FACTS.md, "A run log lives in the // panel's React state". // // Two things about the setup, both forced: // // - Its own channel, not this file's shared `test-transcribe`. This is the // only case here that runs a JOB, and a finished job arms the debounced // snapshot regen (snapshotScheduler). Landing late, that regen writes // channels//snapshot.json after the NEXT test's resetData has copied // its fixture in — and generateReport RETURNS EARLY when a snapshot file // exists (helpers.ts:146), so that test would then read a snapshot built // before its own seed and find no flagged video. Measured: with this case // sharing the slug, "clear incomplete" below failed 1 run in 10 on // "Select incomplete" never appearing; the baseline without this case is // 50/50. A separate slug puts the late write somewhere nobody reads. // - A truncated VTT rather than the whisper transcript the other cases seed: // transcribeOneVideo returns "already-exists" the moment a transcript.json // is on disk (common/controller/transcribeOne.ts:107), so with one there the // fix run re-downloads the audio, writes no new transcript, and nothing // regenerates transcript.cues.json — the flag would never clear. Not this // test's subject. test("the fix run's log survives the refresh that clears the banner", async ({ page, }) => { // See saved-videos.spec.ts: sequential waits, so give the case its own budget // rather than let the 30 s default mask which assertion failed. test.setTimeout(120_000); await resetData(); const slug = "incomplete-log"; const id = "vidVttTrunc"; const root = resolvePath(`test-transcripts/channels/${slug}`); const dir = `${root}/data/${id}`; await mkdir(dir, { recursive: true }); await writeFile( `${root}/config.json`, JSON.stringify({ handling: "transcribe", name: slug, platform: "youtube", url: `https://www.youtube.com/@${slug}`, audioFormat: "m4a", }), ); // fixIncompleteTranscriptOne resolves the source URL from metadata.info.json // or the playlist, and there is no metadata here yet. await writeFile( `${root}/playlist`, `https://www.youtube.com/watch?v=${id}\n`, ); await writeFile( `${dir}/transcript.en.vtt`, "WEBVTT\n\n00:00:00.000 --> 00:06:52.000\nhello\n", ); await writeFile(`${dir}/audio.m4a`, "fake-audio"); await writeFile( `${dir}/transcript.cues.json`, JSON.stringify({ version: 1, source: "vtt", duration: 8541, isLivestream: false, cues: [{ start: 0, end: 412, text: "hello" }], }), ); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); await page.goto(`/channels/${slug}/videos/${id}`); const banner = page.getByLabel("incomplete transcript"); await expect(banner).toContainText("Transcript looks truncated"); const run = banner.getByRole("button", { name: /Re-download & re-transcribe/, }); await expect(run).toBeEnabled(); await run.click(); const log = page.getByLabel(`Re-download & re-transcribe ${id} output`); await expect(log).toContainText(`Re-downloading audio for ${id}`, { timeout: 15_000, }); // The state the run produces: a full transcript over the re-downloaded audio, // so the truncation warning is gone. await expect(page.getByText("Transcript looks truncated")).toHaveCount(0, { timeout: 15_000, }); // And the log is STILL on screen, carrying the run's closing line. await expect(log).toContainText(`Normalized ${slug}/${id}`); // Kept for its log, not for a second run: the transcript is no longer // truncated. (Page-scoped, not banner-scoped — the wrapper drops the alert // label with the warning it belonged to.) await expect( page.getByRole("button", { name: "Re-download & re-transcribe" }), ).toBeDisabled(); }); test("channel bulk bar: clear incomplete resets the video and enables auto-runners", async ({ page, }) => { await seed(); await generateReport(page, CHANNEL); await page.goto(channelVideos(CHANNEL)); // Select the flagged video via the new quick-select, then clear it. await page .getByRole("button", { name: "Select incomplete", exact: true }) .click(); const bar = page.getByLabel("bulk action bar"); await expect(bar).toBeVisible(); await page.getByLabel("bulk action", { exact: true }).selectOption("clear_incomplete"); page.once("dialog", (d) => d.accept()); await page.getByLabel("apply bulk action").click(); // Selection clears on success → the bar hides. await expect(bar).toBeHidden(); // The truncated audio + transcript are gone on disk. await expect .poll(() => pathExists(`${DATA}/vidTrunc/audio.m4a`)) .toBe(false); await expect .poll(() => pathExists(`${DATA}/vidTrunc/transcript.json`)) .toBe(false); await expect .poll(() => pathExists(`${DATA}/vidTrunc/transcript.cues.json`)) .toBe(false); // The auto-download + auto-transcribe runners are now enabled. const settings = await readJson<{ autoQueue: { transcription: { enabled: boolean }; download: { enabled: boolean }; }; }>("test-settings.json"); expect(settings.autoQueue.transcription.enabled).toBe(true); expect(settings.autoQueue.download.enabled).toBe(true); // The video is no longer flagged and now reads as undownloaded ("No audio"). // The channel snapshot regenerates on a ~1s debounce after the clear, and the // channel page serves the persisted snapshot, so re-navigate until it's fresh. const list = page.getByLabel("videos", { exact: true }); await expect .poll( async () => { await page.goto(channelVideos(CHANNEL, { filter: "incomplete_transcript" })); return list.getByLabel("open vidTrunc").count(); }, { timeout: 15000 }, ) .toBe(0); await page.goto(channelVideos(CHANNEL, { filter: "no_audio" })); await expect(list.getByLabel("open vidTrunc")).toBeVisible(); }); test("channel bulk bar: re-download & re-transcribe queues a batch fix", async ({ page, }) => { await seed(); await generateReport(page, CHANNEL); await page.goto(channelVideos(CHANNEL)); await page .getByRole("button", { name: "Select incomplete", exact: true }) .click(); const bar = page.getByLabel("bulk action bar"); await expect(bar).toBeVisible(); await page.getByLabel("bulk action", { exact: true }).selectOption("redownload_incomplete"); await page.getByLabel("apply bulk action").click(); // A streaming bulk action clears the selection on success (the batch job runs // in the background) → the bar hides with no error surfaced. await expect(bar).toBeHidden(); await expect(page.getByLabel("bulk action error")).toBeHidden(); }); test("transcription page: section exposes per-channel + global fix buttons; per-channel re-download queues a job", async ({ page, }) => { await seed(); // Visiting the channel materializes its snapshot so the section lists it. await generateReport(page, CHANNEL); await page.goto(`/channels/${CHANNEL}`); await page.goto(`/operations/transcription`); const section = page.getByRole("region", { name: "incomplete-transcripts", exact: true, }); await expect(section).toBeVisible(); // Per-channel and global buttons are all present. await expect( section.getByLabel(`re-download & re-transcribe ${CHANNEL}`), ).toBeVisible(); await expect( section.getByLabel(`clear & re-queue ${CHANNEL}`), ).toBeVisible(); await expect( section.getByLabel("re-download all incomplete transcripts"), ).toBeVisible(); await expect( section.getByLabel("clear all incomplete transcripts"), ).toBeVisible(); // Per-channel re-download queues a job (the job id is returned synchronously). await section.getByLabel(`re-download & re-transcribe ${CHANNEL}`).click(); await expect( section.getByLabel(`re-download & re-transcribe ${CHANNEL} job`), ).toBeVisible(); }); test("transcription page: global clear-all clears every flagged video and empties the section", async ({ page, }) => { await seed(); // Visiting the channel materializes its snapshot so the section lists it. await generateReport(page, CHANNEL); await page.goto(`/channels/${CHANNEL}`); await page.goto(`/operations/transcription`); const section = page.getByRole("region", { name: "incomplete-transcripts", exact: true, }); await expect(section).toBeVisible(); // Global clear-all clears every flagged video across all channels (synchronous // fs op — no background job to race the assertions below). page.once("dialog", (d) => d.accept()); await section.getByLabel("clear all incomplete transcripts").click(); await expect( section.getByLabel("fix all incomplete transcripts result"), ).toContainText(/Cleared 1/); await expect .poll(() => pathExists(`${DATA}/vidTrunc/transcript.cues.json`)) .toBe(false); // The snapshot regenerates on a ~1s debounce after the clear; the section // reads the persisted snapshot, so re-navigate until it is empty. await expect .poll( async () => { await page.goto(`/operations/transcription`); return page.getByLabel("incomplete-transcripts empty").count(); }, { timeout: 15000 }, ) .toBe(1); });