import { rm, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { buildIndex, readJson, resetData, resolvePath, writeSite, } from "./helpers"; const CHANNEL = "test-youtube"; const VIDEO_DIR = "20240101_test1234567"; // The index keys a record by metadata.info.json's `id`, which the fixture now // keeps equal to its directory name (see that file — a mismatch made snapshot // generation rename the directory mid-spec). const VIDEO_ID = VIDEO_DIR; const DATA = `test-transcripts/channels/${CHANNEL}/data/${VIDEO_DIR}`; const TRANSCRIPTS_PAGE = `test-transcripts/.export-index/shared/transcripts/${CHANNEL}/page-0000.json`; // parseVtt only emits cues from lines carrying YouTube's inline word-timing // tags, so the body needs markers to produce a real cue. const VTT = "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\n" + "<00:00:00.000> hello<00:00:02.000> regional\n"; type Detail = { id: string; cues?: Array }; // The site has to exist before the index build runs; buildIndex() from helpers // is the shared "open the Pool, press the button, wait for Done" (the local name // would shadow the import, hence seedAndBuild). async function seedAndBuild(page: import("@playwright/test").Page) { await writeSite("testsite", { channels: [{ slug: CHANNEL, groupId: "default" }], }); await buildIndex(page); } // YouTube sometimes serves a video's English captions only under a regional or // auto code (transcript.en-US.vtt) with no plain transcript.en.vtt. buildIndex // should still parse it as the primary transcript via the English-VTT fallback, // so the video's shared transcript page carries cues. test("build index parses a video with only transcript.en-US.vtt", async ({ page, }) => { await resetData("one-youtube-channel-with-data"); // Replace the canonical transcript.en.vtt with a regional-only English track. await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); await writeFile(resolvePath(`${DATA}/transcript.en-US.vtt`), VTT); await seedAndBuild(page); const detailsPage = await readJson(TRANSCRIPTS_PAGE); const detail = detailsPage.find((d) => d.id === VIDEO_ID); expect(detail).toBeDefined(); // cues are written only when a primary transcript was recognized. expect(detail?.cues?.length ?? 0).toBeGreaterThan(0); }); // Regression guard: a translation track (transcript.es-en-US.vtt) is NOT English // and must not be mistaken for the primary transcript — the video lands in the // index (it has metadata) but carries no cues. test("build index ignores a non-English translation track", async ({ page }) => { await resetData("one-youtube-channel-with-data"); await rm(resolvePath(`${DATA}/transcript.en.vtt`), { force: true }); await writeFile(resolvePath(`${DATA}/transcript.es-en-US.vtt`), VTT); await seedAndBuild(page); const detailsPage = await readJson(TRANSCRIPTS_PAGE); const detail = detailsPage.find((d) => d.id === VIDEO_ID); expect(detail).toBeDefined(); expect(detail?.cues?.length ?? 0).toBe(0); });