import { dirname, resolve, join } from "node:path"; import { fileURLToPath } from "node:url"; import { cp, mkdir, readFile, rm, stat, utimes, writeFile, } from "node:fs/promises"; import { expect, type Page } from "@playwright/test"; import { baseUrl } from "./baseUrl"; const here = dirname(fileURLToPath(import.meta.url)); const editorRoot = resolve(here, ".."); const fixturesRoot = resolve(editorRoot, "e2e", "fixtures"); const testTranscriptsDir = resolve(editorRoot, "test-transcripts"); const testSettingsFile = resolve(editorRoot, "test-settings.json"); const defaultTestSettingsFile = resolve( fixturesRoot, "test-settings.default.json", ); async function fileExists(p: string): Promise { try { await stat(p); return true; } catch (err) { // Only "it is not there" means false. Any other errno (EACCES, EIO, a // transient ELOOP over a symlinked fixture) is a real fault and must not // be laundered into a confident "gone" that a caller then acts on. if ((err as NodeJS.ErrnoException).code !== "ENOENT") throw err; return false; } } export async function resetData(fixtureName: string | null = null) { // Quiesce the server BEFORE touching the tree, not only after. The // registry/runner singletons live in the one Next server and outlive a spec: // a debounced snapshot regen armed by the previous test's action fires while // this function is copying the fixture back in, and snapshot generation calls // reconcileVideoDirs, which RENAMES a data dir whose name differs from its // canonical video id. Freshly copied fixture, renamed out from under the // spec that just asked for it. Invalidating first cancels that work; the // second call at the end of this function clears caches over the new tree. // // The same quiescing is why the rm below can usually complete, but maxRetries // stays: a runner mid-write can still re-create a file inside a directory the // recursive walk has just emptied, and the rmdir then fails with ENOTEMPTY. // Node retries the whole operation with linear backoff on exactly that errno // set (also EBUSY/EPERM). Observed as two unrelated-looking full-suite // failures at channel-work.spec and pipeline.spec:106; both pass in isolation. const src = fixtureName ? join(fixturesRoot, "test-transcripts", fixtureName) : null; if (src && !(await fileExists(src))) { throw new Error(`Fixture not found: ${fixtureName}`); } // THE COPY IS RETRIED ON EEXIST. fs.cp makes each directory with a plain // mkdir after finding it absent, so a write still landing from the previous // spec's work (a job log, the auto-queue state — anything that mkdirs under // the data root) can create the directory in between. Release 19's // start-mode gate run lost channel-work.spec:208 to exactly that ("EEXIST: // file already exists, mkdir '…/test-transcripts/channels'"), and an earlier // run no-subs-fallback.spec:114. Quiesce again, clear again, copy again. for (let attempt = 1; ; attempt++) { await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); await rm(testTranscriptsDir, { recursive: true, force: true, maxRetries: 10, retryDelay: 100, }); try { if (src) { await mkdir(testTranscriptsDir, { recursive: true }); await cp(src, testTranscriptsDir, { recursive: true }); } else { await mkdir(join(testTranscriptsDir, "channels"), { recursive: true }); } break; } catch (err) { if ((err as NodeJS.ErrnoException).code !== "EEXIST" || attempt >= 4) { throw err; } await new Promise((r) => setTimeout(r, 100 * attempt)); } } await rm(testSettingsFile, { force: true }); await cp(defaultTestSettingsFile, testSettingsFile); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } // A SPEC'S SETTINGS ARE THE FIXTURE'S PLUS WHAT THE SPEC NAMES. // // This used to write the file WHOLESALE, which meant every key in // fixtures/test-settings.default.json — copied in by resetData() moments earlier // — was gone the instant a spec called this. The keys did not fall back to // something neutral: they fell back to the PRODUCT defaults, which is a // different fixture, chosen for operators and not for a test host. Four of them // diverge, and each one makes a spec depend on something it never mentions: // // minFreeDiskGB 0 vs 5 GB — arms the low-disk gate, // so every media action in the spec depends on how much room the HOST has // left, and the refusal is a returned `{ ok: false }` from // pipelineActions.lowDiskError() that no assertion reads. // sleepBetweenDownloadsSeconds 0 vs 10 s — 10 s between every // download in a fixture batch, which is the difference between a spec that // finishes and one that times out on a slow host. // verifyAvailabilityBeforeClean false vs true — an extra source probe on // the cleanup path. // syncScheduler.fullSweepIntervalMinutes 0 vs 1440 — whether a never-swept // channel is due for a full sweep or the cheap paged walk. // // Fourteen spec files call this without naming the sleep; eighteen without // naming the floor. Merging is what makes "a spec writes the settings it cares // about" true. EXPLICIT WINS at every level, which is what disk-space.spec.ts, // widget.spec.ts and backfill.spec.ts's "the disk floor refuses to re-acquire // anything" rely on. // // ONE LEVEL DEEP, and no deeper. `syncScheduler` is the only nested block the // fixture sets, and a spec that names it (scheduler.spec.ts, cadence-ui.spec.ts) // names the whole scheduler except that one key. Arrays and every other object // are REPLACED, not merged — `workers: []` has to mean no workers, and a // half-merged policy tree would be a worse surprise than a replaced one. type SettingsPatch = Record; function isPlainObject(v: unknown): v is SettingsPatch { return typeof v === "object" && v !== null && !Array.isArray(v); } export async function writeSettings(settings: SettingsPatch) { const base = JSON.parse( await readFile(defaultTestSettingsFile, "utf8"), ) as SettingsPatch; const merged: SettingsPatch = { ...base, ...settings }; for (const [key, value] of Object.entries(settings)) { if (isPlainObject(value) && isPlainObject(base[key])) { merged[key] = { ...base[key], ...value }; } } await writeFile(testSettingsFile, JSON.stringify(merged, null, 2)); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } // Write a site config under test-transcripts/sites//site.json (the // default SITES_DIR derived from TRANSCRIPTS_DIR in tests). Pass partial fields; // sensible defaults fill the rest. export async function writeSite( siteId: string, site: Record = {}, ) { const dir = join(testTranscriptsDir, "sites", siteId); await mkdir(dir, { recursive: true }); const full = { siteId, siteTitle: site.siteTitle ?? siteId, siteDescription: site.siteDescription ?? "", headerTitle: site.headerTitle ?? site.siteTitle ?? siteId, homeTagline: site.homeTagline ?? "", // Omit the key entirely when unspecified so the site inherits the global // default; pass an explicit array (incl. []) to override. ...("socialLinks" in site ? { socialLinks: site.socialLinks } : {}), groups: site.groups ?? [ { id: "default", name: "All channels", selectedByDefault: true }, ], defaultGroupId: site.defaultGroupId ?? "default", channels: site.channels ?? [], ...(site.cloudflareProject ? { cloudflareProject: site.cloudflareProject } : {}), ...(site.siteUrl ? { siteUrl: site.siteUrl } : {}), ...(site.listed === false ? { listed: false } : {}), ...(site.search === false ? { search: false } : {}), ...(site.audience !== undefined ? { audience: site.audience } : {}), // The legacy report-only key, written as given (siteSchema migrates it). ...(site.publish !== undefined ? { publish: site.publish } : {}), ...(site.relatedSites ? { relatedSites: site.relatedSites } : {}), }; await writeFile( join(dir, "site.json"), JSON.stringify(full, null, 2), ); await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } export async function readJson(relPath: string): Promise { const full = resolve(editorRoot, relPath); const raw = await readFile(full, "utf8"); return JSON.parse(raw) as T; } export async function pathExists(relPath: string): Promise { return fileExists(resolve(editorRoot, relPath)); } export function resolvePath(relPath: string): string { return resolve(editorRoot, relPath); } // Generate a channel's report, the way a user does. // // This used to be a SIDE EFFECT of viewing the channel page: with no // snapshot.json present the page ran a full channel analysis inline, inside a // GET. On a large channel that is a multi-minute page load caused by nothing // more than clicking a link, so the page now shows a "No report yet" state with // a button instead. Specs that need a report ask for one. // // Idempotent: if a report already exists the page renders normally and this // returns immediately. export async function generateReport( page: Page, slug: string, ): Promise { const snapshotRel = `test-transcripts/channels/${slug}/snapshot.json`; if (await pathExists(snapshotRel)) return; await page.goto(`/channels/${slug}`); const button = page.getByRole("button", { name: /refresh report/i }); // A social (posts) channel has no report and no button — it short-circuits // the whole video pipeline view. Nothing to generate, so say so by leaving. const offered = await button .waitFor({ state: "visible", timeout: 10_000 }) .then(() => true) .catch(() => false); if (!offered) return; // The click is RETRIED, deliberately. A click that lands before React has // hydrated fires nothing at all — no request, no job, no error — which is the // long-standing flake pattern in this suite (see the openChart helper). The // action is idempotent (it regenerates the same report), so clicking twice is // harmless and clicking zero times is not. for (let attempt = 0; attempt < 5; attempt++) { await button.click({ timeout: 5_000 }).catch(() => {}); for (let i = 0; i < 30; i++) { if (await pathExists(snapshotRel)) return; await new Promise((r) => setTimeout(r, 100)); } } throw new Error(`generateReport: no snapshot appeared for "${slug}"`); } export async function copyFixture(name: string) { const dst = join(fixturesRoot, "test-transcripts", name); await rm(dst, { recursive: true, force: true }); await mkdir(dst, { recursive: true }); await cp(testTranscriptsDir, dst, { recursive: true }); } // --------------------------------------------------------------------------- // Jobs table // --------------------------------------------------------------------------- // A /jobs row selected by its MACHINE kind. // // The kind column renders jobKindLabel(kind) (common/jobs/jobKinds.ts), not the // kind itself, so `getByRole("row").filter({ hasText: "whisper-all" })` matches // nothing once a kind gains a label — silently, and only for kinds that have // one. That drift is exactly what made 9 specs permanently red: "whisper-all" // renders "Transcribe all", "sync" renders "Sync", "retry-bucket" renders // "Retry", while label-less kinds like "check-availability" fall back to the raw // string and kept passing. Match the data-kind attribute instead; it is the // kind, so it cannot drift with the copy. export function jobRowByKind( page: import("@playwright/test").Page, kind: string, ): import("@playwright/test").Locator { return page.locator(`tr[data-kind="${kind}"]`); } // --------------------------------------------------------------------------- // Digest fixtures // --------------------------------------------------------------------------- // Write a video that the digest lane will actually accept. // // digestVideo refuses to run unless transcript.cues.json is FRESH — at least as // new as both metadata.info.json and the raw transcript (isCuesJsonFresh) — so a // digest never describes text that is about to be rewritten. Copying a fixture // tree can't guarantee that ordering, because `cp` stamps every file with the // time of the copy and the resulting order is whatever the walk produced. So the // mtimes are set EXPLICITLY here: the sources are backdated and cues.json is // left at "now". Without this the whole digest suite fails intermittently with // "stale-cues", which reads like a product bug and isn't one. export async function writeDigestVideo(opts: { channelSlug: string; videoId: string; title?: string; // Total video length in seconds. Cues are laid down every 5s across it. durationSeconds?: number; channelName?: string; // Shift every cue by this many seconds while keeping the TEXT identical — a // mirror with a longer intro. The alignment gate must refuse to share a digest // onto one of these: the content matches, so a text-only check would pass it, // and every shared chapter would then land at the wrong moment while the // artifact looked perfectly healthy. startOffsetSeconds?: number; // Write the VTT and metadata but NOT transcript.cues.json — the real state of // 1,942 videos on the live corpus, and one nothing produces automatically: a // `handling: "youtube"` channel fetches subtitles with --skip-download and so // never reaches transcribeOne, the only automatic caller of the normalizer. // Such a video is transcribed, published correctly (buildIndex re-parses the // raw VTT when the sidecar is absent) and permanently undigestable. skipCuesJson?: boolean; }) { const { channelSlug, videoId, title = "Synthetic Digest Video", durationSeconds = 600, channelName = channelSlug, startOffsetSeconds = 0, skipCuesJson = false, } = opts; const dir = join(testTranscriptsDir, "channels", channelSlug, "data", videoId); await mkdir(dir, { recursive: true }); // Every cue's words are GLOBALLY UNIQUE. measureAlignment anchors on 8-word // n-grams that occur exactly once on each side, so the obvious fixture — the // same sentence in every cue with only the number changed — yields no unique // anchors at all and reports "too-few-anchors" instead of the offset result a // sharing test is actually trying to observe. const cues = []; let i = 0; for (let t = 0; t + 5 <= durationSeconds; t += 5, i++) { const words = [ "segment", "filing", "deadline", "schedule", "ruling", "hearing", "motion", "brief", "docket", "counsel", "exhibit", "transcript", ].map((w) => `${w}${i}`); cues.push({ start: t + startOffsetSeconds, end: t + 5 + startOffsetSeconds, text: words.join(" "), }); } const meta = { id: videoId, title, channel: channelName, channel_id: `UC${channelSlug}`, uploader: channelName, upload_date: "20240101", duration: durationSeconds + startOffsetSeconds, description: "", is_live: false, was_live: false, live_status: "not_live", age_limit: 0, extractor_key: "Youtube", webpage_url: `https://www.youtube.com/watch?v=${videoId}`, }; const vtt = [ "WEBVTT", "Kind: captions", "Language: en", "", ...cues.flatMap((c) => [ `${vttStamp(c.start)} --> ${vttStamp(c.end)}`, c.text, "", ]), ].join("\n"); const detail = { version: 2, source: "vtt", transcriptFormat: "vtt", id: videoId, slug: `${channelSlug}/${videoId}`, channelSlug, channel: channelName, title, uploadDate: "20240101", duration: durationSeconds + startOffsetSeconds, isLivestream: false, cues, }; const metaPath = join(dir, "metadata.info.json"); const vttPath = join(dir, "transcript.en.vtt"); const cuesPath = join(dir, "transcript.cues.json"); await writeFile(metaPath, JSON.stringify(meta, null, 2)); await writeFile(vttPath, vtt); if (!skipCuesJson) await writeFile(cuesPath, JSON.stringify(detail)); const older = new Date(Date.now() - 60_000); await utimes(metaPath, older, older); await utimes(vttPath, older, older); return { dir, cuesPath, cues }; } function vttStamp(seconds: number): string { const n = Math.max(0, Math.floor(seconds)); const h = String(Math.floor(n / 3600)).padStart(2, "0"); const m = String(Math.floor((n % 3600) / 60)).padStart(2, "0"); const s = String(n % 60).padStart(2, "0"); return `${h}:${m}:${s}.000`; } // ── Channel route helpers ──────────────────────────────────────────────────── // // The channel page shows ONE stage panel at a time, selected by `?stage=`, and // the video browser lives on its own route. Both are URLs a spec has to build, // so they are built here rather than a hundred times inline. export type ChannelStage = | "configure" | "playlist" | "download" | "transcribe" | "digest" | "speakers" | "cleanup" | "diagnostics" | "storage" | "danger"; // The channel overview with one stage panel open. Landing here is what the old // `#stage-` hash plus a click on the section header used to do; the panel is // server-rendered open, so a spec navigates and interacts — there is no expand // step to perform first. export function channelStage(slug: string, stage: ChannelStage): string { return `/channels/${slug}?stage=${stage}`; } // The video workspace: list on the left, selected video on the right. Carries // the same `?video=` / `?filter=` / `?q=` parameters the browser has always // used, so filter and selection semantics survive the move to its own route. export function channelVideos( slug: string, params: { video?: string; filter?: string; q?: string } = {}, ): string { const qs = new URLSearchParams(); if (params.filter) qs.set("filter", params.filter); if (params.q) qs.set("q", params.q); if (params.video) qs.set("video", params.video); const s = qs.toString(); return s ? `/channels/${slug}/videos?${s}` : `/channels/${slug}/videos`; } // Minimal channel config so the channel page renders and the batch can run. export async function writeChannelConfig( channelSlug: string, config: Record = {}, ) { const dir = join(testTranscriptsDir, "channels", channelSlug); await mkdir(dir, { recursive: true }); await writeFile( join(dir, "config.json"), JSON.stringify( { handling: config.handling ?? "youtube", name: config.name ?? channelSlug, url: config.url ?? "https://www.youtube.com/@example/videos", ...config, }, null, 2, ), ); } // Run the corpus-wide Build index from the family page, the way a user does: // open the Pool disclosure, press the button, wait for the stage to end. The // controls sit under
so /sites does not open on six job consoles, // and getByRole ignores what a closed disclosure hides — hence the click first. // goto() resets the disclosure, so calling this twice in one test is fine. // // Build index is the `update-index` publish stage since release 18: the LMDB // index, the stats datasets and the chart templates in one child process, so // the wait is for the STAGE's last line (`[stage] update-index _index: Done`), // not for the index build's own "Done in …" halfway through it — a spec reads // the stats right after. export async function buildIndex(page: Page, timeout = 60_000) { await page.goto("/sites"); await page.getByText("Pool jobs", { exact: true }).click(); await page.getByRole("button", { name: "Build index" }).click(); await expect(page.getByLabel("Build index output")).toContainText( /\[stage\] update-index _index: Done/, { timeout }, ); }