commit e61db76c8e9092f84bac8d0c72e9b38e272ee894
parent 3b95a7bb8ef0efa2197766a949d1b1e56dc5fdc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 26 Sep 2026 16:50:18 -0400
e2e: persist-youtube-handling.spec (Persist source video on a youtube-handling video with a transcript downloads, persists, keeps the transcript, extracts nothing; a changed upstream title + view_count lands in metadata.history.json and on the page) and the full fetch on the same fixture ends done WITH a file; the fake's .fake-ytdlp-metadata.json knob
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 254 insertions(+), 1 deletion(-)
diff --git a/editor/e2e/fetch-window.spec.ts b/editor/e2e/fetch-window.spec.ts
@@ -9,10 +9,11 @@
// Same token as /api/worker/* (the test server runs with
// WORKER_TOKEN=test-worker-token; see package.json dev:test).
-import { readFile, stat } from "node:fs/promises";
+import { readdir, readFile, stat } from "node:fs/promises";
import { test, expect, type APIRequestContext } from "@playwright/test";
import {
generateReport,
+ readJson,
resetData,
resolvePath,
writeChannelConfig,
@@ -395,3 +396,62 @@ test("a full-source fetch lands in the saved-video store and is cached on a seco
expect(Number(cached.bytes)).toBe(Number(pointer.bytes));
expect(await invocations()).toBe(invBefore);
});
+
+// THE SAME ASK ON THE CHANNEL AS IT IS (release 10 slice N). The test above
+// switches the channel to transcribe handling, because that used to be the only
+// way the container reached the store: on this subtitles-first channel the one
+// media pass (the no-subs fallback) was skipped for any video with a
+// transcript, so the job ended `done` with no file — the silent no-op slice M's
+// live proof found. `forceMedia` opens that pass; with a transcript on disk it
+// persists the container and extracts nothing.
+test("a full-source fetch on a youtube-handling video that already has a transcript downloads anyway", async ({
+ request,
+}) => {
+ test.setTimeout(90_000);
+ // The fixture's downloaded video: metadata.info.json + transcript.en.vtt.
+ const HAS_TRANSCRIPT = "fake00000001";
+ const post = await request.post(`${baseUrl}/api/media/fetch-window`, {
+ headers: AUTH,
+ data: {
+ channelSlug: SLUG,
+ videoId: HAS_TRANSCRIPT,
+ full: true,
+ requestedBy: "mcp",
+ manifest: "demo-report",
+ reason: "the whole recording, captions or not",
+ },
+ });
+ expect(post.status(), await post.text()).toBe(202);
+ const { jobId } = (await post.json()) as { jobId: string };
+
+ const finished = await pollJob(request, jobId);
+ expect(finished.status, JSON.stringify(finished)).toBe("done");
+ // A FILE, which is the whole difference.
+ expect(String(finished.file)).toMatch(/source-media\./);
+ expect(Number(finished.bytes)).toBeGreaterThan(0);
+
+ const dir = resolvePath(rel(`data/${HAS_TRANSCRIPT}`));
+ const files = await readdir(dir);
+ expect(files).toContain("transcript.en.vtt");
+ expect(files).toContain("saved-video.json");
+ // Nothing extracted beside the transcript.
+ expect(files.filter((f) => /^(audio|source-media)\./.test(f))).toEqual([]);
+
+ // AND THE METADATA REWRITE IS ON RECORD, with who asked. The fixture's own
+ // info json differs from what the fake's prefetch writes (another title), so
+ // this fetch is a rewrite that changed content.
+ const history = await readJson<{
+ entries: Array<{
+ by: string;
+ requestedBy?: string;
+ changed: Record<string, { from: unknown; to: unknown }>;
+ }>;
+ }>(rel(`data/${HAS_TRANSCRIPT}/metadata.history.json`));
+ expect(history.entries).toHaveLength(1);
+ expect(history.entries[0].by).toBe("prefetch");
+ expect(history.entries[0].requestedBy).toBe("mcp");
+ expect(history.entries[0].changed.title).toEqual({
+ from: "Synthetic Test Video 1",
+ to: `Synthetic ${HAS_TRANSCRIPT}`,
+ });
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -132,12 +132,30 @@ async function writeMetadata(videoDir, id, opts = {}) {
meta.was_live = true;
meta.live_status = "was_live";
}
+ Object.assign(meta, await readMetadataOverrides());
await writeFile(
path.join(videoDir, "metadata.info.json"),
JSON.stringify(meta),
);
}
+// THE SOURCE CHANGED ITS METADATA. A spec that needs the second fetch of a
+// video to differ from the first (the metadata history, release 10 slice N)
+// writes `.fake-ytdlp-metadata.json` in the channel root (the fake's cwd): a
+// JSON object whose keys are laid over every info json this fake writes from
+// then on — `{ "title": "Renamed", "view_count": 150 }`. Absent (every other
+// spec), nothing changes and the output stays byte-stable.
+async function readMetadataOverrides() {
+ try {
+ const parsed = JSON.parse(await readFile(".fake-ytdlp-metadata.json", "utf8"));
+ return parsed && typeof parsed === "object" && !Array.isArray(parsed)
+ ? parsed
+ : {};
+ } catch {
+ return {};
+ }
+}
+
// Sentinels encoded in a video id/URL select metadata variants so a fixture
// can deterministically exercise each filter/fallback branch.
function urlSentinels(url) {
diff --git a/editor/e2e/persist-youtube-handling.spec.ts b/editor/e2e/persist-youtube-handling.spec.ts
@@ -0,0 +1,175 @@
+// THE WHOLE-RECORDING FETCH DOWNLOADS ANYWAY, AND EVERY METADATA REWRITE KEEPS
+// THE OLD VERSION (release 10 slice N).
+//
+// On a youtube-handling channel the only pass that fetches media is the no-subs
+// fallback, and its gate is "no transcript and no captions". So "Persist source
+// video" (and `fetch_clip` with `full: true`, the same function) on a video that
+// already had a transcript ran two subtitle passes and produced no file — the
+// live proof of slice M found it on teamrcn. No spec covered it: every persist
+// spec used a transcribe-handling fixture. This one uses the youtube fixture's
+// already-downloaded video, transcript and all.
+//
+// The second test is the history: two downloads of one video where the source's
+// metadata changed in between (the fake's `.fake-ytdlp-metadata.json` knob),
+// and the video page saying so.
+
+import { readFile, readdir, stat, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { readJson, resetData, resolvePath, writeSettings } from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+const SLUG = "test-youtube";
+const rel = (p: string) => `test-transcripts/channels/${SLUG}/${p}`;
+// The fixture's downloaded video: metadata.info.json + transcript.en.vtt.
+const HAS_TRANSCRIPT = "fake00000001";
+// In the playlist, never downloaded.
+const UNDOWNLOADED = "fake00000002";
+
+type Outcome = {
+ status: string;
+ startedAt: string;
+ fellBackToTranscribe?: boolean;
+ attempts: Array<{ kind: string; ytdlpExitCode: number | null }>;
+};
+
+async function outcomeOf(id: string): Promise<Outcome | null> {
+ try {
+ return await readJson<Outcome>(rel(`data/${id}/download-outcome.json`));
+ } catch {
+ return null;
+ }
+}
+
+test("Persist source video on a youtube-handling video with a transcript downloads the source and leaves the transcript alone", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("youtube-with-playlist");
+ // Inline whisper ON, so the claim "nothing but the container" bites: had the
+ // forced pass extracted audio and fallen through to transcription, the
+ // whisper transcript below would have been rewritten.
+ await writeSettings({ inlineTranscribeOnFallback: true });
+ const whisper = JSON.stringify({
+ transcription: [{ text: "the transcript already on disk" }],
+ });
+ const whisperPath = resolvePath(rel(`data/${HAS_TRANSCRIPT}/transcript.json`));
+ await writeFile(whisperPath, whisper);
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+
+ await page.goto(`/channels/${SLUG}/videos/${HAS_TRANSCRIPT}`);
+ await page.getByLabel("Source video stage summary").click();
+ // The card used to say persistence was for transcribe-handling channels only
+ // and offered no button here at all.
+ const run = page.getByRole("button", {
+ name: "Persist source video",
+ exact: true,
+ });
+ await expect(run).toBeEnabled();
+ await run.click();
+
+ const log = page.getByLabel(`Persist source video for ${HAS_TRANSCRIPT} output`);
+ await expect(log).toContainText(
+ "forceMedia: downloading the source although a transcript is on disk",
+ { timeout: 30_000 },
+ );
+ await expect(page.getByLabel(`unpersist source video ${HAS_TRANSCRIPT}`)).toBeVisible({
+ timeout: 30_000,
+ });
+ await expect(log).toContainText("Persisted source video to");
+
+ // yt-dlp was asked for MEDIA: the fake reaches its source-media branch only
+ // for an invocation without --skip-download.
+ const inv = await readFile(resolvePath(rel("fake-ytdlp.invocations")), "utf8");
+ expect(inv).toContain(
+ `download-source-media:https://www.youtube.com/watch?v=${HAS_TRANSCRIPT}`,
+ );
+
+ // The pointer landed, naming a real file in the store.
+ const pointer = await readJson<{
+ dir: string;
+ file: string;
+ bytes: number;
+ keepReason?: string;
+ }>(rel(`data/${HAS_TRANSCRIPT}/saved-video.json`));
+ expect(pointer.keepReason).toBe("override");
+ expect(pointer.file).toMatch(/^source-media\./);
+ expect((await stat(`${pointer.dir}/${pointer.file}`)).size).toBe(pointer.bytes);
+
+ // The transcript is byte-for-byte what it was, and nothing was extracted
+ // beside it: no audio, no container left in the data dir.
+ expect(await readFile(whisperPath, "utf8")).toBe(whisper);
+ const files = await readdir(resolvePath(rel(`data/${HAS_TRANSCRIPT}`)));
+ expect(files.filter((f) => /^(audio|source-media)\./.test(f))).toEqual([]);
+
+ // A clean download, and not a fallback to transcription.
+ const outcome = await outcomeOf(HAS_TRANSCRIPT);
+ expect(outcome?.status).toBe("ok");
+ expect(outcome?.fellBackToTranscribe).toBeUndefined();
+ expect(outcome?.attempts.map((a) => a.kind)).toEqual([
+ "metadata-prefetch",
+ "primary",
+ "no-subs-fallback",
+ ]);
+});
+
+test("a download whose metadata changed upstream appends to metadata.history.json, and the video page says so", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("youtube-with-playlist");
+ const knob = resolvePath(rel(".fake-ytdlp-metadata.json"));
+ const historyRel = rel(`data/${UNDOWNLOADED}/metadata.history.json`);
+
+ // First download: a first write of metadata.info.json is not a rewrite.
+ await writeFile(knob, JSON.stringify({ view_count: 100 }));
+ await page.goto(`/channels/${SLUG}/videos/${UNDOWNLOADED}`);
+ await page.getByRole("button", { name: /^Run download pipeline$/ }).click();
+ await expect.poll(() => outcomeOf(UNDOWNLOADED), { timeout: 30_000 }).not.toBeNull();
+ const first = (await outcomeOf(UNDOWNLOADED))!;
+ expect(first.status).toBe("ok");
+ await expect(readJson(historyRel)).rejects.toThrow();
+
+ // The source renames the video and it gains views; download again.
+ await writeFile(
+ knob,
+ JSON.stringify({
+ title: `Synthetic ${UNDOWNLOADED} (renamed upstream)`,
+ view_count: 150,
+ }),
+ );
+ await page.reload();
+ await page.getByRole("button", { name: /^Run download pipeline$/ }).click();
+ await expect
+ .poll(async () => (await outcomeOf(UNDOWNLOADED))?.startedAt, {
+ timeout: 30_000,
+ })
+ .not.toBe(first.startedAt);
+
+ type History = {
+ entries: Array<{
+ by: string;
+ changed: Record<string, { from: unknown; to: unknown }>;
+ counters: Record<string, [unknown, unknown]>;
+ }>;
+ };
+ const history = await readJson<History>(historyRel);
+ expect(history.entries).toHaveLength(1);
+ expect(history.entries[0].by).toBe("prefetch");
+ expect(history.entries[0].changed).toEqual({
+ title: {
+ from: `Synthetic ${UNDOWNLOADED}`,
+ to: `Synthetic ${UNDOWNLOADED} (renamed upstream)`,
+ },
+ });
+ expect(history.entries[0].counters).toEqual({ view_count: [100, 150] });
+
+ // The page: one line, and the entry behind it with its from → to.
+ await page.goto(`/channels/${SLUG}/videos/${UNDOWNLOADED}`);
+ const line = page.getByText(/^Metadata rewritten 1× · last .* by prefetch: title$/);
+ await expect(line).toBeVisible();
+ await line.click();
+ const entries = page.getByLabel("metadata history entries");
+ await expect(entries).toContainText(`Synthetic ${UNDOWNLOADED} (renamed upstream)`);
+ await expect(entries).toContainText("view_count");
+ await expect(entries).toContainText("100 → 150");
+});