Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit e61db76c8e9092f84bac8d0c72e9b38e272ee894
parent 3b95a7bb8ef0efa2197766a949d1b1e56dc5fdc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 26 Sep 2026 16:50:18 -0400

e2e: persist-youtube-handling.spec (Persist source video on a youtube-handling video with a transcript downloads, persists, keeps the transcript, extracts nothing; a changed upstream title + view_count lands in metadata.history.json and on the page) and the full fetch on the same fixture ends done WITH a file; the fake's .fake-ytdlp-metadata.json knob

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Meditor/e2e/fetch-window.spec.ts | 62+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 18++++++++++++++++++
Aeditor/e2e/persist-youtube-handling.spec.ts | 175+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
3 files changed, 254 insertions(+), 1 deletion(-)

diff --git a/editor/e2e/fetch-window.spec.ts b/editor/e2e/fetch-window.spec.ts @@ -9,10 +9,11 @@ // Same token as /api/worker/* (the test server runs with // WORKER_TOKEN=test-worker-token; see package.json dev:test). -import { readFile, stat } from "node:fs/promises"; +import { readdir, readFile, stat } from "node:fs/promises"; import { test, expect, type APIRequestContext } from "@playwright/test"; import { generateReport, + readJson, resetData, resolvePath, writeChannelConfig, @@ -395,3 +396,62 @@ test("a full-source fetch lands in the saved-video store and is cached on a seco expect(Number(cached.bytes)).toBe(Number(pointer.bytes)); expect(await invocations()).toBe(invBefore); }); + +// THE SAME ASK ON THE CHANNEL AS IT IS (release 10 slice N). The test above +// switches the channel to transcribe handling, because that used to be the only +// way the container reached the store: on this subtitles-first channel the one +// media pass (the no-subs fallback) was skipped for any video with a +// transcript, so the job ended `done` with no file — the silent no-op slice M's +// live proof found. `forceMedia` opens that pass; with a transcript on disk it +// persists the container and extracts nothing. +test("a full-source fetch on a youtube-handling video that already has a transcript downloads anyway", async ({ + request, +}) => { + test.setTimeout(90_000); + // The fixture's downloaded video: metadata.info.json + transcript.en.vtt. + const HAS_TRANSCRIPT = "fake00000001"; + const post = await request.post(`${baseUrl}/api/media/fetch-window`, { + headers: AUTH, + data: { + channelSlug: SLUG, + videoId: HAS_TRANSCRIPT, + full: true, + requestedBy: "mcp", + manifest: "demo-report", + reason: "the whole recording, captions or not", + }, + }); + expect(post.status(), await post.text()).toBe(202); + const { jobId } = (await post.json()) as { jobId: string }; + + const finished = await pollJob(request, jobId); + expect(finished.status, JSON.stringify(finished)).toBe("done"); + // A FILE, which is the whole difference. + expect(String(finished.file)).toMatch(/source-media\./); + expect(Number(finished.bytes)).toBeGreaterThan(0); + + const dir = resolvePath(rel(`data/${HAS_TRANSCRIPT}`)); + const files = await readdir(dir); + expect(files).toContain("transcript.en.vtt"); + expect(files).toContain("saved-video.json"); + // Nothing extracted beside the transcript. + expect(files.filter((f) => /^(audio|source-media)\./.test(f))).toEqual([]); + + // AND THE METADATA REWRITE IS ON RECORD, with who asked. The fixture's own + // info json differs from what the fake's prefetch writes (another title), so + // this fetch is a rewrite that changed content. + const history = await readJson<{ + entries: Array<{ + by: string; + requestedBy?: string; + changed: Record<string, { from: unknown; to: unknown }>; + }>; + }>(rel(`data/${HAS_TRANSCRIPT}/metadata.history.json`)); + expect(history.entries).toHaveLength(1); + expect(history.entries[0].by).toBe("prefetch"); + expect(history.entries[0].requestedBy).toBe("mcp"); + expect(history.entries[0].changed.title).toEqual({ + from: "Synthetic Test Video 1", + to: `Synthetic ${HAS_TRANSCRIPT}`, + }); +}); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -132,12 +132,30 @@ async function writeMetadata(videoDir, id, opts = {}) { meta.was_live = true; meta.live_status = "was_live"; } + Object.assign(meta, await readMetadataOverrides()); await writeFile( path.join(videoDir, "metadata.info.json"), JSON.stringify(meta), ); } +// THE SOURCE CHANGED ITS METADATA. A spec that needs the second fetch of a +// video to differ from the first (the metadata history, release 10 slice N) +// writes `.fake-ytdlp-metadata.json` in the channel root (the fake's cwd): a +// JSON object whose keys are laid over every info json this fake writes from +// then on — `{ "title": "Renamed", "view_count": 150 }`. Absent (every other +// spec), nothing changes and the output stays byte-stable. +async function readMetadataOverrides() { + try { + const parsed = JSON.parse(await readFile(".fake-ytdlp-metadata.json", "utf8")); + return parsed && typeof parsed === "object" && !Array.isArray(parsed) + ? parsed + : {}; + } catch { + return {}; + } +} + // Sentinels encoded in a video id/URL select metadata variants so a fixture // can deterministically exercise each filter/fallback branch. function urlSentinels(url) { diff --git a/editor/e2e/persist-youtube-handling.spec.ts b/editor/e2e/persist-youtube-handling.spec.ts @@ -0,0 +1,175 @@ +// THE WHOLE-RECORDING FETCH DOWNLOADS ANYWAY, AND EVERY METADATA REWRITE KEEPS +// THE OLD VERSION (release 10 slice N). +// +// On a youtube-handling channel the only pass that fetches media is the no-subs +// fallback, and its gate is "no transcript and no captions". So "Persist source +// video" (and `fetch_clip` with `full: true`, the same function) on a video that +// already had a transcript ran two subtitle passes and produced no file — the +// live proof of slice M found it on teamrcn. No spec covered it: every persist +// spec used a transcribe-handling fixture. This one uses the youtube fixture's +// already-downloaded video, transcript and all. +// +// The second test is the history: two downloads of one video where the source's +// metadata changed in between (the fake's `.fake-ytdlp-metadata.json` knob), +// and the video page saying so. + +import { readFile, readdir, stat, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { readJson, resetData, resolvePath, writeSettings } from "./helpers"; +import { baseUrl } from "./baseUrl"; + +const SLUG = "test-youtube"; +const rel = (p: string) => `test-transcripts/channels/${SLUG}/${p}`; +// The fixture's downloaded video: metadata.info.json + transcript.en.vtt. +const HAS_TRANSCRIPT = "fake00000001"; +// In the playlist, never downloaded. +const UNDOWNLOADED = "fake00000002"; + +type Outcome = { + status: string; + startedAt: string; + fellBackToTranscribe?: boolean; + attempts: Array<{ kind: string; ytdlpExitCode: number | null }>; +}; + +async function outcomeOf(id: string): Promise<Outcome | null> { + try { + return await readJson<Outcome>(rel(`data/${id}/download-outcome.json`)); + } catch { + return null; + } +} + +test("Persist source video on a youtube-handling video with a transcript downloads the source and leaves the transcript alone", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("youtube-with-playlist"); + // Inline whisper ON, so the claim "nothing but the container" bites: had the + // forced pass extracted audio and fallen through to transcription, the + // whisper transcript below would have been rewritten. + await writeSettings({ inlineTranscribeOnFallback: true }); + const whisper = JSON.stringify({ + transcription: [{ text: "the transcript already on disk" }], + }); + const whisperPath = resolvePath(rel(`data/${HAS_TRANSCRIPT}/transcript.json`)); + await writeFile(whisperPath, whisper); + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); + + await page.goto(`/channels/${SLUG}/videos/${HAS_TRANSCRIPT}`); + await page.getByLabel("Source video stage summary").click(); + // The card used to say persistence was for transcribe-handling channels only + // and offered no button here at all. + const run = page.getByRole("button", { + name: "Persist source video", + exact: true, + }); + await expect(run).toBeEnabled(); + await run.click(); + + const log = page.getByLabel(`Persist source video for ${HAS_TRANSCRIPT} output`); + await expect(log).toContainText( + "forceMedia: downloading the source although a transcript is on disk", + { timeout: 30_000 }, + ); + await expect(page.getByLabel(`unpersist source video ${HAS_TRANSCRIPT}`)).toBeVisible({ + timeout: 30_000, + }); + await expect(log).toContainText("Persisted source video to"); + + // yt-dlp was asked for MEDIA: the fake reaches its source-media branch only + // for an invocation without --skip-download. + const inv = await readFile(resolvePath(rel("fake-ytdlp.invocations")), "utf8"); + expect(inv).toContain( + `download-source-media:https://www.youtube.com/watch?v=${HAS_TRANSCRIPT}`, + ); + + // The pointer landed, naming a real file in the store. + const pointer = await readJson<{ + dir: string; + file: string; + bytes: number; + keepReason?: string; + }>(rel(`data/${HAS_TRANSCRIPT}/saved-video.json`)); + expect(pointer.keepReason).toBe("override"); + expect(pointer.file).toMatch(/^source-media\./); + expect((await stat(`${pointer.dir}/${pointer.file}`)).size).toBe(pointer.bytes); + + // The transcript is byte-for-byte what it was, and nothing was extracted + // beside it: no audio, no container left in the data dir. + expect(await readFile(whisperPath, "utf8")).toBe(whisper); + const files = await readdir(resolvePath(rel(`data/${HAS_TRANSCRIPT}`))); + expect(files.filter((f) => /^(audio|source-media)\./.test(f))).toEqual([]); + + // A clean download, and not a fallback to transcription. + const outcome = await outcomeOf(HAS_TRANSCRIPT); + expect(outcome?.status).toBe("ok"); + expect(outcome?.fellBackToTranscribe).toBeUndefined(); + expect(outcome?.attempts.map((a) => a.kind)).toEqual([ + "metadata-prefetch", + "primary", + "no-subs-fallback", + ]); +}); + +test("a download whose metadata changed upstream appends to metadata.history.json, and the video page says so", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData("youtube-with-playlist"); + const knob = resolvePath(rel(".fake-ytdlp-metadata.json")); + const historyRel = rel(`data/${UNDOWNLOADED}/metadata.history.json`); + + // First download: a first write of metadata.info.json is not a rewrite. + await writeFile(knob, JSON.stringify({ view_count: 100 })); + await page.goto(`/channels/${SLUG}/videos/${UNDOWNLOADED}`); + await page.getByRole("button", { name: /^Run download pipeline$/ }).click(); + await expect.poll(() => outcomeOf(UNDOWNLOADED), { timeout: 30_000 }).not.toBeNull(); + const first = (await outcomeOf(UNDOWNLOADED))!; + expect(first.status).toBe("ok"); + await expect(readJson(historyRel)).rejects.toThrow(); + + // The source renames the video and it gains views; download again. + await writeFile( + knob, + JSON.stringify({ + title: `Synthetic ${UNDOWNLOADED} (renamed upstream)`, + view_count: 150, + }), + ); + await page.reload(); + await page.getByRole("button", { name: /^Run download pipeline$/ }).click(); + await expect + .poll(async () => (await outcomeOf(UNDOWNLOADED))?.startedAt, { + timeout: 30_000, + }) + .not.toBe(first.startedAt); + + type History = { + entries: Array<{ + by: string; + changed: Record<string, { from: unknown; to: unknown }>; + counters: Record<string, [unknown, unknown]>; + }>; + }; + const history = await readJson<History>(historyRel); + expect(history.entries).toHaveLength(1); + expect(history.entries[0].by).toBe("prefetch"); + expect(history.entries[0].changed).toEqual({ + title: { + from: `Synthetic ${UNDOWNLOADED}`, + to: `Synthetic ${UNDOWNLOADED} (renamed upstream)`, + }, + }); + expect(history.entries[0].counters).toEqual({ view_count: [100, 150] }); + + // The page: one line, and the entry behind it with its from → to. + await page.goto(`/channels/${SLUG}/videos/${UNDOWNLOADED}`); + const line = page.getByText(/^Metadata rewritten 1× · last .* by prefetch: title$/); + await expect(line).toBeVisible(); + await line.click(); + const entries = page.getByLabel("metadata history entries"); + await expect(entries).toContainText(`Synthetic ${UNDOWNLOADED} (renamed upstream)`); + await expect(entries).toContainText("view_count"); + await expect(entries).toContainText("100 → 150"); +});