Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b6f8d1355b78bdbaafce18b3153353ce0c9affb8
parent 35015500e43d1b37691e6edfe950efd9cb4d6418
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  1 Oct 2026 16:58:32 -0400

editor(e2e): dashboard-answers.spec — / and /jobs answer while a 2,000-video report regenerates; a ghost running meta is closed as interrupted and does not block a move

/api/test/settle-running-metas runs the boot pass over running metas on demand
(guarded by E2E_TEST_ROUTES like the others): the e2e server boots once per
suite.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Aeditor/app/api/test/settle-running-metas/route.ts | 26++++++++++++++++++++++++++
Aeditor/e2e/dashboard-answers.spec.ts | 275+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 301 insertions(+), 0 deletions(-)

diff --git a/editor/app/api/test/settle-running-metas/route.ts b/editor/app/api/test/settle-running-metas/route.ts @@ -0,0 +1,26 @@ +import { NextResponse } from "next/server"; +import { settleRunningJobMetas } from "yt-dlp-transcript-common/jobs/bootQueuedJobs"; +import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { testRouteDenied } from "../_guard"; + +export const dynamic = "force-dynamic"; + +// E2E test harness only, GUARDED BY `E2E_TEST_ROUTES` like every other +// /api/test route (see _guard.ts). It runs the boot pass over `running` metas +// (common/jobs/bootQueuedJobs.ts `settleRunningJobMetas`) NOW, the way +// editor/instrumentation.ts runs it once at boot: the e2e server boots once +// for the whole suite, so a spec that plants a ghost meta (a dead process's +// `running` job) has no other way to watch the pass close it. "This boot" is +// this instant: every meta from before it, not in the registry, whose writer +// is gone. +export async function POST() { + const denied = testRouteDenied(); + if (denied) return denied; + const result = await settleRunningJobMetas({ + paths: getPaths(), + bootedAt: Date.now(), + isLive: (id) => getRegistry().get(id) !== undefined, + }); + return NextResponse.json(result); +} diff --git a/editor/e2e/dashboard-answers.spec.ts b/editor/e2e/dashboard-answers.spec.ts @@ -0,0 +1,275 @@ +import { link, mkdir, readFile, writeFile } from "node:fs/promises"; +import { join } from "node:path"; +import { test, expect, type APIRequestContext, type Page } from "@playwright/test"; +import { ulid } from "yt-dlp-transcript-common/jobs/ulid"; +import { baseUrl } from "./baseUrl"; +import { + channelStage, + generateReport, + resetData, + resolvePath, +} from "./helpers"; + +// THE DASHBOARD ANSWERS WHILE A REPORT REGENERATES (release 17 slice D0). +// +// On 2026-10-01 the live editor's `/`, `/channels` and `/jobs` gave no +// response for over an hour while two `refresh-report` jobs walked 2,000- and +// 3,260-video channels in its own process, side by side, and three more +// `refresh-report` metas still read `running` hours after the process that ran +// them was gone. The first case below is the first half: a regeneration big +// enough to take several seconds, and `/` and `/jobs` polled every 2 s +// throughout — every answer inside 5 s. The second is the ghost: a `running` +// meta a dead process left behind is closed as interrupted by the boot pass, +// one a live process owns is not, and neither stops a media move. + +const OPS_AUTH = { authorization: "Bearer test-worker-token" }; + +// THE BIG CHANNEL. Every video dir hardlinks ONE ~400 KB metadata.info.json — +// the size of a long VOD's, the file the walk parses — so 2,000 of them cost +// one file's bytes and a few seconds to make. No `webpage_url`: the reconcile +// pass at the walk's start parses each file for it and, finding none, renames +// nothing (one shared id would merge every dir into one). The archive names a +// non-YouTube extractor, so the walk parses each file a second time for its +// native id — the Rumble-channel path. +const BIG = "big-channel"; +const BIG_VIDEOS = 2000; + +async function seedBigChannel(): Promise<void> { + const channelDir = resolvePath(`test-transcripts/channels/${BIG}`); + const dataDir = join(channelDir, "data"); + await mkdir(dataDir, { recursive: true }); + await writeFile( + join(channelDir, "config.json"), + JSON.stringify({ handling: "youtube", name: "Big Channel" }), + ); + const formats = Array.from({ length: 400 }, (_, i) => ({ + format_id: `hls-${i}`, + url: `https://example.invalid/${"x".repeat(600)}${i}`, + ext: "mp4", + protocol: "m3u8_native", + width: 1280, + height: 720, + tbr: 1234.5 + i, + http_headers: { "User-Agent": `Mozilla/5.0 ${"y".repeat(80)}`, Accept: "*/*" }, + fragments: [ + { url: "seg0", duration: 6 }, + { url: "seg1", duration: 6 }, + ], + })); + const template = resolvePath("test-transcripts/.big-channel-meta.json"); + await writeFile( + template, + JSON.stringify({ id: "native", title: "A long VOD", duration: 3600, formats }), + ); + const ids = Array.from( + { length: BIG_VIDEOS }, + (_, i) => `big${String(i).padStart(8, "0")}`, + ); + for (let i = 0; i < ids.length; i += 100) { + await Promise.all( + ids.slice(i, i + 100).map(async (id) => { + const dir = join(dataDir, id); + await mkdir(dir); + await link(template, join(dir, "metadata.info.json")); + }), + ); + } + await writeFile( + join(channelDir, "archive"), + ids.map((id) => `rumble ${id}\n`).join(""), + ); + await writeFile(join(channelDir, "playlist"), ""); +} + +type Sample = { path: string; ms: number; status: number }; + +async function timedGet( + request: APIRequestContext, + path: string, +): Promise<Sample> { + const t = Date.now(); + const res = await request.get(`${baseUrl}${path}`, { timeout: 60_000 }); + // The body too: a page that sends its head and stalls is not an answer. + await res.body(); + return { path, ms: Date.now() - t, status: res.status() }; +} + +test("/ and /jobs answer within 5 s while a large report regenerates", async ({ + request, +}) => { + test.setTimeout(300_000); + await resetData("empty"); + await seedBigChannel(); + + // Warm both routes first: under `next dev` the first request compiles the + // page, which is the dev server's cost and not what this measures. + for (const path of ["/", "/jobs"]) { + expect((await timedGet(request, path)).status).toBe(200); + } + + // The regeneration goes in the way "Update all reports" sends it: a + // refresh-report job on the serial queue. The ops call answers when it has + // finished, so it is the clock for "while it regenerates". + const startedAt = Date.now(); + let finishedAt: number | null = null; + const refresh = request + .post(`${baseUrl}/api/ops/refresh-report`, { + headers: OPS_AUTH, + data: { all: true }, + timeout: 280_000, + }) + .then(async (res) => { + finishedAt = Date.now(); + return { status: res.status(), body: (await res.json()) as { queued?: string[] } }; + }); + + const during: Sample[] = []; + while (finishedAt === null) { + const tick = Date.now(); + for (const path of ["/", "/jobs"]) { + const s = await timedGet(request, path); + if (finishedAt === null) during.push(s); + } + const wait = 2_000 - (Date.now() - tick); + if (wait > 0 && finishedAt === null) { + await Promise.race([new Promise((r) => setTimeout(r, wait)), refresh]); + } + } + const done = await refresh; + expect(done.status).toBe(200); + expect(done.body.queued).toEqual([BIG]); + const snapshot = JSON.parse( + await readFile( + resolvePath(`test-transcripts/channels/${BIG}/snapshot.json`), + "utf8", + ), + ) as { totals: { videos: number } }; + expect(snapshot.totals.videos).toBe(BIG_VIDEOS); + + // The fixture has to make the walk long enough to be polled through, or the + // case below proves nothing. + const regenMs = (finishedAt ?? Date.now()) - startedAt; + test.info().annotations.push({ + type: "timings", + description: `regeneration ${regenMs} ms; ${during + .map((s) => `${s.path} ${s.ms}`) + .join(", ")}`, + }); + expect(regenMs, "the regeneration took several seconds").toBeGreaterThan(4_000); + for (const path of ["/", "/jobs"]) { + expect( + during.filter((s) => s.path === path).length, + `${path} was polled during the regeneration`, + ).toBeGreaterThanOrEqual(2); + } + for (const s of during) { + expect(s.status, s.path).toBe(200); + expect(s.ms, `${s.path} answered in ${s.ms} ms`).toBeLessThan(5_000); + } +}); + +// --------------------------------------------------------------------------- +// The ghost. + +const SLUG = "test-youtube"; + +async function quiet(page: Page): Promise<void> { + await expect + .poll( + async () => { + const res = await page.request.get(`${baseUrl}/api/jobs/active`); + const body = await res.json(); + const jobs: { channelSlug?: string; status: string }[] = Array.isArray( + body, + ) + ? body + : (body.jobs ?? []); + return jobs.filter( + (j) => + j.channelSlug === SLUG && + (j.status === "running" || j.status === "queued"), + ).length; + }, + { timeout: 30_000 }, + ) + .toBe(0); +} + +// A `running` refresh-report meta for SLUG, as a process that died mid-walk +// leaves it: queued and started an hour ago, its log last written then. +async function plantGhost(pid: number): Promise<string> { + const id = ulid(Date.now() - 60 * 60 * 1000); + const jobsDir = resolvePath("test-transcripts/.jobs"); + await mkdir(jobsDir, { recursive: true }); + const at = Date.now() - 60 * 60 * 1000; + await writeFile( + join(jobsDir, `${id}.meta.json`), + JSON.stringify({ + id, + kind: "refresh-report", + queueKey: "", + channelSlug: SLUG, + status: "running", + queuedAt: at, + startedAt: at, + pid, + }), + ); + await writeFile( + join(jobsDir, `${id}.log`), + `Regenerating report for ${SLUG}…\n`, + ); + return id; +} + +async function metaOf(id: string): Promise<{ status: string; cancelReason?: string }> { + return JSON.parse( + await readFile(resolvePath(`test-transcripts/.jobs/${id}.meta.json`), "utf8"), + ); +} + +test("a ghost running meta from a dead process is closed as interrupted, and does not block a move", async ({ + page, +}, testInfo) => { + test.setTimeout(120_000); + await resetData("one-youtube-channel-with-data"); + await generateReport(page, SLUG); + await quiet(page); + + // Past pid_max: no process has it. And this test runner's own pid: a live + // process that is not the editor — an `archilyzer run` beside it. + const ghost = await plantGhost(2 ** 22 + 1); + const live = await plantGhost(process.pid); + + const res = await page.request.post(`${baseUrl}/api/test/settle-running-metas`); + expect(res.ok()).toBe(true); + const { interrupted } = (await res.json()) as { interrupted: { id: string }[] }; + expect(interrupted.map((j) => j.id)).toContain(ghost); + expect(interrupted.map((j) => j.id)).not.toContain(live); + + const closed = await metaOf(ghost); + expect(closed.status).toBe("cancelled"); + expect(closed.cancelReason).toMatch(/^interrupted: /); + expect((await metaOf(live)).status).toBe("running"); + + // /jobs says why, on the job's own page. + await page.goto(`/jobs/${ghost}`); + await expect(page.getByTestId("cancel-reason")).toContainText("interrupted"); + + // Neither the closed ghost nor the live one is a writer on the channel: the + // Storage panel offers the move and the move completes. + const root = testInfo.outputPath("ghost-root"); + await mkdir(root, { recursive: true }); + await page.goto(channelStage(SLUG, "storage")); + await page.getByLabel("destination root").fill(root); + await page.getByRole("button", { name: "Preview", exact: true }).click(); + await expect(page.getByLabel("relocation preview")).toBeVisible({ + timeout: 15_000, + }); + const moveButton = page.getByRole("button", { name: "Move media" }); + await expect(moveButton).toBeEnabled(); + await moveButton.click(); + await expect(page.getByLabel("Move media output")).toContainText("Moved", { + timeout: 60_000, + }); +});