Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit d66e39798d1b5a36c1f98cb90d4af88de54c3345
parent b6f8d1355b78bdbaafce18b3153353ce0c9affb8
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  1 Oct 2026 17:39:36 -0400

editor(e2e): dashboard-answers — two 600-video channels regenerated one after the other, the budget floored on the idle page time

2,000 videos did not finish inside five minutes under next dev at a load
average of 30. Two channels prove the serial queue from the jobs' records (the
second starts when the first has ended); each answer is held to 5 s or three
times the slowest idle answer measured just before, whichever is longer.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Meditor/e2e/dashboard-answers.spec.ts | 139+++++++++++++++++++++++++++++++++++++++++++++++++++++++------------------------
1 file changed, 98 insertions(+), 41 deletions(-)

diff --git a/editor/e2e/dashboard-answers.spec.ts b/editor/e2e/dashboard-answers.spec.ts @@ -16,26 +16,37 @@ import { // response for over an hour while two `refresh-report` jobs walked 2,000- and // 3,260-video channels in its own process, side by side, and three more // `refresh-report` metas still read `running` hours after the process that ran -// them was gone. The first case below is the first half: a regeneration big -// enough to take several seconds, and `/` and `/jobs` polled every 2 s -// throughout — every answer inside 5 s. The second is the ghost: a `running` -// meta a dead process left behind is closed as interrupted by the boot pass, -// one a live process owns is not, and neither stops a media move. +// them was gone. The first case below is the first half: two channels whose +// reports take several seconds each, regenerated by "Update all reports" — one +// after the other, never side by side — with `/` and `/jobs` polled every 2 s +// throughout. The second is the ghost: a `running` meta a dead process left +// behind is closed as interrupted by the boot pass, one a live process owns is +// not, and neither stops a media move. +// +// THE BUDGET IS 5 s, OR THREE TIMES WHAT THE SAME PAGE TOOK JUST BEFORE THE +// REGENERATION, WHICHEVER IS LONGER. The suite runs under `next dev` on a +// machine other suites and builds share; at a load average of 30 a page that +// renders in 0.3 s on a quiet machine takes 2–4 s with nothing regenerating at +// all. What this pins is that the regeneration does not starve the pages, not +// how fast the machine is — so the idle measurement taken a moment before sets +// the floor, and on a quiet machine the budget is simply 5 s. const OPS_AUTH = { authorization: "Bearer test-worker-token" }; -// THE BIG CHANNEL. Every video dir hardlinks ONE ~400 KB metadata.info.json — -// the size of a long VOD's, the file the walk parses — so 2,000 of them cost -// one file's bytes and a few seconds to make. No `webpage_url`: the reconcile -// pass at the walk's start parses each file for it and, finding none, renames +// THE BIG CHANNELS. Every video dir hardlinks ONE ~400 KB metadata.info.json +// — the size of a long VOD's, the file the walk parses — so 600 of them cost +// one file's bytes and a second to make. No `webpage_url`: the reconcile pass +// at the walk's start parses each file for it and, finding none, renames // nothing (one shared id would merge every dir into one). The archive names a // non-YouTube extractor, so the walk parses each file a second time for its -// native id — the Rumble-channel path. -const BIG = "big-channel"; -const BIG_VIDEOS = 2000; +// native id — the Rumble-channel path. 600 is several seconds of walk under +// `next start` on a quiet machine and well over a minute under `next dev` at a +// load average of 30; 2,000 did not finish inside five minutes there. +const BIG = ["big-channel-a", "big-channel-b"]; +const BIG_VIDEOS = 600; -async function seedBigChannel(): Promise<void> { - const channelDir = resolvePath(`test-transcripts/channels/${BIG}`); +async function seedBigChannel(slug: string): Promise<void> { + const channelDir = resolvePath(`test-transcripts/channels/${slug}`); const dataDir = join(channelDir, "data"); await mkdir(dataDir, { recursive: true }); await writeFile( @@ -56,14 +67,14 @@ async function seedBigChannel(): Promise<void> { { url: "seg1", duration: 6 }, ], })); - const template = resolvePath("test-transcripts/.big-channel-meta.json"); + const template = resolvePath(`test-transcripts/.${slug}-meta.json`); await writeFile( template, JSON.stringify({ id: "native", title: "A long VOD", duration: 3600, formats }), ); const ids = Array.from( { length: BIG_VIDEOS }, - (_, i) => `big${String(i).padStart(8, "0")}`, + (_, i) => `${slug.slice(-1)}big${String(i).padStart(8, "0")}`, ); for (let i = 0; i < ids.length; i += 100) { await Promise.all( @@ -94,33 +105,51 @@ async function timedGet( return { path, ms: Date.now() - t, status: res.status() }; } -test("/ and /jobs answer within 5 s while a large report regenerates", async ({ +type JobMetaOnDisk = { + id: string; + kind: string; + channelSlug?: string; + status: string; + startedAt?: number; + endedAt?: number; +}; + +test("/ and /jobs answer while two large reports regenerate, one after the other", async ({ request, }) => { - test.setTimeout(300_000); + test.setTimeout(420_000); await resetData("empty"); - await seedBigChannel(); + for (const slug of BIG) await seedBigChannel(slug); // Warm both routes first: under `next dev` the first request compiles the - // page, which is the dev server's cost and not what this measures. + // page, which is the dev server's cost and not what this measures. Then the + // idle floor: the slowest of three more rounds, with nothing regenerating. for (const path of ["/", "/jobs"]) { expect((await timedGet(request, path)).status).toBe(200); } + const idle: Sample[] = []; + for (let i = 0; i < 3; i++) { + for (const path of ["/", "/jobs"]) idle.push(await timedGet(request, path)); + } + const budgetMs = Math.max(5_000, 3 * Math.max(...idle.map((s) => s.ms))); - // The regeneration goes in the way "Update all reports" sends it: a - // refresh-report job on the serial queue. The ops call answers when it has - // finished, so it is the clock for "while it regenerates". + // "Update all reports", through the ops API: a refresh-report job per + // channel on the serial queue. The call answers when both have finished, so + // it is the clock for "while they regenerate". const startedAt = Date.now(); let finishedAt: number | null = null; const refresh = request .post(`${baseUrl}/api/ops/refresh-report`, { headers: OPS_AUTH, data: { all: true }, - timeout: 280_000, + timeout: 400_000, }) .then(async (res) => { finishedAt = Date.now(); - return { status: res.status(), body: (await res.json()) as { queued?: string[] } }; + return { + status: res.status(), + body: (await res.json()) as { queued?: string[]; jobIds?: string[] }, + }; }); const during: Sample[] = []; @@ -128,7 +157,10 @@ test("/ and /jobs answer within 5 s while a large report regenerates", async ({ const tick = Date.now(); for (const path of ["/", "/jobs"]) { const s = await timedGet(request, path); - if (finishedAt === null) during.push(s); + if (finishedAt === null) { + during.push(s); + console.log(`[dashboard-answers] +${tick - startedAt} ms ${path} ${s.ms} ms`); + } } const wait = 2_000 - (Date.now() - tick); if (wait > 0 && finishedAt === null) { @@ -137,25 +169,50 @@ test("/ and /jobs answer within 5 s while a large report regenerates", async ({ } const done = await refresh; expect(done.status).toBe(200); - expect(done.body.queued).toEqual([BIG]); - const snapshot = JSON.parse( - await readFile( - resolvePath(`test-transcripts/channels/${BIG}/snapshot.json`), - "utf8", - ), - ) as { totals: { videos: number } }; - expect(snapshot.totals.videos).toBe(BIG_VIDEOS); + expect([...(done.body.queued ?? [])].sort()).toEqual(BIG); + for (const slug of BIG) { + const snapshot = JSON.parse( + await readFile( + resolvePath(`test-transcripts/channels/${slug}/snapshot.json`), + "utf8", + ), + ) as { totals: { videos: number } }; + expect(snapshot.totals.videos).toBe(BIG_VIDEOS); + } + + // ONE AFTER THE OTHER: the second started when the first had ended. Read + // off the jobs' records on disk; the terminal write lands just after the + // job's stream closes, so it is waited for. + const readMetas = () => + Promise.all( + (done.body.jobIds ?? []).map( + async (id) => + JSON.parse( + await readFile(resolvePath(`test-transcripts/.jobs/${id}.meta.json`), "utf8"), + ) as JobMetaOnDisk, + ), + ); + await expect + .poll(async () => (await readMetas()).map((m) => `${m.kind} ${m.status}`), { + timeout: 10_000, + }) + .toEqual(["refresh-report done", "refresh-report done"]); + const metas = await readMetas(); + const [first, second] = [...metas].sort( + (a, b) => (a.startedAt ?? 0) - (b.startedAt ?? 0), + ); + expect(second.startedAt ?? 0).toBeGreaterThanOrEqual(first.endedAt ?? Infinity); // The fixture has to make the walk long enough to be polled through, or the - // case below proves nothing. + // budget below proves nothing. const regenMs = (finishedAt ?? Date.now()) - startedAt; test.info().annotations.push({ type: "timings", - description: `regeneration ${regenMs} ms; ${during - .map((s) => `${s.path} ${s.ms}`) - .join(", ")}`, + description: + `idle max ${Math.max(...idle.map((s) => s.ms))} ms, budget ${budgetMs} ms; ` + + `regeneration ${regenMs} ms; ${during.map((s) => `${s.path} ${s.ms}`).join(", ")}`, }); - expect(regenMs, "the regeneration took several seconds").toBeGreaterThan(4_000); + expect(regenMs, "the regenerations took several seconds").toBeGreaterThan(4_000); for (const path of ["/", "/jobs"]) { expect( during.filter((s) => s.path === path).length, @@ -164,7 +221,7 @@ test("/ and /jobs answer within 5 s while a large report regenerates", async ({ } for (const s of during) { expect(s.status, s.path).toBe(200); - expect(s.ms, `${s.path} answered in ${s.ms} ms`).toBeLessThan(5_000); + expect(s.ms, `${s.path} answered in ${s.ms} ms (budget ${budgetMs} ms)`).toBeLessThan(budgetMs); } }); @@ -231,7 +288,7 @@ async function metaOf(id: string): Promise<{ status: string; cancelReason?: stri test("a ghost running meta from a dead process is closed as interrupted, and does not block a move", async ({ page, }, testInfo) => { - test.setTimeout(120_000); + test.setTimeout(180_000); await resetData("one-youtube-channel-with-data"); await generateReport(page, SLUG); await quiet(page);