Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 2db970950e401e0605ecd2e4ae139a248a60acaf
parent 888310546a552f16b0400a137270e03b3d36bea8
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  4 Aug 2026 22:34:38 -0400

Let the pulse notice a regenerated report

/api/pulse builds its change token from in-memory state only — it must never
walk the corpus — so it could not see that a snapshot file had been rewritten.
Reports regenerate on the snapshot scheduler's debounce, which lands AFTER the
job that triggered it reached its final status, so the pages whose counts come
from snapshots (/channels, the dashboard) could sit stale until some unrelated
thing changed.

The scheduler now bumps an integer when it rewrites a report and the pulse
includes it. Zero I/O, closes the gap.

channels-counts.spec.ts asserted those counts immediately after a download,
which was only ever true while the page counted from disk on every render. It
now polls with reload — the same pattern the channel page's post-mutation
assertions already use — and the helper says why.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/jobs/snapshotScheduler.ts | 12++++++++++++
Meditor/app/api/pulse/route.ts | 8++++++++
Meditor/e2e/channels-counts.spec.ts | 33+++++++++++++++++++++++++--------
3 files changed, 45 insertions(+), 8 deletions(-)

diff --git a/common/jobs/snapshotScheduler.ts b/common/jobs/snapshotScheduler.ts @@ -53,6 +53,16 @@ type SchedulerState = { timer: ReturnType<typeof setTimeout> | null; // When the current (uncleared) dirty batch started, for the max-wait cap. firstDirtyAt: number | null; + // Bumped every time a report is actually rewritten. + // + // The editor's /api/pulse change token is deliberately built from in-memory + // state only — it must never walk the corpus — so it cannot notice that a + // snapshot FILE changed on disk. Without this counter, a report regenerated + // on the debounce AFTER the triggering job already reached its final status + // would move nothing the pulse can see, and pages showing snapshot-derived + // counts (/channels, the dashboard) would sit stale until the next unrelated + // change. One integer closes that gap at zero cost. + generation: number; }; declare global { @@ -66,6 +76,7 @@ function getState(): SchedulerState { dirty: new Map(), timer: null, firstDirtyAt: null, + generation: 0, }; } return globalThis.__yttSnapshotScheduler__; @@ -149,6 +160,7 @@ async function fire(): Promise<void> { fn: async (onLog) => { onLog(`Regenerating report for ${slug}…`); const snap = await generateChannelSnapshot(paths, slug); + getState().generation++; const excluded = excludedDownloadIdSet(snap); const awaitingTranscription = excluded.size ? snap.buckets.downloadedNoTranscript.filter( diff --git a/editor/app/api/pulse/route.ts b/editor/app/api/pulse/route.ts @@ -105,6 +105,14 @@ function computeRev(): { rev: string; activeJobs: number; runningJobs: number; b for (const w of workers) parts.push(`w:${w.id}:${w.busy ? 1 : 0}`); } + // A report regenerated on the snapshot scheduler's debounce lands AFTER the + // job that triggered it has already reached its final status — so without + // this counter, the pages whose counts come from snapshots (/channels, the + // dashboard) would go stale until some unrelated thing changed. One integer, + // in memory, and it means "a report was rewritten" without this endpoint + // having to stat 65 files to find out. + parts.push(`snap:${globalThis.__yttSnapshotScheduler__?.generation ?? 0}`); + // Files the layout renders from. mtime only — neither is read here. parts.push(`s:${mtime(paths.settingsFile)}`); parts.push(`c:${mtime(paths.editorChangelogFile)}`); diff --git a/editor/e2e/channels-counts.spec.ts b/editor/e2e/channels-counts.spec.ts @@ -1,6 +1,29 @@ import { test, expect } from "@playwright/test"; import { resetData, generateReport } from "./helpers"; +// /channels counts come from each channel's last generated REPORT, not from a +// live disk walk — that swap is what took the page from ~4.5s to ~0.1s. The +// report is rewritten on the snapshot scheduler's debounce, so a count checked +// immediately after a job finishes can legitimately still be the old one. +// Reload until it lands rather than asserting once. (Same pattern the channel +// page's post-mutation assertions use.) +async function expectCountEventually( + page: import("@playwright/test").Page, + label: string, + text: string, +): Promise<void> { + await expect + .poll( + async () => { + await page.goto("/channels"); + return page.getByLabel(label).textContent(); + }, + { timeout: 30_000, intervals: [500, 1000, 2000] }, + ) + .toBe(text); +} + + test("shows zero/em-dash when no data exists", async ({ page }) => { await resetData("one-youtube-channel"); await page.goto("/channels"); @@ -49,10 +72,7 @@ test("counts update after store-playlist and download", async ({ page }) => { { timeout: 20_000 }, ); - await page.goto("/channels"); - await expect( - page.getByLabel("playlist count for test-pipeline"), - ).toHaveText("5"); + await expectCountEventually(page, "playlist count for test-pipeline", "5"); await expect( page.getByLabel("downloads count for test-pipeline"), ).toHaveText("0"); @@ -64,10 +84,7 @@ test("counts update after store-playlist and download", async ({ page }) => { page.getByLabel("Download from playlist output"), ).toContainText("Managed download complete", { timeout: 60_000 }); - await page.goto("/channels"); - await expect( - page.getByLabel("downloads count for test-pipeline"), - ).toHaveText("5"); + await expectCountEventually(page, "downloads count for test-pipeline", "5"); // YouTube channels: each downloaded video also counts as a transcript // because the .vtt is the destination file. await expect(