Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b64af6e282408a71af4082f9c0605e58a6d20b8b
parent 597669b68bd1ae676c384c4c88f91e59d2c09fbb
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  8 Sep 2026 01:05:09 -0400

editor: the bucket lanes' entries are pinned, and the bands are pinned against them

The twin of the digest-entry test beside it: half one asserts
backfill.download and backfill.transcription really land, with the ids in
bucket-union order and missingInput on transcription alone; half two renders
the /channels Download and Transcribe cells, strips the two entries back out of
the snapshot the way every un-regenerated live one has them, and requires the
cells to read the same. Invariance in one run, not numbers copied from an
older build.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Meditor/e2e/backfill.spec.ts | 109+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 109 insertions(+), 0 deletions(-)

diff --git a/editor/e2e/backfill.spec.ts b/editor/e2e/backfill.spec.ts @@ -1247,3 +1247,112 @@ test("a digest entry in the snapshot does not move the backfill instrument", asy body.backfill.needsMedia, ); }); + +// (O) THE SAME REGRESSION, ONE SLICE LATER — and the same two halves. +// +// Slice 1.5 gave the two BUCKET lanes a snapshot entry too: +// `backfill.download` and `backfill.transcription` are the fold of each lane's +// default buckets, so `snapshot.backfill[op].ids` is where every lane's work +// list lives. Six operations now write into a map four surfaces once summed +// generically. +// +// The /channels pipeline columns are the surface with the most to lose, because +// they draw a Download and a Transcribe band and could plausibly be "fixed" to +// read the new entries. They must not: the band's transcription `reachable` is +// downloadedNoTranscript ALONE while the lane's work list also carries the +// retry bucket (1,873 videos against 881 on the live corpus), and the band's +// `blocked` is what the entry calls `missingInput`. +// +// So this asserts invariance directly rather than against numbers copied out of +// an older build: render the row with the entries present, strip them back out +// of the snapshot exactly as every un-regenerated live snapshot has them, and +// require the two cells to read the same both times. +test("the bucket lanes' work lists land in the snapshot and move no band", async ({ + page, +}) => { + test.setTimeout(SLOW); + await resetData("one-transcribe-channel-with-audio"); + await writeSettings(backfillSettings()); + await rm(resolvePath(`test-transcripts/channels/${SLUG}/snapshot.json`), { + force: true, + }); + await generateReport(page, SLUG); + + // HALF ONE: the entries are really there. Without this the invariance below + // would pass vacuously over a feature that never shipped. + const snapRel = `test-transcripts/channels/${SLUG}/snapshot.json`; + type Snap = { + backfill?: Record< + string, + { missing?: number; ids?: string[]; missingInput?: number } + >; + buckets?: Record<string, string[] | undefined>; + undownloadedIds?: string[]; + }; + const snapshot = await readJson<Snap>(snapRel); + const download = snapshot.backfill?.download; + const transcription = snapshot.backfill?.transcription; + expect(download, "snapshot.backfill.download must exist").toBeTruthy(); + expect(transcription, "snapshot.backfill.transcription must exist").toBeTruthy(); + // The fold, restated from the buckets in the same file: partialDownloads then + // undownloadedIds, downloadedNoTranscript then failedListed, deduped, in that + // order. Asserting the LIST and not just a count is the point — the download + // lane's order is playlist order and sorting it would reorder the queue. + const union = (names: string[]): string[] => { + const out: string[] = []; + for (const name of names) { + const ids = + name === "undownloadedIds" + ? (snapshot.undownloadedIds ?? []) + : (snapshot.buckets?.[name] ?? []); + for (const id of ids) if (!out.includes(id)) out.push(id); + } + return out; + }; + expect(download?.ids).toEqual(union(["partialDownloads", "undownloadedIds"])); + expect(transcription?.ids).toEqual( + union(["downloadedNoTranscript", "failedListed"]), + ); + expect(download?.missing).toBe(download?.ids?.length); + expect(transcription?.missing).toBe(transcription?.ids?.length); + // Download's input is the channel listing, which is never missing. A video + // with no audio is the TRANSCRIPTION lane's missing input. + expect(download?.missingInput).toBe(0); + expect(transcription?.missingInput).toBe( + snapshot.buckets?.noTranscript?.length ?? 0, + ); + // The opt-in auto-captions bucket is a POLICY switch, never folded into a + // corpus fact. + for (const id of snapshot.buckets?.downloadedAutoSubsOnly ?? []) { + expect(transcription?.ids).not.toContain(id); + } + + // HALF TWO: the two bands read the same with the entries and without them. + const cells = async (): Promise<string[]> => { + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); + await page.goto("/channels"); + const out: string[] = []; + for (const label of ["downloads count", "transcripts count"]) { + const cell = page.getByLabel(`${label} for ${SLUG}`); + await expect(cell).toBeVisible(); + out.push( + `${label}=${(await cell.textContent()) ?? ""}|${await cell.getAttribute("title")}`, + ); + } + return out; + }; + const withEntries = await cells(); + + // Every live snapshot looks like this today: the slice regenerates nothing, + // so the bands and the runner both have to read a file with no entry. + const stripped = JSON.parse( + await readFile(resolvePath(snapRel), "utf8"), + ) as Snap; + delete stripped.backfill?.download; + delete stripped.backfill?.transcription; + await writeFile( + resolvePath(snapRel), + JSON.stringify(stripped, null, 2) + "\n", + ); + expect(await cells()).toEqual(withEntries); +});