Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 0a17900565c266621f0e141f66e6386923575717
parent 3c8562a91ac4aabfa6829a6846594381a69cce2c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 00:35:40 -0400

e2e: the scan, the settlement, and the proof that no video directory is made

`title-filter-channel` has two matching ids, two non-matching, and one the fake
refuses without cookies, so the scan has something it cannot read and the
snapshot has to say so without settling it.

The load-bearing assertion is the negative one: after a scan, `data/<id>/`
exists for none of the five. A directory with a metadata.info.json is admitted
to the LMDB index and the exported site by buildIndex, and its name is what
deriveChannelSets reads as "ever fetched" — so "the scan created no directory"
is the invariant the whole design rests on, and it is checked directly rather
than inferred from a count.

The other two that earn their place: a settled video is never PREFETCHED (the
invocations file, not just the absence of media — the cost being avoided is the
metadata request, once per non-match per run, forever), and flipping the regex
re-decides the channel with a byte-identical invocations file, which is what
"derived, not stored" means operationally.

fake-ytdlp gets a `--skip-download --print -a` branch, placed BEFORE the plain
`-a` branch that would otherwise swallow it and "download" the whole batch.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelSnapshot.ts | 18++++++++++++++++--
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 56++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist | 1+
Meditor/e2e/title-filter.spec.ts | 334++++++++++++++++++++++++++++++++++++++++++++++++-------------------------------
4 files changed, 276 insertions(+), 133 deletions(-)

diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -1035,7 +1035,14 @@ export async function generateChannelSnapshot( // rejection left a prefetched metadata.info.json behind for, which would // otherwise land in noTranscript AND noMetadata AND the retryable // skippedByFilter at once. - if (settledIds.has(id)) { + // + // ONLY WHEN THERE IS NOTHING ON DISK. A video that was downloaded before the + // filter was written (or before it was edited to reject it) is DOWNLOADED — + // that is a fact, not a preference — and short-circuiting it here would drop + // it from transcribedWithAudio and the cleanup estimates while it still + // counted in totals.transcribed, i.e. a channel reporting more transcripts + // than videos. Settlement only ever decides what NOT to fetch. + if (settledIds.has(id) && !videoHasAnyArtifact(files)) { settledOnDisk.push(id); continue; } @@ -1327,7 +1334,14 @@ export async function generateChannelSnapshot( corruptFullSource: corruptFullSource.sort(), nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), - skippedByTitleFilter: [...settledIds].sort(), + // The settled set minus anything already on disk — same rule as the + // short-circuit above, so the bucket and the classification cannot disagree. + skippedByTitleFilter: [...settledIds] + .filter((id) => { + const f = filesById.get(id); + return !f || !videoHasAnyArtifact(f); + }) + .sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), autoSubsOnly: autoSubsOnly.sort(), diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -565,6 +565,62 @@ async function main() { return; } + // Metadata scan: --skip-download --print <json template> -a <batch file>. + // MUST come before the plain `-a` branch below, which would otherwise swallow + // it and "download" every URL in the batch. Prints one JSON line per URL and + // creates NO video directory — that is the whole contract of the scan. + if ( + has("--skip-download") && + has("--print") && + arg("-a") && + !has("--flat-playlist") + ) { + const batch = arg("-a"); + const text = await readFile(batch, "utf8"); + const urls = text.split("\n").map((s) => s.trim()).filter(Boolean); + await appendFile( + "fake-ytdlp.invocations", + `metadata-scan:${urls.length} cookies=${cookieArg()}\n`, + ); + let printed = 0; + for (const url of urls) { + const id = urlIdYouTube(url); + if (!id) continue; + // The auth sentinels behave exactly as cookieGateBlocked does elsewhere: + // an ERROR line without cookies, a normal record with them. Per-video, so + // the batch keeps going (yt-dlp --ignore-errors). + const lower = url.toLowerCase(); + const gated = + (lower.includes("cookiegated") || lower.includes("needsauth")) && + !cookieArg(); + if (gated) { + process.stderr.write( + `ERROR: [youtube] ${id}: Sign in to confirm your age. This video may be inappropriate for some users.\n`, + ); + continue; + } + const sent = urlSentinels(url); + const record = { + id, + title: `Synthetic ${id}`, + description: `Synthetic video ${id}`, + upload_date: "20240101", + live_status: sent.isLive + ? "is_live" + : sent.isUpcoming + ? "is_upcoming" + : sent.wasLive + ? "was_live" + : "not_live", + duration: sent.longDuration ? 6000 : 60, + }; + process.stdout.write(`${JSON.stringify(record)}\n`); + printed++; + } + process.stderr.write(`[fake-ytdlp] metadata scan printed ${printed}\n`); + return; + } + const audioFmt = has("-x") ? arg("--audio-format") : undefined; const playlistFile = arg("-a"); diff --git a/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist b/editor/e2e/fixtures/test-transcripts/title-filter-channel/channels/test-filter/playlist @@ -2,3 +2,4 @@ https://www.youtube.com/watch?v=guestvid0001 https://www.youtube.com/watch?v=guestvid0002 https://www.youtube.com/watch?v=plainvid0001 https://www.youtube.com/watch?v=plainvid0002 +https://www.youtube.com/watch?v=needsauthvid01 diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts @@ -1,5 +1,5 @@ import { readFile, writeFile } from "node:fs/promises"; -import { test, expect } from "@playwright/test"; +import { test, expect, type Page } from "@playwright/test"; import { channelStage, generateReport, @@ -10,26 +10,27 @@ import { } from "./helpers"; import { baseUrl } from "./baseUrl"; -// The per-channel download filter, end to end. The fake yt-dlp titles every -// video "Synthetic <id>" (fixtures/bin/fake-ytdlp.mjs), so an include of -// /guest/ matches guestvid0001 and guestvid0002 and misses plainvid0001 and -// plainvid0002 — no fixture-specific metadata needed. +// The per-channel download filter and the metadata scan that feeds it. +// +// The fake yt-dlp titles every video "Synthetic <id>" (fixtures/bin/fake-ytdlp.mjs), +// so an include of /guest/ matches guestvid0001 and guestvid0002 and misses +// plainvid0001 and plainvid0002 — no fixture-specific metadata needed. +// `needsauthvid01` is the sentinel the fake refuses without cookies, so the scan +// has one video it cannot read. const CHANNEL = "test-filter"; const ROOT = `test-transcripts/channels/${CHANNEL}`; const GUEST = ["guestvid0001", "guestvid0002"]; const PLAIN = ["plainvid0001", "plainvid0002"]; +const NEEDS_AUTH = "needsauthvid01"; -// The signature the fixture's filter settles under: "v1\0<include>\0<exclude>". -const GUEST_SIGNATURE = "v1\0guest\0"; - -type DownloadOutcome = { - status: string; - filter?: { - name: string; - reason: string; - permanent?: boolean; - signature?: string; - }; +type MetadataScan = { + version: number; + entries: Record< + string, + { title: string; description: string; uploadDate: string; scannedAt: string } + >; + errors: Record<string, { class: string; message: string; at: string }>; + lastRun: { scanned: number; errors: number; stopped?: string } | null; }; type Snapshot = { @@ -37,13 +38,20 @@ type Snapshot = { totals: { videos: number; transcribed: number; downloaded: number }; buckets: { skippedByTitleFilter?: string[]; skippedByFilter?: string[] }; undownloadedIds: string[]; + metadataScan?: { scanned: number; errors: number; unscanned: number }; }; +async function readInvocations(): Promise<string> { + return readFile(resolvePath(`${ROOT}/fake-ytdlp.invocations`), "utf8").catch( + () => "", + ); +} + async function readArchive(): Promise<string> { return readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(() => ""); } -async function writeConfig(filter: Record<string, string> | null) { +async function writeFilterConfig(filter: Record<string, string> | null) { await writeFile( resolvePath(`${ROOT}/config.json`), JSON.stringify( @@ -60,13 +68,10 @@ async function writeConfig(filter: Record<string, string> | null) { await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); } -// Regenerate the report and wait for the snapshot to actually advance. The -// click is retried for the usual reason (a pre-hydration click fires nothing); +// Regenerate the report and wait for the snapshot to actually advance. The click +// is retried for the usual reason (a pre-hydration click fires nothing); // regenerating is idempotent. -async function refreshReport( - page: import("@playwright/test").Page, - after: string, -): Promise<Snapshot> { +async function refreshReport(page: Page, after: string): Promise<Snapshot> { await page.goto("/channels"); const refresh = page.getByRole("button", { name: `refresh report ${CHANNEL}`, @@ -77,9 +82,9 @@ async function refreshReport( async () => { await refresh.click({ timeout: 5_000 }).catch(() => {}); for (let i = 0; i < 20; i++) { - const cur = await readJson<Snapshot>( - `${ROOT}/snapshot.json`, - ).catch(() => null); + const cur = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch( + () => null, + ); if (cur && cur.generatedAt > after) return true; await new Promise((r) => setTimeout(r, 250)); } @@ -91,121 +96,153 @@ async function refreshReport( return readJson<Snapshot>(`${ROOT}/snapshot.json`); } -async function download(page: import("@playwright/test").Page) { +async function runScan(page: Page): Promise<void> { + await page.goto(channelStage(CHANNEL, "playlist")); + await page.getByRole("button", { name: "Scan metadata" }).click(); + await expect(page.getByLabel("Scan metadata output")).toContainText( + "Metadata scan complete", + { timeout: 60_000 }, + ); +} + +async function download(page: Page): Promise<void> { await page.goto(channelStage(CHANNEL, "download")); await page.getByRole("button", { name: "Download videos" }).click(); - const log = page.getByLabel("Download videos output"); - await expect(log).toContainText("Managed download complete", { - timeout: 60_000, - }); - return log; + await expect(page.getByLabel("Download videos output")).toContainText( + "Managed download complete", + { timeout: 60_000 }, + ); } -test("a non-matching video is settled, not re-queued", async ({ page }) => { - test.setTimeout(150_000); +test("the scan reads titles without creating a single video directory", async ({ + page, +}) => { + test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); - await page.goto(channelStage(CHANNEL, "download")); - await page.getByRole("button", { name: "Download videos" }).click(); - const log = page.getByLabel("Download videos output"); + await runScan(page); - await expect(log).toContainText( - "Skipping plainvid0001: title/description does not match include /guest/i [filter=titleFilter, settled]", - { timeout: 60_000 }, + const scan = await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`); + expect(Object.keys(scan.entries).sort()).toEqual([...GUEST, ...PLAIN].sort()); + expect(scan.entries.guestvid0001.title).toBe("Synthetic guestvid0001"); + // The one video the fake refuses without cookies. No cookie spec is + // configured, so there is no auth retry to make and it stays an error. + expect(Object.keys(scan.errors)).toEqual([NEEDS_AUTH]); + expect(scan.errors[NEEDS_AUTH].class).toBe("needs_auth"); + expect(scan.lastRun?.scanned).toBe(4); + + // THE LOAD-BEARING ASSERTION. A video directory holding a metadata.info.json + // is admitted to the LMDB index and the exported site by buildIndex, and its + // dir name is what deriveChannelSets reads as "ever fetched". The scan must + // create none of them. + for (const id of [...GUEST, ...PLAIN, NEEDS_AUTH]) { + expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); + } + expect(await readArchive()).toBe(""); + + // The report then settles the non-matches, with nothing written per video. + const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const snapshot = await refreshReport(page, before.generatedAt); + expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN); + expect([...snapshot.undownloadedIds].sort()).toEqual( + [...GUEST, NEEDS_AUTH].sort(), ); - await expect(log).toContainText("Managed download complete", { - timeout: 60_000, - }); + expect(snapshot.metadataScan?.scanned).toBe(4); + expect(snapshot.metadataScan?.errors).toBe(1); + // The unreadable video is not re-queued for a scan for a day — otherwise a + // members-only video would keep the backlog above zero forever. It is still + // ordinary download work, asserted above. + expect(snapshot.metadataScan?.unscanned).toBe(0); +}); + +test("a settled video is never downloaded", async ({ page }) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await generateReport(page, CHANNEL); + await runScan(page); + const afterScan = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + await refreshReport(page, afterScan.generatedAt); + + await download(page); - // The matching videos downloaded and were archived, exactly as with no filter. for (const id of GUEST) { expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); expect(await readArchive()).toContain(id); } - - // The non-matching ones are SETTLED: a terminal outcome carrying the filter's - // signature, and deliberately NO archive line (an archive line means - // "downloaded" to verifyTranscripts and the missingFromArchive bucket). + // Settled: no directory, no archive line, and — the part that matters on a + // channel with ~1,800 of them — yt-dlp was never invoked for them at all. + const invocations = await readInvocations(); for (const id of PLAIN) { - const outcome = await readJson<DownloadOutcome>( - `${ROOT}/data/${id}/download-outcome.json`, - ); - expect(outcome.status).toBe("skipped-filtered"); - expect(outcome.filter?.name).toBe("titleFilter"); - expect(outcome.filter?.permanent).toBe(true); - expect(outcome.filter?.signature).toBe(GUEST_SIGNATURE); - expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe( - false, - ); + expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); expect(await readArchive()).not.toContain(id); + expect(invocations).not.toContain(`prefetch:https://www.youtube.com/watch?v=${id}`); } - - // The report is regenerated by the debounced scheduler after the job, so poll - // for the bucket rather than reading the instant the job ends. - await expect - .poll( - async () => { - const snap = await readJson<Snapshot>( - `${ROOT}/snapshot.json`, - ).catch(() => null); - return snap?.buckets.skippedByTitleFilter ?? null; - }, - { timeout: 20_000, intervals: [200, 300, 500] }, - ) - .toEqual(PLAIN); - - const snapshot = await readJson<Snapshot>(`${ROOT}/snapshot.json`); - // Out of the runner's work queue — the line that makes the settlement stick, - // since undownloadedIds is artifact-derived and an archive line would not - // have kept them out of it. - for (const id of PLAIN) expect(snapshot.undownloadedIds).not.toContain(id); - // And out of the RETRYABLE skip bucket: a settled video is not "we'll try - // again", and it must not be double-counted. - expect(snapshot.buckets.skippedByFilter ?? []).toEqual([]); - // Two videos, not four: the settled pair are metadata stubs the operator - // asked us not to fetch. - expect(snapshot.totals.videos).toBe(2); - expect(snapshot.totals.downloaded).toBe(2); + await expect(page.getByLabel("Download videos output")).toContainText( + "Settled by this channel's download filter: 2 skipped", + ); }); -test("changing the filter un-settles every video it decided", async ({ +test("changing the filter re-decides the channel with no rescan", async ({ page, }) => { test.setTimeout(180_000); await resetData("title-filter-channel"); await generateReport(page, CHANNEL); - await download(page); + await runScan(page); + const afterScan = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const settled = await refreshReport(page, afterScan.generatedAt); + expect(settled.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN); - await expect - .poll( - async () => { - const snap = await readJson<Snapshot>( - `${ROOT}/snapshot.json`, - ).catch(() => null); - return snap?.buckets.skippedByTitleFilter ?? null; - }, - { timeout: 20_000, intervals: [200, 300, 500] }, - ) - .toEqual(PLAIN); - const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const invocationsBefore = await readInvocations(); - // Flip the filter on disk. The stored outcomes still carry the OLD signature, - // so nothing about them matches the new one — no migration, no sweep. - await writeConfig({ include: "plain" }); - const snapshot = await refreshReport(page, before.generatedAt); + // Flip the filter on disk. Nothing stored per video has to change, because + // nothing about the verdict was ever stored per video. + await writeFilterConfig({ include: "plain" }); + const flipped = await refreshReport(page, settled.generatedAt); - expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual([]); - for (const id of PLAIN) expect(snapshot.undownloadedIds).toContain(id); - // All four are videos again. - expect(snapshot.totals.videos).toBe(4); + expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(GUEST); + for (const id of PLAIN) expect(flipped.undownloadedIds).toContain(id); + for (const id of GUEST) expect(flipped.undownloadedIds).not.toContain(id); + // No rescan: re-deciding is pure computation over what the scan already read. + expect(await readInvocations()).toBe(invocationsBefore); - // And the next download actually fetches them — the settlement is gone, not - // merely hidden from the report. + // And the download now fetches the other half. await download(page); for (const id of PLAIN) { expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); - expect(await readArchive()).toContain(id); } + for (const id of GUEST) { + expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); + } +}); + +test("the Playlist stage reports the scan and what the filter makes of it", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await generateReport(page, CHANNEL); + await page.goto(channelStage(CHANNEL, "playlist")); + + const summary = page.getByLabel("metadata scan summary"); + await expect(summary).toContainText("Scanned 0 of 5 listed"); + + await page.getByRole("button", { name: "Scan metadata" }).click(); + await expect(page.getByLabel("Scan metadata output")).toContainText( + "Metadata scan complete", + { timeout: 60_000 }, + ); + + await page.goto(channelStage(CHANNEL, "playlist")); + await expect(summary).toContainText("Scanned 4 of 5 listed"); + await expect(summary).toContainText("1 could not be read"); + const verdict = page.getByLabel("download filter verdict"); + await expect(verdict).toContainText("2 match"); + await expect(verdict).toContainText("2 filtered out"); + await page.getByText("Matched titles").click(); + const matched = page.getByLabel("matched titles"); + await expect(matched).toContainText("Synthetic guestvid0001"); + await expect(matched).not.toContainText("Synthetic plainvid0001"); }); test("the form round-trips both patterns and refuses an invalid regex", async ({ @@ -216,25 +253,26 @@ test("the form round-trips both patterns and refuses an invalid regex", async ({ await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "configure")); - // The two fields live under Advanced. await page.locator("summary").filter({ hasText: "Advanced" }).click(); - const include = page.getByLabel(/^download filter: include/i); - const exclude = page.getByLabel(/^download filter: exclude/i); - await expect(include).toHaveValue("guest"); - await expect(exclude).toHaveValue(""); - - // An unparseable pattern is refused by the form, because a bad pattern on - // disk makes the filter silently inert at download time. - await exclude.fill("elf("); + await expect(page.getByLabel(/^download filter: include/i)).toHaveValue( + "guest", + ); + await expect(page.getByLabel(/^download filter: exclude/i)).toHaveValue(""); + + // An unparseable pattern is refused by the form, because a bad pattern on disk + // makes the filter silently inert at download time. + await page.getByLabel(/^download filter: exclude/i).fill("elf("); await page.getByRole("button", { name: "Save changes" }).click(); await expect( page.getByText(/Download filter exclude is not a valid regular expression/), ).toBeVisible({ timeout: 30_000 }); - // Nothing was written. - const unchanged = await readJson<{ - downloadFilter?: { include?: string; exclude?: string }; - }>(`${ROOT}/config.json`); - expect(unchanged.downloadFilter).toEqual({ include: "guest" }); + expect( + ( + await readJson<{ downloadFilter?: Record<string, string> }>( + `${ROOT}/config.json`, + ) + ).downloadFilter, + ).toEqual({ include: "guest" }); // A valid pair round-trips. await page.goto(channelStage(CHANNEL, "configure")); @@ -244,12 +282,12 @@ test("the form round-trips both patterns and refuses an invalid regex", async ({ await page.getByRole("button", { name: "Save changes" }).click(); await expect .poll( - async () => { - const cfg = await readJson<{ - downloadFilter?: { include?: string; exclude?: string }; - }>(`${ROOT}/config.json`).catch(() => null); - return cfg?.downloadFilter ?? null; - }, + async () => + ( + await readJson<{ downloadFilter?: Record<string, string> }>( + `${ROOT}/config.json`, + ).catch(() => ({}) as { downloadFilter?: Record<string, string> }) + ).downloadFilter ?? null, { timeout: 30_000, intervals: [250, 500] }, ) .toEqual({ include: "guest|special", exclude: "rerun" }); @@ -271,7 +309,7 @@ test("the form round-trips both patterns and refuses an invalid regex", async ({ await expect .poll( async () => { - const cfg = await readJson<{ downloadFilter?: unknown }>( + const cfg = await readJson<Record<string, unknown>>( `${ROOT}/config.json`, ).catch(() => null); return cfg ? "downloadFilter" in cfg : null; @@ -280,3 +318,37 @@ test("the form round-trips both patterns and refuses an invalid regex", async ({ ) .toBe(false); }); + +test("/operations/metadata-scan lists the channel and can run it", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await generateReport(page, CHANNEL); + // The backlog is a snapshot field, so the report has to have seen the + // unscanned playlist before the operation page can offer the work. + const first = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const seeded = await refreshReport(page, first.generatedAt); + expect(seeded.metadataScan?.unscanned).toBe(5); + + await page.goto("/operations/metadata-scan"); + await expect( + page.getByRole("heading", { name: "Metadata scan" }), + ).toBeVisible(); + await expect(page.getByRole("link", { name: CHANNEL })).toBeVisible(); + + await page.getByRole("button", { name: "Scan metadata" }).first().click(); + await expect + .poll( + async () => + Object.keys( + ( + await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`).catch( + () => null, + ) + )?.entries ?? {}, + ).length, + { timeout: 60_000, intervals: [500, 1000] }, + ) + .toBe(4); +});