Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fff366c4bfff0ea9bdaa84337985ba9e5eb6ff9d
parent e28f06b689347c543bf04d24fd63f641fd61390e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 04:00:02 -0400

metadata scan: "consecutive" means consecutive in the batch, not in the pipes

The soft-block streak was counted as errors arrived, and reset whenever a
record arrived. But stdout and stderr reach the parent independently: a record
read late could break a streak it was never part of, so a real block looked
like ten separate dead videos — and the e2e failed under load while passing in
isolation, which is the signature of exactly that.

Unavailable errors are now held keyed by id, and the verdict is computed after
the pass by walking the ids IN BATCH ORDER, where the order is a fact rather
than a scheduling accident. An id with no outcome at all (the pass was killed)
neither extends nor breaks a run. The in-stream counter survives as what it
always should have been: a trigger to stop hammering a source that is refusing
us, not the decision.

This also makes the rule more honest for real channels. Ten genuinely deleted
videos scattered through a listing have records between them, so they are not
consecutive in batch order and are recorded normally — where the arrival-order
count could have called them a block and thrown them away.

The spec's refreshReport helper gets the matching fix: "newer" is not "reflects
my change". The debounced snapshot scheduler arms on every job completion, so a
regen triggered by the scan can land after a config edit carrying the config
from before it. Callers that just changed the filter now wait for a report that
has actually seen it, and still assert the exact contents.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/ytdlp/metadataScan.ts | 107++++++++++++++++++++++++++++++++++++++++++++++++++++---------------------------
Meditor/e2e/title-filter.spec.ts | 30++++++++++++++++++++++++++----
2 files changed, 96 insertions(+), 41 deletions(-)

diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts @@ -300,25 +300,51 @@ export async function runMetadataScan( // Native extractor id -> canonical id, learned from the records this run // reads, so an `ERROR:` line naming a native id can be attributed correctly. const nativeToCanonical = new Map<string, string>(); - // "Video unavailable" errors seen back-to-back, HELD rather than recorded - // until the streak is broken — so a soft block's ids are never written to the - // store at all, not written and then deleted. See SOFT_BLOCK_STREAK. - let unavailableStreak: Array<{ id: string; message: string }> = []; - - // The streak ended without reaching the threshold: these really were - // unavailable videos, so record them. - const commitStreak = () => { - for (const e of unavailableStreak) { + // "Video unavailable" errors, HELD rather than recorded until the pass ends — + // so a soft block's ids are never written to the store at all, rather than + // written and then deleted. See SOFT_BLOCK_STREAK. + // + // KEYED BY ID, NOT A LIST, because "consecutive" has to mean consecutive IN + // THE BATCH and not in the order the parent happened to read two pipes. + // stdout and stderr arrive independently: a record read late could break a + // streak that never actually contained it, which made a real soft block look + // like ten separate dead videos (and the e2e flaky under load). The verdict is + // computed from the batch order after the pass, where the order is a fact. + let unavailableById = new Map<string, string>(); + + // Longest run of consecutive unavailable ids in BATCH order, over the ids this + // pass was asked for. Nothing else decides whether this was a block. + const longestUnavailableRun = (ids: ReadonlyArray<string>): number => { + let best = 0; + let run = 0; + for (const id of ids) { + if (unavailableById.has(id)) { + run++; + if (run > best) best = run; + } else if (entries[id] || errors[id]) { + // A video we DID read, or that failed for another reason, genuinely + // breaks the run. + run = 0; + } + // An id with no outcome at all was never reached (the pass was killed); + // it neither extends nor breaks the run. + } + return best; + }; + + // These really were unavailable videos, so record them. + const commitUnavailable = () => { + for (const [id, message] of unavailableById) { const err: MetadataScanError = { class: "deleted", - message: e.message, + message, at: new Date().toISOString(), }; - errors[e.id] = err; - unflushedErrors[e.id] = err; + errors[id] = err; + unflushedErrors[id] = err; pending++; } - unavailableStreak = []; + unavailableById = new Map(); }; // WHAT HAS NOT REACHED DISK YET. `entries`/`errors` above are the run's full @@ -424,9 +450,6 @@ export async function runMetadataScan( }; entries[id] = entry; unflushedEntries[id] = entry; - // A readable video proves we are not blocked, so whatever ran before it - // was a real streak of unavailable videos. - commitStreak(); scannedTotal++; pending++; opts.setProgress?.({ @@ -450,6 +473,8 @@ export async function runMetadataScan( }); let stderrBuf = ""; + // See the comment in takeError: a stop signal, not a verdict. + let unavailableRun = 0; const takeError = (line: string) => { // Once this pass is blocked, the rest of the stream is that block // talking. Nothing more is recorded from it. @@ -478,25 +503,19 @@ export async function runMetadataScan( if (!id) return; // "Video unavailable", possibly the soft block. Held, not recorded. if (UNAVAILABLE_LINE.test(message)) { - unavailableStreak.push({ id, message }); - if (unavailableStreak.length >= SOFT_BLOCK_STREAK) { - block = { - kind: "soft-block", - message: - `${unavailableStreak.length} consecutive "Video unavailable" errors — ` + - `treating this as a soft block rather than ${unavailableStreak.length} deleted videos. ` + - `Their ids were NOT recorded, so the next run re-reads them.`, - }; - unavailableStreak = []; - child.kill("SIGTERM"); - } + unavailableById.set(id, message); + // A LIVE COUNTER, used only to STOP FETCHING — not to decide anything. + // stderr is ordered within itself, so this is a good enough trigger to + // stop hammering a source that is refusing us; the verdict is computed + // from batch order once the pass is over. + unavailableRun++; + if (unavailableRun >= SOFT_BLOCK_STREAK) child.kill("SIGTERM"); return; } - // Anything else ends the streak and is recorded on its own terms. A - // members-only video is the case this protects: it errors every time, it - // is genuinely not fetchable, and it must not be thrown away with a - // soft block's ids. - commitStreak(); + // Any other error breaks the run — a members-only video is the case this + // protects: it errors every time, it is genuinely not fetchable, and it + // must not be thrown away with a soft block's ids. + unavailableRun = 0; if (cls === "needs_auth") needsAuthIds.push(id); const err: MetadataScanError = { class: cls, @@ -521,9 +540,23 @@ export async function runMetadataScan( const result = await child; if (stdoutBuf) takeRecord(stdoutBuf); if (stderrBuf) takeError(stderrBuf); - // A streak that never reached the threshold is just a run of dead videos. - if (!block) commitStreak(); - else unavailableStreak = []; + // THE VERDICT, from the batch order. A run of consecutive unavailable ids + // long enough to be a block is discarded so the next run re-reads them; + // anything shorter is what it looks like — videos that really are gone. + if (!block) { + const run = longestUnavailableRun(ids); + if (run >= SOFT_BLOCK_STREAK) { + block = { + kind: "soft-block", + message: + `${run} consecutive "Video unavailable" errors — ` + + `treating this as a soft block rather than ${run} deleted videos. ` + + `Their ids were NOT recorded, so the next run re-reads them.`, + }; + } + } + if (block) unavailableById = new Map(); + else commitUnavailable(); await flushChain; await flush(true); @@ -579,7 +612,7 @@ export async function runMetadataScan( `retrying the remaining ${remaining.length} ids with --cookies-from-browser ${retryCookies}.\n`, ); block = null; - unavailableStreak = []; + unavailableById = new Map(); usedCookies = true; if (remaining.length > 0) { await runPass(remaining, retryCookies, "cookie-retry"); diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts @@ -75,7 +75,18 @@ async function writeFilterConfig(filter: Record<string, string> | null) { // Regenerate the report and wait for the snapshot to actually advance. The click // is retried for the usual reason (a pre-hydration click fires nothing); // regenerating is idempotent. -async function refreshReport(page: Page, after: string): Promise<Snapshot> { +// +// `until` exists because "newer" is not the same as "reflects my change". The +// debounced snapshot scheduler arms itself whenever a job finishes, so a regen +// triggered by the SCAN can land after a config edit and carry the config from +// before it — a newer generatedAt with the old verdict. Callers that just +// changed the filter pass a predicate and get a report that has actually seen +// it; they still assert the exact contents afterwards. +async function refreshReport( + page: Page, + after: string, + until?: (snapshot: Snapshot) => boolean, +): Promise<Snapshot> { await page.goto("/channels"); const refresh = page.getByRole("button", { name: `refresh report ${CHANNEL}`, @@ -89,7 +100,9 @@ async function refreshReport(page: Page, after: string): Promise<Snapshot> { const cur = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch( () => null, ); - if (cur && cur.generatedAt > after) return true; + if (cur && cur.generatedAt > after && (!until || until(cur))) { + return true; + } await new Promise((r) => setTimeout(r, 250)); } return false; @@ -203,8 +216,13 @@ test("changing the filter re-decides the channel with no rescan", async ({ // Flip the filter on disk. Nothing stored per video has to change, because // nothing about the verdict was ever stored per video. + const settledBefore = JSON.stringify(settled.buckets.skippedByTitleFilter ?? []); await writeFilterConfig({ include: "plain" }); - const flipped = await refreshReport(page, settled.generatedAt); + const flipped = await refreshReport( + page, + settled.generatedAt, + (s) => JSON.stringify(s.buckets.skippedByTitleFilter ?? []) !== settledBefore, + ); expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual( [...GUEST, LIVE].sort(), @@ -394,7 +412,11 @@ test("Include every livestream lets the VOD through, with no rescan", async ({ ) .toEqual({ include: "guest", includeLivestreams: true }); - const flipped = await refreshReport(page, settled.generatedAt); + const flipped = await refreshReport( + page, + settled.generatedAt, + (s) => !(s.buckets.skippedByTitleFilter ?? []).includes(LIVE), + ); expect(flipped.buckets.skippedByTitleFilter ?? []).toEqual(PLAIN); expect(flipped.undownloadedIds).toContain(LIVE); // A second positive selector, evaluated over what the scan already read.