Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 0086f0411b524e09652af94bbc2336ddfc2c6901
parent 195ac4c3e440459c37feb8ff677474c1dd2d5191
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 03:12:51 -0400

metadata scan: store records under the canonical id, not yt-dlp's native one

yt-dlp's `id` is the NATIVE extractor id; every consumer of this store keys by
the CANONICAL id — extractVideoId of the URL, the name `data/<id>/` carries.
They coincide on YouTube and diverge everywhere else: Rumble's native id is the
embed id (`v2apmfn`) while the canonical id is the URL slug (`v2846lb`), and
Twitch's native id drops the leading `v`. Keyed natively, the settled set on
those platforms would be full of ids no playlist entry and no directory has
ever been called — so nothing would ever settle and every run would re-fetch
the entire listing.

`webpage_url` joins the print template so each record can canonicalize itself.
An `ERROR:` line has no URL on it, only the native id, so it resolves through
three fallbacks: a record already read this run, an id that IS a canonical id,
or a target whose URL contains it. An unrecognizable one is kept rather than
dropped — a wrong key on an error row costs one re-read, dropping it makes the
backlog never reach zero.

The fake now reports a deliberately different native id for Rumble URLs, and
the bot-check fixture carries one, so the e2e fails if the canonicalization is
removed.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/ytdlp/metadataScan.test.ts | 81+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/metadataScan.ts | 55++++++++++++++++++++++++++++++++++++++++++++++++++++---
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 8+++++++-
Meditor/e2e/fixtures/test-transcripts/metadata-scan-botcheck/channels/test-botcheck/playlist | 1+
Meditor/e2e/metadata-scan-botcheck.spec.ts | 13+++++++++----
5 files changed, 150 insertions(+), 8 deletions(-)

diff --git a/common/ytdlp/metadataScan.test.ts b/common/ytdlp/metadataScan.test.ts @@ -0,0 +1,81 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { canonicalScanId, resolveScanErrorId } from "./metadataScan"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test common/ytdlp/metadataScan.test.ts + +// yt-dlp's `id` is the NATIVE extractor id. Every consumer of the scan store +// keys by the CANONICAL id (extractVideoId of the URL — the name data/<id>/ +// carries). They coincide on YouTube and diverge everywhere else, and keyed +// natively nothing on those platforms would ever settle: the settled set would +// be full of ids no playlist entry and no directory has ever been called. + +test("YouTube: native and canonical are the same id", () => { + assert.equal( + canonicalScanId({ + id: "dQw4w9WgXcQ", + webpage_url: "https://www.youtube.com/watch?v=dQw4w9WgXcQ", + }), + "dQw4w9WgXcQ", + ); +}); + +test("Rumble: the URL slug wins over the embed id", () => { + // The case that motivated this: the archive keys by the embed id, a local cue + // dir is named for the URL slug (see AGENTS.md). + assert.equal( + canonicalScanId({ + id: "v2apmfn", + webpage_url: "https://rumble.com/v2846lb-some-title.html", + }), + "v2846lb", + ); +}); + +test("Twitch: the canonical id keeps the leading v the native id drops", () => { + assert.equal( + canonicalScanId({ + id: "2735747903", + webpage_url: "https://www.twitch.tv/videos/v2735747903", + }), + "v2735747903", + ); +}); + +test("no usable URL falls back to the native id rather than dropping the record", () => { + assert.equal(canonicalScanId({ id: "abc123" }), "abc123"); + assert.equal(canonicalScanId({ id: "abc123", webpage_url: "not a url" }), "abc123"); + assert.equal(canonicalScanId({}), ""); +}); + +// --- error lines ------------------------------------------------------------- +// yt-dlp names the NATIVE id on an ERROR: line and gives no URL to canonicalize. + +const URLS = new Map([ + ["v2846lb", "https://rumble.com/v2846lb-some-title.html"], + ["dQw4w9WgXcQ", "https://www.youtube.com/watch?v=dQw4w9WgXcQ"], +]); + +test("an error id resolves through a record already read this run", () => { + assert.equal( + resolveScanErrorId("v2apmfn", new Map([["v2apmfn", "v2846lb"]]), URLS), + "v2846lb", + ); +}); + +test("an error id that IS a canonical id is taken as-is (the YouTube case)", () => { + assert.equal(resolveScanErrorId("dQw4w9WgXcQ", new Map(), URLS), "dQw4w9WgXcQ"); +}); + +test("an error id found inside a target URL resolves to that target", () => { + // Nothing was read for it this run — the very first video failing is exactly + // when this matters — so the only link left is the URL itself. + assert.equal(resolveScanErrorId("v2846lb-some", new Map(), URLS), "v2846lb"); +}); + +test("an unrecognizable error id is kept, not dropped", () => { + // A wrong key for an ERROR row costs one re-read next run. Dropping it would + // make the backlog never reach zero. + assert.equal(resolveScanErrorId("whoknows", new Map(), URLS), "whoknows"); +}); diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts @@ -47,8 +47,15 @@ import { // yt-dlp's JSON-subset print template: one JSON object per entry, so the scan // parses lines instead of whole info jsons — and so nothing is ever written to // disk by yt-dlp itself. +// `webpage_url` is in here for a reason that is not cosmetic: yt-dlp's `id` is +// the NATIVE extractor id, and every consumer of this store keys by the +// CANONICAL id (extractVideoId of the URL — the name data/<id>/ carries). The +// two coincide on YouTube and diverge everywhere else: Rumble's native id is the +// embed id (`v2apmfn`) while the canonical id is the URL slug (`v2846lb`); +// Twitch's native id drops the leading `v`. Keyed natively, nothing on those +// platforms would ever settle and every run would re-fetch the whole listing. const PRINT_TEMPLATE = - "%(.{id,title,description,upload_date,live_status,duration})j"; + "%(.{id,title,description,upload_date,live_status,duration,webpage_url})j"; // Flush to the store every N records. A killed or rate-limited job keeps what it // learned; 25 is small enough that a run stopped early has lost almost nothing @@ -75,6 +82,40 @@ const ERROR_LINE = /^ERROR:\s*(?:\[[^\]]+\]\s*)?([^\s:]+)\s*:\s*(.*)$/; const SOFT_BLOCK_STREAK = 10; const UNAVAILABLE_LINE = /video unavailable/i; +// The id this record must be STORED under: the canonical id derived from the +// video's own URL, falling back to yt-dlp's native id when there is no usable +// URL. Pure and exported so the divergence is testable without spawning +// anything. +export function canonicalScanId(record: { + id?: unknown; + webpage_url?: unknown; +}): string { + const url = typeof record.webpage_url === "string" ? record.webpage_url : ""; + const fromUrl = url ? extractVideoId(url) : null; + if (fromUrl) return fromUrl; + return typeof record.id === "string" ? record.id : ""; +} + +// yt-dlp names the NATIVE id on an `ERROR:` line, and there is no URL on it to +// canonicalize. Three ways back, cheapest first: a record we already read this +// run, a target whose canonical id is literally this string (the YouTube case), +// or a target whose URL contains it (the Rumble/Twitch case). Failing all three +// the native id is stored as-is — a wrong key for an error row is a re-read next +// run, not a lost video. +export function resolveScanErrorId( + native: string, + nativeToCanonical: ReadonlyMap<string, string>, + urlsById: ReadonlyMap<string, string>, +): string { + const known = nativeToCanonical.get(native); + if (known) return known; + if (urlsById.has(native)) return native; + for (const [canonical, url] of urlsById) { + if (url.includes(native)) return canonical; + } + return native; +} + export type MetadataScanResult = { scanned: number; errors: number; @@ -237,6 +278,9 @@ export async function runMetadataScan( let block: { kind: "bot-check" | "soft-block" | "rate-limit"; message: string } | null = null; const needsAuthIds: string[] = []; + // Native extractor id -> canonical id, learned from the records this run + // reads, so an `ERROR:` line naming a native id can be attributed correctly. + const nativeToCanonical = new Map<string, string>(); // "Video unavailable" errors seen back-to-back, HELD rather than recorded // until the streak is broken — so a soft block's ids are never written to the // store at all, not written and then deleted. See SOFT_BLOCK_STREAK. @@ -341,8 +385,10 @@ export async function runMetadataScan( } catch { return; } - const id = typeof parsed.id === "string" ? parsed.id : ""; + const id = canonicalScanId(parsed); if (!id) return; + const native = typeof parsed.id === "string" ? parsed.id : ""; + if (native && native !== id) nativeToCanonical.set(native, id); const entry: MetadataScanEntry = { title: typeof parsed.title === "string" ? parsed.title : "", description: @@ -405,7 +451,10 @@ export async function runMetadataScan( return; } const m = ERROR_LINE.exec(trimmed); - const id = m?.[1] ?? ""; + const native = m?.[1] ?? ""; + const id = native + ? resolveScanErrorId(native, nativeToCanonical, urlsById) + : ""; const message = (m?.[2] || trimmed).trim().slice(0, 300); if (!id) return; // "Video unavailable", possibly the soft block. Held, not recorded. diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -627,8 +627,14 @@ async function main() { continue; } const sent = urlSentinels(url); + // yt-dlp's `id` is the NATIVE extractor id. On Rumble it is the EMBED id, + // which is not the URL slug the rest of the app keys by — so the fake + // reports a deliberately different one there, and the scan has to + // canonicalize through webpage_url or nothing on Rumble ever settles. + const nativeId = url.includes("rumble.com") ? `embed${id}` : id; const record = { - id, + id: nativeId, + webpage_url: url, title: `Synthetic ${id}`, description: `Synthetic video ${id}`, upload_date: "20240101", diff --git a/editor/e2e/fixtures/test-transcripts/metadata-scan-botcheck/channels/test-botcheck/playlist b/editor/e2e/fixtures/test-transcripts/metadata-scan-botcheck/channels/test-botcheck/playlist @@ -2,3 +2,4 @@ https://www.youtube.com/watch?v=botcheck0001 https://www.youtube.com/watch?v=guestvid0001 https://www.youtube.com/watch?v=guestvid0002 https://www.youtube.com/watch?v=plainvid0001 +https://rumble.com/v2846lb-rumble-guest.html diff --git a/editor/e2e/metadata-scan-botcheck.spec.ts b/editor/e2e/metadata-scan-botcheck.spec.ts @@ -25,6 +25,11 @@ const ALL = [ "guestvid0001", "guestvid0002", "plainvid0001", + // A Rumble URL, whose NATIVE extractor id (the embed id) is not the canonical + // id the rest of the app keys by. The fake reports a deliberately different + // one, so this entry only lands under "v2846lb" if the scan canonicalizes + // through webpage_url. Keyed natively, nothing on Rumble would ever settle. + "v2846lb", ]; type MetadataScan = { @@ -55,7 +60,7 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing const log = page.getByLabel("Scan metadata output"); await expect(log).toContainText( - "bot check without cookies — retrying the remaining 4 ids with --cookies-from-browser firefox", + "bot check without cookies — retrying the remaining 5 ids with --cookies-from-browser firefox", { timeout: 60_000 }, ); await expect(log).toContainText("Metadata scan complete", { timeout: 60_000 }); @@ -68,7 +73,7 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing // NOT a refusal: no stop, and no cooldown. A cooldown here would cost the // operator an hour for a problem the retry solved in seconds. expect(scan.lastRun?.stopped).toBeUndefined(); - expect(scan.lastRun?.scanned).toBe(4); + expect(scan.lastRun?.scanned).toBe(5); await expect(log).not.toContainText("STOPPED"); await expect(log).not.toContainText("cooldown has been recorded"); @@ -76,8 +81,8 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing // ordinary cookie-less scan, which is what "when-required" means. const invocations = await scanInvocations(); expect(invocations).toHaveLength(2); - expect(invocations[0]).toBe("metadata-scan:4 cookies="); - expect(invocations[1]).toBe("metadata-scan:4 cookies=firefox"); + expect(invocations[0]).toBe("metadata-scan:5 cookies="); + expect(invocations[1]).toBe("metadata-scan:5 cookies=firefox"); // And the cooldown really was not recorded: a second scan starts rather than // being refused with the rate-limit notice.