commit 0086f0411b524e09652af94bbc2336ddfc2c6901
parent 195ac4c3e440459c37feb8ff677474c1dd2d5191
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 03:12:51 -0400
metadata scan: store records under the canonical id, not yt-dlp's native one
yt-dlp's `id` is the NATIVE extractor id; every consumer of this store keys by
the CANONICAL id — extractVideoId of the URL, the name `data/<id>/` carries.
They coincide on YouTube and diverge everywhere else: Rumble's native id is the
embed id (`v2apmfn`) while the canonical id is the URL slug (`v2846lb`), and
Twitch's native id drops the leading `v`. Keyed natively, the settled set on
those platforms would be full of ids no playlist entry and no directory has
ever been called — so nothing would ever settle and every run would re-fetch
the entire listing.
`webpage_url` joins the print template so each record can canonicalize itself.
An `ERROR:` line has no URL on it, only the native id, so it resolves through
three fallbacks: a record already read this run, an id that IS a canonical id,
or a target whose URL contains it. An unrecognizable one is kept rather than
dropped — a wrong key on an error row costs one re-read, dropping it makes the
backlog never reach zero.
The fake now reports a deliberately different native id for Rumble URLs, and
the bot-check fixture carries one, so the e2e fails if the canonicalization is
removed.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
5 files changed, 150 insertions(+), 8 deletions(-)
diff --git a/common/ytdlp/metadataScan.test.ts b/common/ytdlp/metadataScan.test.ts
@@ -0,0 +1,81 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { canonicalScanId, resolveScanErrorId } from "./metadataScan";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/ytdlp/metadataScan.test.ts
+
+// yt-dlp's `id` is the NATIVE extractor id. Every consumer of the scan store
+// keys by the CANONICAL id (extractVideoId of the URL — the name data/<id>/
+// carries). They coincide on YouTube and diverge everywhere else, and keyed
+// natively nothing on those platforms would ever settle: the settled set would
+// be full of ids no playlist entry and no directory has ever been called.
+
+test("YouTube: native and canonical are the same id", () => {
+ assert.equal(
+ canonicalScanId({
+ id: "dQw4w9WgXcQ",
+ webpage_url: "https://www.youtube.com/watch?v=dQw4w9WgXcQ",
+ }),
+ "dQw4w9WgXcQ",
+ );
+});
+
+test("Rumble: the URL slug wins over the embed id", () => {
+ // The case that motivated this: the archive keys by the embed id, a local cue
+ // dir is named for the URL slug (see AGENTS.md).
+ assert.equal(
+ canonicalScanId({
+ id: "v2apmfn",
+ webpage_url: "https://rumble.com/v2846lb-some-title.html",
+ }),
+ "v2846lb",
+ );
+});
+
+test("Twitch: the canonical id keeps the leading v the native id drops", () => {
+ assert.equal(
+ canonicalScanId({
+ id: "2735747903",
+ webpage_url: "https://www.twitch.tv/videos/v2735747903",
+ }),
+ "v2735747903",
+ );
+});
+
+test("no usable URL falls back to the native id rather than dropping the record", () => {
+ assert.equal(canonicalScanId({ id: "abc123" }), "abc123");
+ assert.equal(canonicalScanId({ id: "abc123", webpage_url: "not a url" }), "abc123");
+ assert.equal(canonicalScanId({}), "");
+});
+
+// --- error lines -------------------------------------------------------------
+// yt-dlp names the NATIVE id on an ERROR: line and gives no URL to canonicalize.
+
+const URLS = new Map([
+ ["v2846lb", "https://rumble.com/v2846lb-some-title.html"],
+ ["dQw4w9WgXcQ", "https://www.youtube.com/watch?v=dQw4w9WgXcQ"],
+]);
+
+test("an error id resolves through a record already read this run", () => {
+ assert.equal(
+ resolveScanErrorId("v2apmfn", new Map([["v2apmfn", "v2846lb"]]), URLS),
+ "v2846lb",
+ );
+});
+
+test("an error id that IS a canonical id is taken as-is (the YouTube case)", () => {
+ assert.equal(resolveScanErrorId("dQw4w9WgXcQ", new Map(), URLS), "dQw4w9WgXcQ");
+});
+
+test("an error id found inside a target URL resolves to that target", () => {
+ // Nothing was read for it this run — the very first video failing is exactly
+ // when this matters — so the only link left is the URL itself.
+ assert.equal(resolveScanErrorId("v2846lb-some", new Map(), URLS), "v2846lb");
+});
+
+test("an unrecognizable error id is kept, not dropped", () => {
+ // A wrong key for an ERROR row costs one re-read next run. Dropping it would
+ // make the backlog never reach zero.
+ assert.equal(resolveScanErrorId("whoknows", new Map(), URLS), "whoknows");
+});
diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts
@@ -47,8 +47,15 @@ import {
// yt-dlp's JSON-subset print template: one JSON object per entry, so the scan
// parses lines instead of whole info jsons — and so nothing is ever written to
// disk by yt-dlp itself.
+// `webpage_url` is in here for a reason that is not cosmetic: yt-dlp's `id` is
+// the NATIVE extractor id, and every consumer of this store keys by the
+// CANONICAL id (extractVideoId of the URL — the name data/<id>/ carries). The
+// two coincide on YouTube and diverge everywhere else: Rumble's native id is the
+// embed id (`v2apmfn`) while the canonical id is the URL slug (`v2846lb`);
+// Twitch's native id drops the leading `v`. Keyed natively, nothing on those
+// platforms would ever settle and every run would re-fetch the whole listing.
const PRINT_TEMPLATE =
- "%(.{id,title,description,upload_date,live_status,duration})j";
+ "%(.{id,title,description,upload_date,live_status,duration,webpage_url})j";
// Flush to the store every N records. A killed or rate-limited job keeps what it
// learned; 25 is small enough that a run stopped early has lost almost nothing
@@ -75,6 +82,40 @@ const ERROR_LINE = /^ERROR:\s*(?:\[[^\]]+\]\s*)?([^\s:]+)\s*:\s*(.*)$/;
const SOFT_BLOCK_STREAK = 10;
const UNAVAILABLE_LINE = /video unavailable/i;
+// The id this record must be STORED under: the canonical id derived from the
+// video's own URL, falling back to yt-dlp's native id when there is no usable
+// URL. Pure and exported so the divergence is testable without spawning
+// anything.
+export function canonicalScanId(record: {
+ id?: unknown;
+ webpage_url?: unknown;
+}): string {
+ const url = typeof record.webpage_url === "string" ? record.webpage_url : "";
+ const fromUrl = url ? extractVideoId(url) : null;
+ if (fromUrl) return fromUrl;
+ return typeof record.id === "string" ? record.id : "";
+}
+
+// yt-dlp names the NATIVE id on an `ERROR:` line, and there is no URL on it to
+// canonicalize. Three ways back, cheapest first: a record we already read this
+// run, a target whose canonical id is literally this string (the YouTube case),
+// or a target whose URL contains it (the Rumble/Twitch case). Failing all three
+// the native id is stored as-is — a wrong key for an error row is a re-read next
+// run, not a lost video.
+export function resolveScanErrorId(
+ native: string,
+ nativeToCanonical: ReadonlyMap<string, string>,
+ urlsById: ReadonlyMap<string, string>,
+): string {
+ const known = nativeToCanonical.get(native);
+ if (known) return known;
+ if (urlsById.has(native)) return native;
+ for (const [canonical, url] of urlsById) {
+ if (url.includes(native)) return canonical;
+ }
+ return native;
+}
+
export type MetadataScanResult = {
scanned: number;
errors: number;
@@ -237,6 +278,9 @@ export async function runMetadataScan(
let block: { kind: "bot-check" | "soft-block" | "rate-limit"; message: string } | null =
null;
const needsAuthIds: string[] = [];
+ // Native extractor id -> canonical id, learned from the records this run
+ // reads, so an `ERROR:` line naming a native id can be attributed correctly.
+ const nativeToCanonical = new Map<string, string>();
// "Video unavailable" errors seen back-to-back, HELD rather than recorded
// until the streak is broken — so a soft block's ids are never written to the
// store at all, not written and then deleted. See SOFT_BLOCK_STREAK.
@@ -341,8 +385,10 @@ export async function runMetadataScan(
} catch {
return;
}
- const id = typeof parsed.id === "string" ? parsed.id : "";
+ const id = canonicalScanId(parsed);
if (!id) return;
+ const native = typeof parsed.id === "string" ? parsed.id : "";
+ if (native && native !== id) nativeToCanonical.set(native, id);
const entry: MetadataScanEntry = {
title: typeof parsed.title === "string" ? parsed.title : "",
description:
@@ -405,7 +451,10 @@ export async function runMetadataScan(
return;
}
const m = ERROR_LINE.exec(trimmed);
- const id = m?.[1] ?? "";
+ const native = m?.[1] ?? "";
+ const id = native
+ ? resolveScanErrorId(native, nativeToCanonical, urlsById)
+ : "";
const message = (m?.[2] || trimmed).trim().slice(0, 300);
if (!id) return;
// "Video unavailable", possibly the soft block. Held, not recorded.
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -627,8 +627,14 @@ async function main() {
continue;
}
const sent = urlSentinels(url);
+ // yt-dlp's `id` is the NATIVE extractor id. On Rumble it is the EMBED id,
+ // which is not the URL slug the rest of the app keys by — so the fake
+ // reports a deliberately different one there, and the scan has to
+ // canonicalize through webpage_url or nothing on Rumble ever settles.
+ const nativeId = url.includes("rumble.com") ? `embed${id}` : id;
const record = {
- id,
+ id: nativeId,
+ webpage_url: url,
title: `Synthetic ${id}`,
description: `Synthetic video ${id}`,
upload_date: "20240101",
diff --git a/editor/e2e/fixtures/test-transcripts/metadata-scan-botcheck/channels/test-botcheck/playlist b/editor/e2e/fixtures/test-transcripts/metadata-scan-botcheck/channels/test-botcheck/playlist
@@ -2,3 +2,4 @@ https://www.youtube.com/watch?v=botcheck0001
https://www.youtube.com/watch?v=guestvid0001
https://www.youtube.com/watch?v=guestvid0002
https://www.youtube.com/watch?v=plainvid0001
+https://rumble.com/v2846lb-rumble-guest.html
diff --git a/editor/e2e/metadata-scan-botcheck.spec.ts b/editor/e2e/metadata-scan-botcheck.spec.ts
@@ -25,6 +25,11 @@ const ALL = [
"guestvid0001",
"guestvid0002",
"plainvid0001",
+ // A Rumble URL, whose NATIVE extractor id (the embed id) is not the canonical
+ // id the rest of the app keys by. The fake reports a deliberately different
+ // one, so this entry only lands under "v2846lb" if the scan canonicalizes
+ // through webpage_url. Keyed natively, nothing on Rumble would ever settle.
+ "v2846lb",
];
type MetadataScan = {
@@ -55,7 +60,7 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing
const log = page.getByLabel("Scan metadata output");
await expect(log).toContainText(
- "bot check without cookies — retrying the remaining 4 ids with --cookies-from-browser firefox",
+ "bot check without cookies — retrying the remaining 5 ids with --cookies-from-browser firefox",
{ timeout: 60_000 },
);
await expect(log).toContainText("Metadata scan complete", { timeout: 60_000 });
@@ -68,7 +73,7 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing
// NOT a refusal: no stop, and no cooldown. A cooldown here would cost the
// operator an hour for a problem the retry solved in seconds.
expect(scan.lastRun?.stopped).toBeUndefined();
- expect(scan.lastRun?.scanned).toBe(4);
+ expect(scan.lastRun?.scanned).toBe(5);
await expect(log).not.toContainText("STOPPED");
await expect(log).not.toContainText("cooldown has been recorded");
@@ -76,8 +81,8 @@ test("a bot check on a cookie-less pass retries with cookies instead of backing
// ordinary cookie-less scan, which is what "when-required" means.
const invocations = await scanInvocations();
expect(invocations).toHaveLength(2);
- expect(invocations[0]).toBe("metadata-scan:4 cookies=");
- expect(invocations[1]).toBe("metadata-scan:4 cookies=firefox");
+ expect(invocations[0]).toBe("metadata-scan:5 cookies=");
+ expect(invocations[1]).toBe("metadata-scan:5 cookies=firefox");
// And the cooldown really was not recorded: a second scan starts rather than
// being refused with the rate-limit notice.