commit 51969c53350c63744703192c1a1035a96a0f2756
parent 1a347f5166dd7c491a66e76611151fbf773ad446
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 22:40:41 -0400
Merge main (09d0f4de, release 5 slice R) into one-core/r5-exports
Conflicts only in editor/CHANGELOG.md [Unreleased] and plans/release-5.md
## Record; both slices' text kept, R's first.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
13 files changed, 630 insertions(+), 26 deletions(-)
diff --git a/common/controller/checkAvailability.ts b/common/controller/checkAvailability.ts
@@ -25,6 +25,8 @@ import { getSettings } from "../lib/settings";
// digest registry's claude-code lane, which has the same problem.
import { parseStdoutJson } from "../lib/parseStdoutJson";
import { readChannelConfig } from "./channels";
+import { channelExtraArgs, platformArgs } from "../ytdlp/channelArgs";
+import { detectPlatform } from "../lib/platform";
import { resolveShardItems } from "./shard";
import type { Paths } from "../lib/paths";
@@ -117,7 +119,6 @@ export async function runAvailabilityCheck({
const channelDir = path.join(paths.channelsDir, channelSlug);
const dataDir = path.join(channelDir, "data");
const config = await readChannelConfig(paths, channelSlug);
- const extraArgs = config?.ytdlpExtraArgs ?? [];
// Probe cookies in "always" mode ONLY. when-required/defer probes stay
// cookie-free deliberately, so auth gating keeps being OBSERVED as
// needs_auth — defer mode's exclusion + Needs-cookies bucket depend on that
@@ -214,8 +215,12 @@ export async function runAvailabilityCheck({
"--dump-json",
"--skip-download",
"--no-warnings",
- ...cookieArgs(probeCookies),
- ...extraArgs,
+ ...(config
+ ? channelExtraArgs(config, probeCookies)
+ : [
+ ...cookieArgs(probeCookies),
+ ...platformArgs(detectPlatform(url)),
+ ]),
"--",
url,
];
diff --git a/common/lib/availability.test.ts b/common/lib/availability.test.ts
@@ -0,0 +1,22 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { classifyDownloadFailure } from "./availability";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/availability.test.ts
+
+test("a bare HTTP 403 is a network failure (backs the platform off)", () => {
+ assert.equal(
+ classifyDownloadFailure(
+ "ERROR: [Rumble] v6abc: Unable to download webpage: HTTP Error 403: Forbidden",
+ undefined,
+ ),
+ "network",
+ );
+ // 429 still wins as rate_limit.
+ assert.equal(
+ classifyDownloadFailure("HTTP Error 429: Too Many Requests", undefined),
+ "rate_limit",
+ );
+ assert.equal(classifyDownloadFailure("ERROR: Unsupported URL", undefined), "unknown");
+});
diff --git a/common/lib/availability.ts b/common/lib/availability.ts
@@ -249,6 +249,10 @@ export function classifyDownloadFailure(
return "rate_limit";
}
if (
+ // A bare 403 (Cloudflare's fingerprint block on Rumble) backs the platform
+ // off like any other transport failure. Before this it matched only when
+ // the traceback happened to contain "ssl".
+ /http error 403/.test(s) ||
/econnrefused/.test(s) ||
/etimedout/.test(s) ||
/enetunreach/.test(s) ||
diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts
@@ -0,0 +1,63 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { channelExtraArgs, platformArgs, PLATFORM_ARGS } from "./channelArgs";
+import type { ChannelConfig } from "../lib/channelConfig";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/channelArgs.test.ts
+
+const RUMBLE = ["--impersonate", "chrome", "--sleep-requests", "1"];
+
+function cfg(over: Partial<ChannelConfig>): ChannelConfig {
+ return { handling: "youtube", ...over } as ChannelConfig;
+}
+
+test("a rumble URL gets the four platform args", () => {
+ assert.deepEqual(
+ channelExtraArgs(cfg({ url: "https://rumble.com/c/TheQuartering" })),
+ RUMBLE,
+ );
+});
+
+test("youtube gets no platform args", () => {
+ assert.deepEqual(
+ channelExtraArgs(cfg({ url: "https://www.youtube.com/@x" })),
+ [],
+ );
+ assert.deepEqual(platformArgs("youtube"), []);
+ assert.deepEqual(platformArgs(null), []);
+});
+
+test("an explicit config.platform wins over the URL's host", () => {
+ assert.deepEqual(
+ channelExtraArgs(
+ cfg({ url: "https://example.com/feed", platform: "rumble" }),
+ ),
+ RUMBLE,
+ );
+});
+
+test("order: cookies, then platform args, then the channel's own (so an override wins)", () => {
+ const args = channelExtraArgs(
+ cfg({
+ url: "https://rumble.com/c/x",
+ ytdlpExtraArgs: ["--sleep-requests", "3"],
+ }),
+ "firefox",
+ );
+ assert.deepEqual(args, [
+ "--cookies-from-browser",
+ "firefox",
+ ...RUMBLE,
+ "--sleep-requests",
+ "3",
+ ]);
+ // yt-dlp is last-flag-wins: the channel's 3 is the one that applies.
+ assert.equal(args.lastIndexOf("--sleep-requests"), args.length - 2);
+});
+
+test("platformArgs returns a copy, not the table's array", () => {
+ const a = platformArgs("rumble");
+ a.push("--mutated");
+ assert.deepEqual(PLATFORM_ARGS.rumble, RUMBLE);
+});
diff --git a/common/ytdlp/channelArgs.ts b/common/ytdlp/channelArgs.ts
@@ -1,21 +1,54 @@
// The per-channel argv every yt-dlp invocation in this repo appends: the cookie
-// source when the resolved policy calls for one, then the channel's own
-// `ytdlpExtraArgs` verbatim.
+// source when the resolved policy calls for one, then the platform's fixed
+// args (PLATFORM_ARGS), then the channel's own `ytdlpExtraArgs` verbatim.
//
// Lifted out of ytdlp/downloadOneManaged.ts (where it was `channelConfigArgs`,
// which still delegates here) because the clip-window fetch has to honour the
// same two things and must not grow its own idea of them: an operator who set
// `--limit-rate` on a channel meant it for every byte that channel costs, not
// only for the bytes a full download costs.
+//
+// This is the ONE builder. `configArgs` (runYtdlp.ts), the metadata scan, the
+// quick availability check and the new-channel probe all go through it (the
+// probe has no config yet, so it calls `platformArgs` directly). Before
+// release 5 there were four copies and the probe had none, so a Rumble
+// channel could not even be created.
import type { ChannelConfig } from "../lib/channelConfig";
+import { cookieArgs } from "../lib/cookiePolicy";
+import { detectPlatform, type Platform } from "../lib/platform";
+
+// Args a platform needs on EVERY yt-dlp spawn, whatever the job. Code, not a
+// setting: per-channel `ytdlpExtraArgs` is the tweak surface, and it comes
+// AFTER these, so a channel override wins (yt-dlp is last-flag-wins for
+// `--sleep-requests` and `--impersonate`).
+//
+// rumble: every request 403s at Cloudflare without a browser TLS fingerprint
+// (yt-dlp #17496); a probe on 2026-09-24 got 200 with `--impersonate chrome`
+// and 403 without. `--sleep-requests 1` paces the listing walk — the
+// the-quartering-rumble full sweep 429'd at page 155 unpaced the same day.
+export const PLATFORM_ARGS: Partial<Record<Platform, readonly string[]>> = {
+ rumble: ["--impersonate", "chrome", "--sleep-requests", "1"],
+};
+
+export function platformArgs(platform: Platform | null | undefined): string[] {
+ if (!platform) return [];
+ return [...(PLATFORM_ARGS[platform] ?? [])];
+}
+
+export function channelPlatform(
+ config: Pick<ChannelConfig, "platform" | "url">,
+): Platform | null {
+ return config.platform ?? detectPlatform(config.url);
+}
export function channelExtraArgs(
config: ChannelConfig,
cookies?: string,
): string[] {
const args: string[] = [];
- if (cookies) args.push("--cookies-from-browser", cookies);
+ args.push(...cookieArgs(cookies));
+ args.push(...platformArgs(channelPlatform(config)));
if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
return args;
}
diff --git a/common/ytdlp/metadataScan.ts b/common/ytdlp/metadataScan.ts
@@ -28,9 +28,9 @@ import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilter
import {
alwaysCookies,
authRetryCookies,
- cookieArgs,
resolveCookiePolicy,
} from "../lib/cookiePolicy";
+import { channelExtraArgs } from "./channelArgs";
import type { Paths } from "../lib/paths";
import { getSettings } from "../lib/settings";
import { extractVideoId } from "../lib/videoId";
@@ -408,8 +408,7 @@ export async function runMetadataScan(
PRINT_TEMPLATE,
"-a",
batchFile,
- ...cookieArgs(cookies),
- ...(channelConfig.ytdlpExtraArgs ?? []),
+ ...channelExtraArgs(channelConfig, cookies),
];
opts.onLog(`$ ${paths.ytdlpBin} ${args.join(" ")}\n`);
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -13,6 +13,7 @@ import { getSettings } from "../lib/settings";
import { patchChannelConfig } from "../controller/channels";
import { diskGate } from "../lib/diskSpace";
import { detectPlatform } from "../lib/platform";
+import { channelExtraArgs, platformArgs } from "./channelArgs";
import { isRealAudioFile } from "../lib/videoStatus";
import { readVttProvenance } from "../lib/subtitleProvenance";
import { extractVideoId } from "../lib/videoId";
@@ -24,7 +25,6 @@ import {
} from "../lib/availability";
import {
alwaysCookies,
- cookieArgs,
resolveCookiePolicy,
type ResolvedCookiePolicy,
} from "../lib/cookiePolicy";
@@ -53,6 +53,7 @@ import {
isFullSweepDue,
resolveFullSweepIntervalMinutes,
} from "../jobs/deepSync";
+import { platformCooldownRemainingMs } from "../jobs/downloadBackoff";
import { computeKeepWindow, type KeepWindow } from "../controller/keptVideos";
import { downloadOneManaged } from "./downloadOneManaged";
import {
@@ -219,11 +220,10 @@ function abortableSleep(ms: number, signal: AbortSignal): Promise<void> {
});
}
+// Cookies, then the platform's fixed args, then the channel's own — the one
+// builder in channelArgs.ts.
function configArgs(config: ChannelConfig, cookies?: string): string[] {
- const args: string[] = [];
- args.push(...cookieArgs(cookies));
- if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
- return args;
+ return channelExtraArgs(config, cookies);
}
// The run-level cookie policy: settings + channel overrides, with the
@@ -313,6 +313,58 @@ export function outputArgsForUrl(
return OUTPUT_ARGS;
}
+// Thrown by enumeratePlaylistUrls when a listing is rate-limited part-way.
+//
+// A full enumeration that the platform rate-limited part-way through. What it
+// printed is a PREFIX of the listing, not a listing: acting on it as one would
+// flag every entry past the cut-off as missing, and failing the sync on it
+// throws away a download walk that would have worked. Thrown only for a
+// `rate_limit`-classified exit (HTTP 429 and friends, classifyDownloadFailure);
+// every other non-zero exit still throws the plain error.
+//
+// `platform` is the cooldown key the Sync gate and the auto-download runner
+// share (`detectPlatform(url) ?? "unknown"`); `pagesReached` is the last
+// `Downloading page N` the extractor logged (null for an extractor that does
+// not log pages); `count` is how many entries were printed before the cut.
+export class EnumerationIncompleteError extends Error {
+ readonly platform: string;
+ readonly pagesReached: number | null;
+ readonly count: number;
+ constructor(
+ exitCode: number | undefined,
+ platform: string,
+ pagesReached: number | null,
+ count: number,
+ ) {
+ super(
+ `yt-dlp exited with code ${exitCode} (rate-limited${
+ pagesReached !== null ? ` at page ${pagesReached}` : ""
+ } of the listing, ${count} entries)`,
+ );
+ this.name = "EnumerationIncompleteError";
+ this.platform = platform;
+ this.pagesReached = pagesReached;
+ this.count = count;
+ }
+}
+
+// The last `Downloading page N` an extractor logged — yt-dlp's paged channel
+// extractors print `[RumbleChannel] <name>: Downloading page 155` per page.
+export function lastListingPage(stderr: string): number | null {
+ let last: number | null = null;
+ for (const m of stderr.matchAll(/Downloading page (\d+)/g)) {
+ last = Number(m[1]);
+ }
+ return last;
+}
+
+// The per-platform cooldown key: the one the Sync gate (pipelineActions) and
+// the auto-download runner both use, so a sweep's 429 pauses exactly what
+// theirs would.
+function cooldownPlatformKey(config: ChannelConfig): string {
+ return detectPlatform(config.url) ?? "unknown";
+}
+
// Enumerate a channel's video URLs via `--flat-playlist --print url` (metadata
// only, no downloads). Pass `range` to fetch a single newest-first page via
// `-I start:end`; sync uses this to walk the channel incrementally.
@@ -350,6 +402,10 @@ async function enumeratePlaylistUrls(
});
child.stderr?.on("data", (c: Buffer) => opts.onLog(c.toString("utf8")));
const result = await child;
+ const urls = String(result.stdout ?? "")
+ .split("\n")
+ .map((s) => s.trim())
+ .filter(Boolean);
// yt-dlp exit code convention: 101 = "break-on-existing" / "max-downloads"
// (clean stop, not an error). Treat it the same as 0. A non-zero/101 exit
@@ -359,16 +415,22 @@ async function enumeratePlaylistUrls(
result.exitCode !== 101 &&
!opts.signal.aborted
) {
+ const stderr = String(result.stderr ?? "");
+ if (classifyDownloadFailure(stderr, undefined) === "rate_limit") {
+ throw new EnumerationIncompleteError(
+ result.exitCode,
+ cooldownPlatformKey(opts.channelConfig),
+ lastListingPage(stderr),
+ urls.length,
+ );
+ }
throw new Error(`yt-dlp exited with code ${result.exitCode}`);
}
if (result.exitCode === 101) {
opts.onLog(`yt-dlp stopped on existing entry (exit 101).\n`);
}
- return String(result.stdout ?? "")
- .split("\n")
- .map((s) => s.trim())
- .filter(Boolean);
+ return urls;
}
// Thin exported wrapper over the module-private enumeratePlaylistUrls for
@@ -418,6 +480,9 @@ export async function probeChannelMeta(opts: {
"1",
"--print",
"%(channel,uploader,playlist_title,playlist,uploader_id,title)s",
+ // No channel config exists yet; the platform's fixed args still apply
+ // (a Rumble probe 403s without `--impersonate`).
+ ...platformArgs(detectPlatform(url)),
url,
];
log(`$ ${paths.ytdlpBin} ${args.join(" ")}\n`);
@@ -1330,23 +1395,33 @@ const SYNC_PAGE_SIZE = 50;
// more expensive on a large channel, so it runs at most once per interval.
// `forceFullSweep` overrides the gate in both directions.
async function sync(opts: RunYtdlpOpts): Promise<void> {
- const sweep = opts.forceFullSweep ?? fullSweepDue(opts);
+ const sweep = opts.forceFullSweep ?? (await fullSweepDue(opts));
return sweep ? syncFullSweep(opts) : syncPaged(opts);
}
// Whether this sync should upgrade itself to a full sweep, resolved from the
-// per-channel override, the global cadence and the channel's lastFullSweepAt.
-function fullSweepDue(opts: RunYtdlpOpts): boolean {
+// per-channel override, the global cadence and the channel's lastFullSweepAt —
+// and never while the channel's platform is in a rate-limit cooldown: a sweep
+// is the longest request run a channel makes, and re-trying one into the
+// window that just 429'd is how the cooldown gets extended. The paged walk
+// still runs; `forceFullSweep` still overrides.
+export async function fullSweepDue(
+ opts: Pick<RunYtdlpOpts, "channelConfig" | "paths">,
+ now: number = Date.now(),
+): Promise<boolean> {
const scheduler = getSettings().syncScheduler;
const interval = resolveFullSweepIntervalMinutes(
opts.channelConfig,
scheduler,
);
- return isFullSweepDue(
- opts.channelConfig.lastFullSweepAt,
- interval,
- Date.now(),
+ if (!isFullSweepDue(opts.channelConfig.lastFullSweepAt, interval, now)) {
+ return false;
+ }
+ const cooling = await platformCooldownRemainingMs(
+ cooldownPlatformKey(opts.channelConfig),
+ opts.paths,
);
+ return cooling <= 0;
}
// The per-page download filter, shared by both passes so they can never drift:
@@ -1554,7 +1629,34 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
// 1. One enumeration, no range, then the gate: record everything it saw in
// the roster (additive, so this is safe unconditionally) and decide
// whether the listing itself may be acted on.
- const urls = await enumeratePlaylistUrls(opts, root);
+ let urls: string[];
+ try {
+ urls = await enumeratePlaylistUrls(opts, root);
+ } catch (err) {
+ if (!(err instanceof EnumerationIncompleteError)) throw err;
+ // A 429 part-way through the listing: what was printed is a prefix, not a
+ // listing, so it never reaches acceptEnumeration (no playlist rewrite, no
+ // missing set, no lastFullSweepAt). Record the platform's cooldown the way
+ // a download's 429 does, then do this sync's job the cheap way.
+ // Best-effort, as on the download path: a failed state write must not fail
+ // the sync this branch exists to rescue.
+ try {
+ await opts.onPlatformBackoff?.("rate_limit");
+ } catch (backoffErr) {
+ opts.onLog(
+ `Warning: could not record the ${err.platform} cooldown (${
+ backoffErr instanceof Error ? backoffErr.message : String(backoffErr)
+ }); continuing.\n`,
+ );
+ }
+ opts.onLog(
+ `Full sweep incomplete: 429 at ${
+ err.pagesReached !== null ? `page ${err.pagesReached}` : "an unknown page"
+ } of the listing, ${err.count} entries — not a listing. Ran the paged walk instead; syncs wait for the ${err.platform} cooldown, then the sweep is retried.\n`,
+ );
+ if (opts.signal.aborted) return;
+ return syncPaged(opts);
+ }
if (opts.signal.aborted) return;
const { decision, roster, listedIds, now } = await acceptEnumeration(
diff --git a/common/ytdlp/sweepIncomplete.test.ts b/common/ytdlp/sweepIncomplete.test.ts
@@ -0,0 +1,111 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtempSync, writeFileSync, chmodSync, mkdirSync } from "node:fs";
+import os from "node:os";
+import path from "node:path";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test ytdlp/sweepIncomplete.test.ts
+//
+// A 429 part-way through a full enumeration is "incomplete", not a listing and
+// not a failure (release 5 slice R). getPaths() memoizes, so the env is set
+// before anything imports it.
+const ROOT = mkdtempSync(path.join(os.tmpdir(), "sweep-incomplete-"));
+process.env.TRANSCRIPTS_DIR = ROOT;
+process.env.SETTINGS_FILE = path.join(ROOT, "settings.json");
+writeFileSync(process.env.SETTINGS_FILE, "{}\n");
+mkdirSync(path.join(ROOT, "channels", "c"), { recursive: true });
+
+const {
+ fetchFlatPlaylistUrls,
+ EnumerationIncompleteError,
+ lastListingPage,
+ fullSweepDue,
+} = await import("./runYtdlp");
+const { getPaths } = await import("../lib/paths");
+const { recordDownloadBackoff } = await import("../jobs/downloadBackoff");
+
+// A stand-in yt-dlp: prints `urls` on stdout, `stderr` on stderr, exits `code`.
+function fakeBin(name: string, urls: string[], stderr: string, code: number) {
+ const bin = path.join(ROOT, name);
+ writeFileSync(
+ bin,
+ `#!/bin/sh\ncat <<'OUT'\n${urls.join("\n")}\nOUT\ncat >&2 <<'ERR'\n${stderr}\nERR\nexit ${code}\n`,
+ );
+ chmodSync(bin, 0o755);
+ return bin;
+}
+
+function run(bin: string) {
+ return fetchFlatPlaylistUrls({
+ channelConfig: { handling: "youtube", url: "https://rumble.com/c/x" } as never,
+ paths: { ...getPaths(), ytdlpBin: bin },
+ channelSlug: "c",
+ onLog: () => {},
+ signal: new AbortController().signal,
+ });
+}
+
+const PAGES = [1, 2, 3]
+ .map((n) => `[RumbleChannel] x: Downloading page ${n}`)
+ .join("\n");
+const URLS = Array.from({ length: 15 }, (_, i) => `https://rumble.com/v${i}-t.html`);
+
+test("lastListingPage reads the extractor's last page line", () => {
+ assert.equal(lastListingPage(PAGES), 3);
+ assert.equal(lastListingPage("nothing here"), null);
+});
+
+test("a 429 mid-listing throws EnumerationIncompleteError with page and count", async () => {
+ const bin = fakeBin(
+ "ytdlp-429",
+ URLS,
+ `${PAGES}\nERROR: x: Unable to download webpage: HTTP Error 429: Too Many Requests`,
+ 1,
+ );
+ await assert.rejects(run(bin), (err: unknown) => {
+ assert.ok(err instanceof EnumerationIncompleteError);
+ assert.equal(err.platform, "rumble");
+ assert.equal(err.pagesReached, 3);
+ assert.equal(err.count, 15);
+ assert.match(err.message, /^yt-dlp exited with code 1/);
+ return true;
+ });
+});
+
+test("any other non-zero exit keeps throwing the plain error", async () => {
+ const bin = fakeBin("ytdlp-other", URLS, "ERROR: Unsupported URL", 1);
+ await assert.rejects(run(bin), (err: unknown) => {
+ assert.ok(err instanceof Error);
+ assert.ok(!(err instanceof EnumerationIncompleteError));
+ assert.equal(err.message, "yt-dlp exited with code 1");
+ return true;
+ });
+});
+
+test("fullSweepDue is false while the channel's platform cools down", async () => {
+ const paths = getPaths();
+ const due = {
+ channelConfig: {
+ handling: "youtube",
+ url: "https://rumble.com/c/x",
+ fullSweepIntervalMinutes: 60,
+ lastFullSweepAt: "2020-01-01T00:00:00.000Z",
+ } as never,
+ paths,
+ };
+ const youtube = {
+ channelConfig: {
+ handling: "youtube",
+ url: "https://www.youtube.com/@x",
+ fullSweepIntervalMinutes: 60,
+ lastFullSweepAt: "2020-01-01T00:00:00.000Z",
+ } as never,
+ paths,
+ };
+ assert.equal(await fullSweepDue(due), true);
+ await recordDownloadBackoff("rumble", paths);
+ assert.equal(await fullSweepDue(due), false);
+ // Another platform's sweep is unaffected.
+ assert.equal(await fullSweepDue(youtube), true);
+});
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Rumble works again, and a Rumble full sweep that gets rate-limited no longer fails the sync.** Every Rumble request had started coming back 403 from Cloudflare unless yt-dlp presents a browser fingerprint (yt-dlp #17496), so Rumble downloads failed and a Rumble channel could not even be added. Every yt-dlp run for a Rumble channel — sync, download, metadata scan, availability check, the clip-window fetch and the new-channel probe — now passes `--impersonate chrome --sleep-requests 1`, from one table in the code; a channel's own extra yt-dlp arguments still come last and still win. Separately, a full sweep that hits HTTP 429 part-way through the listing used to fail the whole sync and try again on the next one, so a large channel (The Quartering on Rumble, 44 days) never synced at all. What it read is now treated as *incomplete* — not a listing, so nothing is flagged missing and the stored playlist is untouched: the job records the platform's rate-limit cooldown, says "sweep incomplete: 429 at page N of the listing, M entries" in its log, does the ordinary newest-first sync instead, and succeeds. Syncs for that platform are then refused until its cooldown ends, and the full sweep is tried again after that. Any other yt-dlp failure still fails the sync as before.
- **A site can turn off its visitors' per-video transcript downloads.** The transcript viewer on a published site has always offered three ways to take a video's text away: a **Download** menu (txt, srt, json), **Copy MD**, and **Copy download command** (a `yt-dlp` line for a marked clip). A site's settings form now has a checkbox for them, *Per-video transcript downloads*, beside the archive zips one. Unticked, the site's next build shows none of the three; **Share** and the clip marks stay. It is on by default, so a site nobody touches is unchanged, and the file stores `"transcriptDownloads": false` only when it is off (`SITE.md` has the key). The site's machine contract (`/corpus.json`, `llms.txt`, the manifests and shards the MCP server and report-to-video read) is published either way. The hub follows the same switch: the hub form on **Sites** has the same checkbox, stored as `"transcriptDownloads": false` in the hub's `homepage.json`, and it hides the three controls on the hub's Browse and Ask pages. The editor's own video pages are unaffected.
- **Channel rows no longer scroll over a group's controls on `/channels`.** Scrolled down and to the right, the pinned Slug column of every row painted over the pinned group header and its five station buttons (Sync, Download, Transcribe, Digest and the speaker lane), and took the clicks. The pinned Slug cell and the group header sat at the same stacking level, and the later rows won. The rack now has one named layer order, kept in one file: the Advanced panel, then the column header, then the group header, then the pinned checkbox and Slug cells. Nothing ties any more. The screenshot audit found four more problems, fixed as well. A group header's name and buttons now stay on screen however far the columns scroll across (they used to scroll off to the left). An Advanced panel opened near the bottom or the right edge scrolls itself into view instead of being cut off. The rule above a pinned group header moves with it instead of leaving a gap the rows showed through. On a phone, the column header no longer paints over the selection bar pinned to the bottom of the screen.
- **A group's Transcribe works for YouTube channels, and it counts what it queues.** The station used to be disabled for every `youtube`-handling channel with the message "a youtube-handling channel never runs whisper". That was wrong. A YouTube video that came down with no captions is transcription work like any other, and the automatic runner already treats it that way. Transcribe now counts two kinds of video, after the usual members-only, deleted and private exclusions: downloaded videos with no transcript at all, and downloaded videos whose only transcript is YouTube's auto-captions. Pressing it queues exactly those videos, by id, as the channel page does: up to two jobs per channel on the transcription queue. A video downloaded before it went private, members-only or deleted is no longer transcribed by the group button, because it was never in the figure. Pressing it again while either job runs says *already running*. The wording names no method ("…has downloaded audio to transcribe", "…each takes minutes"). **This figure can now be higher than the Transcription band in the same rack on channels with many auto-caption-only videos.** The band counts videos with no transcript at all, while the station counts everything its button would queue. That is intended.
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -535,6 +535,31 @@ async function main() {
has("--print") &&
arg("--print") === "url"
) {
+ // Every enumeration records its WHOLE argv, in order, so a spec can assert
+ // the platform args (a rumble.com URL gets `--impersonate chrome
+ // --sleep-requests 1`) and their position before the channel's own.
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `flat-playlist:${has("-I") ? "paged" : "full"} argv=${argv.join(" ")}\n`,
+ );
+ // `sweep429`: a FULL enumeration (no -I range) prints three pages of five
+ // urls, logging each page the way yt-dlp's RumbleChannel extractor does,
+ // then hits HTTP 429 and exits 1 — the 2026-09-24 the-quartering-rumble
+ // sweep in miniature. A paged walk (-I) of the same channel lists normally.
+ if ((lastNonFlag() ?? "").toLowerCase().includes("sweep429") && !has("-I")) {
+ for (let page = 1; page <= 3; page++) {
+ process.stderr.write(`[RumbleChannel] sweep429: Downloading page ${page}\n`);
+ for (let i = 1; i <= 5; i++) {
+ process.stdout.write(
+ `https://rumble.com/vsweep${page}${i}-listed.html\n`,
+ );
+ }
+ }
+ process.stderr.write(
+ "ERROR: sweep429: Unable to download webpage: HTTP Error 429: Too Many Requests (caused by <HTTPError 429: Too Many Requests>)\n",
+ );
+ process.exit(1);
+ }
await modeStorePlaylist();
return;
}
diff --git a/editor/e2e/rumble-sweep.spec.ts b/editor/e2e/rumble-sweep.spec.ts
@@ -0,0 +1,166 @@
+import { readFile, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import {
+ channelStage,
+ generateReport,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSettings,
+} from "./helpers";
+
+// Release 5 slice R. A Rumble channel's full sweep that hits HTTP 429 part-way
+// through the listing (the 2026-09-24 the-quartering-rumble run, in miniature):
+//
+// - every yt-dlp spawn for a rumble.com URL carries the platform args
+// (`--impersonate chrome --sleep-requests 1`, channelArgs.ts PLATFORM_ARGS);
+// - the partial listing is "incomplete", not a listing and not a failure: the
+// job records the rumble cooldown, logs one line, runs the paged walk
+// instead and SUCCEEDS, and lastFullSweepAt is never stamped;
+// - the next Sync is refused for the cooldown, like any other 429.
+//
+// The fake yt-dlp's `sweep429` URL sentinel prints three pages of five urls
+// with RumbleChannel-style "Downloading page N" lines, then the 429, exit 1 —
+// for a FULL enumeration only; the paged walk (-I) lists normally.
+
+const CHANNEL = "availability-test";
+const FIXTURE = "availability-baseline";
+const channelRoot = `test-transcripts/channels/${CHANNEL}`;
+const RUMBLE_URL = "https://rumble.com/c/sweep429";
+
+const ALL_IDS = [
+ "vidpublic1",
+ "vidpublic2",
+ "viddeleted1",
+ "vidprivate1",
+ "vidmembers1",
+ "vidneedsauth1",
+];
+
+type ChannelConfigFile = {
+ url?: string;
+ lastSyncedAt?: string;
+ lastFullSweepAt?: string;
+};
+
+type AutoQueueStateFile = {
+ download: {
+ platformBackoff: Record<string, { until: number; fails: number }>;
+ };
+};
+
+async function readConfig(): Promise<ChannelConfigFile> {
+ return readJson<ChannelConfigFile>(`${channelRoot}/config.json`);
+}
+
+async function readInvocations(): Promise<string[]> {
+ const raw = await readFile(
+ resolvePath(`${channelRoot}/fake-ytdlp.invocations`),
+ "utf8",
+ ).catch(() => "");
+ return raw.split("\n").filter(Boolean);
+}
+
+test("a Rumble sweep that 429s mid-listing is incomplete: paged walk, cooldown, no sweep stamp", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ // The sweep is off in the e2e default settings; this spec needs it due.
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ verifyAvailabilityBeforeClean: false,
+ syncScheduler: {
+ enabled: false,
+ fullSweepIntervalMinutes: 1440,
+ fullSweepConfirmMaxSuspects: 25,
+ fullSweepShrinkGuardPercent: 10,
+ },
+ });
+ // Point the fixture channel at a Rumble URL carrying the sentinel.
+ const cfgPath = resolvePath(`${channelRoot}/config.json`);
+ const cfg = JSON.parse(await readFile(cfgPath, "utf8")) as ChannelConfigFile;
+ await writeFile(cfgPath, JSON.stringify({ ...cfg, url: RUMBLE_URL }, null, 2));
+ // The paged walk's listing: everything already archived, so it stops on its
+ // first page and downloads nothing — the download set is not the point here.
+ await writeFile(
+ resolvePath(`${channelRoot}/.fake-ytdlp-flat-playlist.json`),
+ JSON.stringify({ ids: ALL_IDS }),
+ );
+ await writeFile(
+ resolvePath(`${channelRoot}/archive`),
+ ALL_IDS.map((id) => `youtube ${id}`).join("\n") + "\n",
+ );
+ const storedPlaylist = "https://rumble.com/vstale-stored.html\n";
+ await writeFile(resolvePath(`${channelRoot}/playlist`), storedPlaylist);
+
+ await generateReport(page, CHANNEL);
+ await page.goto(channelStage(CHANNEL, "playlist"));
+ // Retried: a click landing before hydration fires nothing (sync-deep's idiom).
+ // lastSyncedAt advancing is the success signal — a failed sync never stamps it.
+ let synced = false;
+ for (let attempt = 0; attempt < 5 && !synced; attempt++) {
+ await page
+ .getByRole("button", { name: "Sync", exact: true })
+ .click({ timeout: 5_000 })
+ .catch(() => {});
+ for (let i = 0; i < 60; i++) {
+ const c = await readConfig().catch(() => null);
+ if (c?.lastSyncedAt) {
+ synced = true;
+ break;
+ }
+ await new Promise((r) => setTimeout(r, 250));
+ }
+ }
+ expect(synced, "the sync job succeeded (lastSyncedAt stamped)").toBe(true);
+
+ // The one "incomplete" line, naming the page, the count and the platform.
+ const output = page.getByLabel("Sync output");
+ await expect(output).toContainText(
+ "sweep incomplete: 429 at page 3 of the listing, 15 entries",
+ { timeout: 15_000 },
+ );
+ await expect(output).toContainText("rumble cooldown");
+
+ // Platform args on every spawn: the full enumeration AND the paged walk.
+ const inv = await readInvocations();
+ const full = inv.filter((l) => l.startsWith("flat-playlist:full"));
+ const paged = inv.filter((l) => l.startsWith("flat-playlist:paged"));
+ expect(full).toHaveLength(1);
+ expect(paged.length).toBeGreaterThan(0);
+ for (const line of [...full, ...paged]) {
+ expect(line).toContain("--impersonate chrome");
+ expect(line).toContain("--sleep-requests 1");
+ expect(line).toContain(RUMBLE_URL);
+ }
+
+ // The rumble cooldown is recorded where the Sync gate and the runner read it.
+ const state = await readJson<AutoQueueStateFile>(
+ "test-transcripts/.auto-queue/state.json",
+ );
+ expect(state.download.platformBackoff.rumble?.until).toBeGreaterThan(
+ Date.now(),
+ );
+
+ // Not a listing: no sweep stamp, stored playlist untouched.
+ const after = await readConfig();
+ expect(after.lastFullSweepAt).toBeUndefined();
+ expect(await readFile(resolvePath(`${channelRoot}/playlist`), "utf8")).toBe(
+ storedPlaylist,
+ );
+
+ // The next Sync is refused for the cooldown, as a notice.
+ await page.goto(channelStage(CHANNEL, "playlist"));
+ await expect(async () => {
+ await page
+ .getByRole("button", { name: "Sync", exact: true })
+ .click({ timeout: 5_000 });
+ await expect(page.getByLabel("Sync notice")).toContainText(
+ "rumble is in a rate-limit cooldown",
+ { timeout: 3_000 },
+ );
+ }).toPass({ timeout: 30_000 });
+});
diff --git a/plans/release-5.md b/plans/release-5.md
@@ -148,6 +148,74 @@ STATE. umtool: rebuild only if `git diff --stat <live>..<new> -- umtool` is non-
## Record
+### Slice R, as shipped — Rumble: impersonation everywhere, paced sweeps, incomplete ≠ failed (2026-09-24)
+
+Branch `one-core/r5-rumble` off `main` `f4da04a9`, seven commits (four, then three after review), unmerged. Every Rumble request
+403s at Cloudflare without a browser TLS fingerprint (yt-dlp #17496), and the
+`the-quartering-rumble` full sweep 429'd at page 155 and failed the whole sync every time for 44
+days. There is now one arg builder with a platform table, and a 429 part-way through a full
+enumeration is *incomplete*: cooldown recorded, one log line, the paged walk runs, the job succeeds.
+
+| sha | what |
+|---|---|
+| `bca929a1` | `common/ytdlp/channelArgs.ts`: `PLATFORM_ARGS` (`rumble: --impersonate chrome --sleep-requests 1`, comment names #17496 and the 2026-09-24 probe), `platformArgs(platform)`, `channelPlatform(config)` = `config.platform ?? detectPlatform(config.url)`; `channelExtraArgs` = cookies → platform args → `ytdlpExtraArgs` (override last, wins). `configArgs` (runYtdlp) delegates to it; the metadata scan's and `checkAvailability`'s inline copies call it (`checkAvailability` falls back to `cookieArgs` only when the channel has no config); `probeChannelMeta` gets `platformArgs(detectPlatform(url))` — `detectPlatform("https://rumble.com/c/…")` is `"rumble"` (host suffix). `channelArgs.test.ts` (5) |
+| `0a4b9999` | `EnumerationIncompleteError {platform, pagesReached, count}` + `lastListingPage(stderr)` (last `Downloading page N` — the real wording is `[RumbleChannel] TheQuartering: Downloading page 155`, job `01M3AVGZC5EZ9GCD04R9QW6NWX`). `enumeratePlaylistUrls` classifies a non-0/101 exit's buffered stderr with `classifyDownloadFailure`; `rate_limit` → the typed error (message still starts `yt-dlp exited with code N`), anything else → the plain error as before. `syncFullSweep` catches it before `acceptEnumeration`, calls `opts.onPlatformBackoff?.("rate_limit")` (= `recordDownloadBackoff(platform, paths)`, wired by `pipelineActions.ts:215`, the same helper and `nextBackoff` schedule a download's 429 uses), logs `Full sweep incomplete: 429 at page N of the listing, M entries — not a listing; next syncs are paged walks until the rumble cooldown ends.`, then `return syncPaged(opts)`. No playlist write, no missing set, no `lastFullSweepAt`. `fullSweepDue` is now exported, async, and false while `platformCooldownRemainingMs(detectPlatform(url) ?? "unknown")` > 0 — the Sync gate's key. `sweepIncomplete.test.ts` (4) |
+| `84788fff` | Fake yt-dlp: every `--flat-playlist --print url` records `flat-playlist:full\|paged argv=<whole argv>` in `fake-ytdlp.invocations` (no argv log existed for enumeration; the per-cwd invocations file is the fixture's existing record, so no new env var); `sweep429` sentinel — a full enumeration prints 3 pages × 5 urls with RumbleChannel page lines on stderr, then the real 429 `ERROR:` line, exit 1; a paged walk (`-I`) lists normally. New `rumble-sweep.spec.ts` |
+| `7b50cb45` | record, `[Unreleased]` bullet, superseded note on `plans/rumble-sweep-pacing.md` |
+| `c145f7df` | Review fixes: the cooldown write in the incomplete branch is best-effort (try/catch as on the download path; a failure logs `Warning: could not record the <platform> cooldown …` and the paged walk still runs); the log line now reads `… M entries — not a listing. Ran the paged walk instead; syncs wait for the rumble cooldown, then the sweep is retried.` (syncs are REFUSED during the cooldown, `pipelineActions.ts:133-145`); enumeration comment back above `enumeratePlaylistUrls`, the error class has its own one-liner; `checkAvailability` adds `platformArgs(detectPlatform(url))` when the channel has no config (no unit test file exists for it) |
+| `a70df836` | `classifyDownloadFailure`: `/http error 403/` joins the `network` group, so a bare `HTTP Error 403: Forbidden` backs the platform off — no new class, no `download-outcome.json` change. New `common/lib/availability.test.ts` (1). Changelog bullet reworded to match |
+| (this) | record: commit table, gates, caveats |
+
+**Gates.** tsc (`pnpm -r … exec tsc --noEmit`) clean at every commit. common **1747** (1738 + 9 new),
+editor unit **72**, test:scripts **156 + 1 skip**, mcp **219**. `next build` editor and export both
+green. e2e, the 17-spec list (none missing: `rumble-sweep queues fetch-window availability
+availability-backfill maybe-missing reconcile scheduler pipeline sync-deep sync-break-on-existing
+channels-actions cookies-mode audio-check-scenarios metadata-scan-botcheck metadata-scan-softblock
+new-channel-onboarding`): **82 passed, 1 failed, 7.8 min** (after 8 min in the queue) — the
+failure is `pipeline.spec.ts:164`, `EEXIST: mkdir …/test-transcripts/channels` inside
+`resetData`'s `cp` at `helpers.ts:69` (a fixture-reset race before the test body ran, 301 ms);
+rerun `pipeline.spec.ts rumble-sweep.spec.ts`: **8 passed, 56 s**. `fetch-window.spec.ts`'s 429 →
+cooldown test green. **After review** (`c145f7df`, `a70df836`): tsc clean per commit; common
+**1748** (+1); editor unit **72**; `next build` editor and export green; e2e `rumble-sweep
+availability availability-backfill queues fetch-window pipeline`: **33 passed, 0 failed,
+2.5 min** (after ~7 min in the queue). Numbers: none (no file format changes).
+
+**Found and left.**
+- **403 has no class of its own (item 4, partly done).** `DownloadFailureClass` is not cheap to widen:
+ it is persisted in every per-video `download-outcome.json` (`common/lib/downloadOutcome.ts:105`
+ — a new value is a sidecar format change), and it is a decision in the auto-runner
+ (`autoRunner.ts:1941-1944`, backoff on `rate_limit | network`), in `runYtdlp.ts` (backoff +
+ `abortOnError`, the managed-download loop ~:1029-1051) and in `downloadOneManaged.ts:1372`
+ (where it is produced). Before `a70df836` a Cloudflare 403 read as `network` only when the
+ traceback happened to contain "ssl" (the 2026-09-24 job did; a bare `HTTP Error 403:
+ Forbidden` did not, and read `unknown`). It is now in the `network` group, so it backs the
+ platform off. With `--impersonate` the 403 should stop occurring; if it recurs, a `blocked`
+ class wants its own decision (no backoff, a sentence on the channel), not a pattern.
+- **The metadata scan passes `--sleep-requests 1` twice for Rumble** (its own, then the platform
+ table's). Harmless (same value, last wins); left so the scan's own pacing stays for every
+ platform.
+- **The paged walk can also 429.** A `rate_limit` on a ranged page throws the typed error out of
+ `syncPaged`, and the sync fails exactly as before (its message is unchanged in prefix). The
+ fallback deliberately runs only once per sync.
+- **Retry cadence after an incomplete sweep.** No last-attempt stamp is kept: once the platform
+ cooldown ends (`nextBackoff`: 1 min, doubling, at most 30 min — `platformBackoff.ts:22-37`) the
+ sweep is due again on the next sync, and any successful Rumble download clears the escalation
+ (`clearBackoff`, `autoRunner.ts:1957`), so it restarts at 1 min. If live sweeps keep coming back
+ incomplete, a `lastFullSweepAttemptAt` (retry no sooner than N hours) is the follow-up.
+- **Outside the slice: umtool still spawns yt-dlp without `--impersonate`.**
+ `umtool/report-to-video/build-video.mjs` (`YTDLP` at :75) and `check-availability.mjs` (:32)
+ build their own argv, so Rumble clip fetches and availability checks there will still 403.
+ (The editor's clip-window fetch, `fetchWindowManaged.ts`, is covered.)
+- **`pipeline.spec.ts:164` EEXIST reset race** — first sighting: `resetData`'s `cp` hit
+ `EEXIST: mkdir …/test-transcripts/channels` (`helpers.ts:69`) before the test body; green on
+ rerun and in the post-review run. Watch for a second sighting.
+- **Commit trailers** name `Claude Opus 5.5 (1M context)` — the model that wrote them — not the
+ `Claude Fable 5.1` line in `plans/tools/implementer-rules.md`.
+
+**Rollout.** Item 6 above stands: after R is live remove `fullSweepIntervalMinutes: 0` from
+`the-quartering-rumble` and watch one paced sweep (at 1 req/s, ~155+ pages ≈ 3 minutes of
+listing); the first accepted listing after 44 days may shrink — the two-observation guard owns it.
+
### Slice X, as shipped — visitor exports off, per site (2026-09-24)
Branch `one-core/r5-exports` off `main` `f4da04a9`. One new `site.json` key, `transcriptDownloads`
diff --git a/plans/rumble-sweep-pacing.md b/plans/rumble-sweep-pacing.md
@@ -1,5 +1,10 @@
# Plan — a full sweep of a large Rumble channel must be paced, or it never completes
+> **Superseded by `plans/release-5.md` slice R** (shipped on `one-core/r5-rumble`, 2026-09-24). The
+> release plan corrects this one: four arg-builders plus the unargued probe, not one `configArgs`;
+> impersonation for every Rumble spawn; a 429 mid-sweep is "incomplete" and falls back to the paged
+> walk. Kept for its 2026-09-24 findings; do not implement from it.
+
**Found 2026-09-24** (operator: "job `01M3AVGZC5EZ9GCD04R9QW6NWX` keeps trying and failing a full
playlist fetch … no sync has happened in 44 days"). Facts verified on the live corpus and
`4130aca1`: