commit e99effb6bd1d8876325d484478afb94b1ae41480
parent 2155234e871fd4be0a4e5b994ba18a7a6f68576f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 22:29:14 -0400
common: a bare HTTP 403 classifies as "network" and backs the platform off
Before, a 403 matched only when yt-dlp's traceback happened to contain "ssl".
No new class, so no download-outcome.json format change. Changelog bullet
reworded: syncs are refused during the cooldown, the sweep retries after.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 27 insertions(+), 1 deletion(-)
diff --git a/common/lib/availability.test.ts b/common/lib/availability.test.ts
@@ -0,0 +1,22 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { classifyDownloadFailure } from "./availability";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test lib/availability.test.ts
+
+test("a bare HTTP 403 is a network failure (backs the platform off)", () => {
+ assert.equal(
+ classifyDownloadFailure(
+ "ERROR: [Rumble] v6abc: Unable to download webpage: HTTP Error 403: Forbidden",
+ undefined,
+ ),
+ "network",
+ );
+ // 429 still wins as rate_limit.
+ assert.equal(
+ classifyDownloadFailure("HTTP Error 429: Too Many Requests", undefined),
+ "rate_limit",
+ );
+ assert.equal(classifyDownloadFailure("ERROR: Unsupported URL", undefined), "unknown");
+});
diff --git a/common/lib/availability.ts b/common/lib/availability.ts
@@ -249,6 +249,10 @@ export function classifyDownloadFailure(
return "rate_limit";
}
if (
+ // A bare 403 (Cloudflare's fingerprint block on Rumble) backs the platform
+ // off like any other transport failure. Before this it matched only when
+ // the traceback happened to contain "ssl".
+ /http error 403/.test(s) ||
/econnrefused/.test(s) ||
/etimedout/.test(s) ||
/enetunreach/.test(s) ||
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,7 +1,7 @@
# Changelog
## [Unreleased]
-- **Rumble works again, and a Rumble full sweep that gets rate-limited no longer fails the sync.** Every Rumble request had started coming back 403 from Cloudflare unless yt-dlp presents a browser fingerprint (yt-dlp #17496), so Rumble downloads failed and a Rumble channel could not even be added. Every yt-dlp run for a Rumble channel — sync, download, metadata scan, availability check, the clip-window fetch and the new-channel probe — now passes `--impersonate chrome --sleep-requests 1`, from one table in the code; a channel's own extra yt-dlp arguments still come last and still win. Separately, a full sweep that hits HTTP 429 part-way through the listing used to fail the whole sync and try again on the next one, so a large channel (The Quartering on Rumble, 44 days) never synced at all. What it read is now treated as *incomplete* — not a listing, so nothing is flagged missing and the stored playlist is untouched: the job records the platform's rate-limit cooldown, says "sweep incomplete: 429 at page N of the listing, M entries" in its log, does the ordinary newest-first sync instead, and succeeds. No full sweep is attempted while that platform is cooling down. Any other yt-dlp failure still fails the sync as before.
+- **Rumble works again, and a Rumble full sweep that gets rate-limited no longer fails the sync.** Every Rumble request had started coming back 403 from Cloudflare unless yt-dlp presents a browser fingerprint (yt-dlp #17496), so Rumble downloads failed and a Rumble channel could not even be added. Every yt-dlp run for a Rumble channel — sync, download, metadata scan, availability check, the clip-window fetch and the new-channel probe — now passes `--impersonate chrome --sleep-requests 1`, from one table in the code; a channel's own extra yt-dlp arguments still come last and still win. Separately, a full sweep that hits HTTP 429 part-way through the listing used to fail the whole sync and try again on the next one, so a large channel (The Quartering on Rumble, 44 days) never synced at all. What it read is now treated as *incomplete* — not a listing, so nothing is flagged missing and the stored playlist is untouched: the job records the platform's rate-limit cooldown, says "sweep incomplete: 429 at page N of the listing, M entries" in its log, does the ordinary newest-first sync instead, and succeeds. Syncs for that platform are then refused until its cooldown ends, and the full sweep is tried again after that. Any other yt-dlp failure still fails the sync as before.
- **Channel rows no longer scroll over a group's controls on `/channels`.** Scrolled down and to the right, the pinned Slug column of every row painted over the pinned group header and its five station buttons (Sync, Download, Transcribe, Digest and the speaker lane), and took the clicks. The pinned Slug cell and the group header sat at the same stacking level, and the later rows won. The rack now has one named layer order, kept in one file: the Advanced panel, then the column header, then the group header, then the pinned checkbox and Slug cells. Nothing ties any more. The screenshot audit found four more problems, fixed as well. A group header's name and buttons now stay on screen however far the columns scroll across (they used to scroll off to the left). An Advanced panel opened near the bottom or the right edge scrolls itself into view instead of being cut off. The rule above a pinned group header moves with it instead of leaving a gap the rows showed through. On a phone, the column header no longer paints over the selection bar pinned to the bottom of the screen.
- **A group's Transcribe works for YouTube channels, and it counts what it queues.** The station used to be disabled for every `youtube`-handling channel with the message "a youtube-handling channel never runs whisper". That was wrong. A YouTube video that came down with no captions is transcription work like any other, and the automatic runner already treats it that way. Transcribe now counts two kinds of video, after the usual members-only, deleted and private exclusions: downloaded videos with no transcript at all, and downloaded videos whose only transcript is YouTube's auto-captions. Pressing it queues exactly those videos, by id, as the channel page does: up to two jobs per channel on the transcription queue. A video downloaded before it went private, members-only or deleted is no longer transcribed by the group button, because it was never in the figure. Pressing it again while either job runs says *already running*. The wording names no method ("…has downloaded audio to transcribe", "…each takes minutes"). **This figure can now be higher than the Transcription band in the same rack on channels with many auto-caption-only videos.** The band counts videos with no transcript at all, while the station counts everything its button would queue. That is intended.
- **The editor's atomic JSON, text and binary writes now go one way, and a failed write no longer leaves a temp file behind.** Nineteen JSON write sites and seven text and binary ones each wrote `<file>.tmp-<pid>` and renamed it over the original — the channel roster, maybe-missing and metadata-scan records, the scheduler and auto-queue state, worker defaults, widget presets, the homepage config, relocation markers, shard configs, the duplicate and media-scan reports and their review decisions, the saved-video backup manifest, both cue normalizers, the playlist, the failed-transcriptions list, the X cookie jar, a site's CHANGELOG cut, a saved video copied into its store across drives, and the video page's VTT promote and remark. They now all go through one writer (`common/lib/jsonFile-server.ts`), which gives every write its own temp name and queues writes to the same file one behind another, so two jobs touching one channel's roster at once cannot trip over each other's temp file. A write that fails now removes its temp: the live `.auto-queue/` holds 175 `state.json.tmp-…` files (173 of them empty) from the day `/home` filled up (2026-09-11), each one a failed write the old code left behind; nothing deletes those old ones for you — `find transcripts -name '*.tmp-*'` lists them. Four temp names stay, on purpose: the export build's two page writers `buildIndex.ts` (a streaming page writer) and `buildStats.ts` (a hand-joined array) — folding them is a restructuring, not a swap — and `transcode.ts` / `transcribeOne.ts` name the output file ffmpeg or the transcription app writes, which is not our write to fold. No file's contents change — every writer puts the same bytes on disk it did before, measured over the live corpus. The cookie jar is still created readable only by you.