Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fd752389bd3e47b6d7feaf3573e64c317cb8452f
parent 56ff1ce13a13dbaee0cae4b996415e97899762c2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  6 Oct 2026 07:22:22 -0400

sources: BitChute is a platform (bitchute) — detection, ids, the extractor's label, the playable file, polite pacing

- bitchute in Platform/PLATFORM_VALUES and the one host table (bitchute.com
  and every subdomain: www., old., api., the seedNNN. media hosts)
- ids: /video/<id>/ and /embed/<id>/ give the id; a channel, playlist or
  profile page gives none; the canonical page is /video/<id>/
- platformFromMetadata: yt-dlp's BitChute and BitChuteChannel extractors,
  and a record whose own page is on BitChute, are bitchute
- summarize: a BitChute record carries its mp4 (the one format, on a
  bitchute.com media host) as mediaUrl, for the file player
- PLATFORM_ARGS.bitchute: --sleep-requests 3 and exponential --retry-sleep
  (http and extractor, 2 s doubling to 120 s); no impersonation, no parallel
  transfer
- PLATFORM_MIN_GAP_SECONDS: a per-platform floor under the gap between two
  videos (BitChute 60 s), stretched by up to half again at random; batch
  downloads, the subtitle pass, persists and the auto-download lane all
  honour it through downloadGapMs
- duplicates: bitchute ranks after twitch; moment links on BitChute are the
  bare page (no start parameter)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/controller/autoRunner.ts | 6+++++-
Mcommon/controller/persistVideos.ts | 6+++++-
Mcommon/jobs/platformBackoff.test.ts | 17+++++++++++++++++
Mcommon/jobs/platformBackoff.ts | 14+++++++++++++-
Acommon/lib/bitchute.test.ts | 117+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/bitchute.ts | 40++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/detectPlatform.mjs | 3+++
Mcommon/lib/duplicates.ts | 1+
Mcommon/lib/momentUrl.ts | 14++++++++------
Mcommon/lib/platform.ts | 5+++++
Mcommon/lib/transcripts-server.ts | 22++++++++++++++++------
Mcommon/lib/videoId.ts | 6++++++
Mcommon/ytdlp/channelArgs.test.ts | 21+++++++++++++++++++++
Mcommon/ytdlp/channelArgs.ts | 2++
Mcommon/ytdlp/downloadFormat.ts | 3++-
Mcommon/ytdlp/platformArgs.mjs | 41+++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 10+++++++++-
17 files changed, 311 insertions(+), 17 deletions(-)

diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -80,7 +80,10 @@ import { prunePlatformPacing, pruneSubtitleDeferrals, } from "../jobs/platformBackoff"; -import { staticSleepRequestsSeconds } from "../ytdlp/platformArgs.mjs"; +import { + platformMinGapSeconds, + staticSleepRequestsSeconds, +} from "../ytdlp/platformArgs.mjs"; import { applyUnitOutcome } from "../jobs/unitOutcome"; import { type DownloadFailureClass } from "../lib/availability"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; @@ -2048,6 +2051,7 @@ async function runLoop( unitSettings.sleepBetweenDownloadsSeconds, currentPaceSeconds(kindState.platformPace, unitPlatform, base), base, + { minSeconds: platformMinGapSeconds(unitPlatform) }, ); if (gap > 0) platformNextStartAt.set(unitPlatform, settledAt + gap); else platformNextStartAt.delete(unitPlatform); diff --git a/common/controller/persistVideos.ts b/common/controller/persistVideos.ts @@ -31,7 +31,10 @@ import { channelPlatform, pacingPlatformKey, } from "../ytdlp/channelArgs"; -import { staticSleepRequestsSeconds } from "../ytdlp/platformArgs.mjs"; +import { + platformMinGapSeconds, + staticSleepRequestsSeconds, +} from "../ytdlp/platformArgs.mjs"; import { downloadGapMs } from "../jobs/platformBackoff"; import { recordDownloadBackoff } from "../jobs/downloadBackoff"; @@ -408,6 +411,7 @@ export async function persistVideos({ config.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds, channelPaceSeconds(config), staticSleepRequestsSeconds(channelPlatform(config)), + { minSeconds: platformMinGapSeconds(channelPlatform(config)) }, ); if (gap > 0) { log(`Sleeping ${gap / 1000}s before the next download...`); diff --git a/common/jobs/platformBackoff.test.ts b/common/jobs/platformBackoff.test.ts @@ -198,6 +198,23 @@ test("the lane gap is the operator's sleep plus the pace above its base", () => assert.equal(downloadGapMs(-5, 0.5, 1), 0); }); +test("a platform floor raises the operator's sleep to it and jitters it by up to half again", () => { + // Below the floor: the floor, stretched by random * 50 %. + assert.equal(downloadGapMs(0, 3, 3, { minSeconds: 60, random: 0 }), 60_000); + assert.equal(downloadGapMs(30, 3, 3, { minSeconds: 60, random: 0.5 }), 75_000); + assert.equal(downloadGapMs(30, 3, 3, { minSeconds: 60, random: 1 }), 90_000); + // Above the floor the operator's sleep is the base; a raised pace still adds. + assert.equal(downloadGapMs(120, 3, 3, { minSeconds: 60, random: 0 }), 120_000); + assert.equal(downloadGapMs(0, 6, 3, { minSeconds: 60, random: 0 }), 63_000); + // Out-of-range randoms are clamped; an absent random still lands in range. + assert.equal(downloadGapMs(0, 3, 3, { minSeconds: 60, random: 7 }), 90_000); + const g = downloadGapMs(0, 3, 3, { minSeconds: 60 }); + assert.ok(g >= 60_000 && g <= 90_000, String(g)); + // No floor: exactly the old gap, never jittered. + assert.equal(downloadGapMs(30, 1, 1, { minSeconds: 0, random: 1 }), 30_000); + assert.equal(downloadGapMs(30, 1, 1, {}), 30_000); +}); + test("a held platform's backoff survives the prune; a hold with no backoff is dropped", () => { const now = 10 * BACKOFF_MAX_MS; const state = { diff --git a/common/jobs/platformBackoff.ts b/common/jobs/platformBackoff.ts @@ -303,12 +303,24 @@ export function currentPaceSeconds( // sleeps between two videos: the operator's `sleepBetweenDownloadsSeconds` // plus the pace ABOVE ITS BASE. The base pace is already paid inside every // spawn (`--sleep-requests`); what the gap adds is the part a rate limit added. +// +// A PLATFORM FLOOR (`floor.minSeconds`, the platform's PLATFORM_MIN_GAP_SECONDS +// in ytdlp/platformArgs.mjs — BitChute's 60 s): the operator's sleep is raised +// to at least the floor and then stretched by up to half again at random +// (`floor.random`, in [0, 1)), so a platform that rate-limits is never asked on +// a fixed beat. With no floor the gap is exactly what it always was. export function downloadGapMs( sleepBetweenDownloadsSeconds: number, paceSeconds: number, baseSeconds: number, + floor: { minSeconds?: number; random?: number } = {}, ): number { - const sleep = Math.max(0, sleepBetweenDownloadsSeconds); + let sleep = Math.max(0, sleepBetweenDownloadsSeconds); + const min = floor.minSeconds ?? 0; + if (min > 0) { + const r = Math.min(1, Math.max(0, floor.random ?? Math.random())); + sleep = Math.max(min, sleep) * (1 + 0.5 * r); + } const extra = Math.max(0, paceSeconds - baseSeconds); return Math.round((sleep + extra) * 1000); } diff --git a/common/lib/bitchute.test.ts b/common/lib/bitchute.test.ts @@ -0,0 +1,117 @@ +// BitChute as a platform: detection, ids, the extractor's label, the playable +// file, the queue, moment links and the "auto" format. Synthetic ids only. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/bitchute.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + PLATFORM_VALUES, + defaultWebpageUrl, + detectPlatform, + isSocialPlatform, + queueKeyForUrl, +} from "./platform"; +import { extractVideoId } from "./videoId"; +import { bitchutePlayableUrl } from "./bitchute"; +import { platformFromMetadata, summarize } from "./transcripts-server"; +import { platformMomentBaseUrl, platformMomentUrl } from "./momentUrl"; +import { resolveDownloadFormatSelector } from "../ytdlp/downloadFormat"; +import { dataDirIdForUrl } from "../ytdlp/runYtdlp"; + +const ID = "Zq3xVb7Kp2Lm"; +const PAGE = `https://www.bitchute.com/video/${ID}/`; +const FILE = `https://seed901.bitchute.com/AbCdEfGhIjKl/${ID}.mp4`; + +test("bitchute is a video platform", () => { + assert.ok(PLATFORM_VALUES.includes("bitchute")); + assert.equal(isSocialPlatform("bitchute"), false); +}); + +test("every BitChute host is detected as bitchute; look-alikes are not", () => { + for (const url of [ + PAGE, + `https://bitchute.com/video/${ID}/`, + `https://old.bitchute.com/video/${ID}/`, + `https://www.bitchute.com/embed/${ID}/`, + "https://www.bitchute.com/channel/examplechannel/", + "https://api.bitchute.com/api/beta/video", + FILE, + ]) { + assert.equal(detectPlatform(url), "bitchute", url); + } + assert.equal(detectPlatform("https://notbitchute.com/video/x/"), null); + assert.equal(detectPlatform("https://www.youtube.com/watch?v=AbC123xyz_9"), "youtube"); +}); + +test("a video or embed URL gives the video id; a channel, playlist or profile gives none", () => { + assert.equal(extractVideoId(PAGE), ID); + assert.equal(extractVideoId(`https://www.bitchute.com/video/${ID}`), ID); + assert.equal(extractVideoId(`https://old.bitchute.com/video/${ID}/?list=x`), ID); + assert.equal(extractVideoId(`https://www.bitchute.com/embed/${ID}/`), ID); + assert.equal(extractVideoId("https://www.bitchute.com/channel/examplechannel/"), null); + assert.equal(extractVideoId("https://www.bitchute.com/playlist/Pl4yL1st0001/"), null); + assert.equal(extractVideoId("https://www.bitchute.com/profile/Pr0f1le00001/"), null); + // A single-video spawn pins its output to data/<id>/. + assert.equal(dataDirIdForUrl(PAGE), ID); +}); + +test("the canonical page is /video/<id>/, and BitChute has its own queue", () => { + assert.equal(defaultWebpageUrl("bitchute", ID), PAGE); + assert.equal(queueKeyForUrl(PAGE), "platform:bitchute"); + assert.equal(queueKeyForUrl("https://www.bitchute.com/channel/examplechannel/"), "platform:bitchute"); +}); + +test("yt-dlp's BitChute extractors label a record bitchute, as does a BitChute page", () => { + assert.equal(platformFromMetadata({ extractor_key: "BitChute" }), "bitchute"); + assert.equal(platformFromMetadata({ extractor_key: "BitChuteChannel" }), "bitchute"); + assert.equal(platformFromMetadata({ extractor: "BitChute" }), "bitchute"); + assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: PAGE }), "bitchute"); + // An unknown host still lands where it always has. + assert.equal(platformFromMetadata({ extractor_key: "Generic", webpage_url: "https://example.com/a.mp4" }), "youtube"); +}); + +test("the playable file is the offered mp4 on a bitchute.com host", () => { + assert.equal(bitchutePlayableUrl({ formats: [{ format_id: "0", ext: "mp4", url: FILE }] } as never), FILE); + // The top-level url is the fallback. + assert.equal(bitchutePlayableUrl({ url: FILE, ext: "mp4" }), FILE); + // A live stream's manifest, or a file on another host, is not played. + assert.equal(bitchutePlayableUrl({ formats: [{ ext: "mp4", url: "https://seed901.bitchute.com/x/live.m3u8" }] }), undefined); + assert.equal(bitchutePlayableUrl({ formats: [{ ext: "mp4", url: "https://cdn.example.com/a.mp4" }] }), undefined); + assert.equal(bitchutePlayableUrl({}), undefined); +}); + +test("summarize: a BitChute record keeps its id and page and carries the file to play", () => { + const s = summarize("example-channel", ID, { + id: ID, + title: "A synthetic upload", + upload_date: "20240102", + duration: 61, + extractor: "BitChute", + extractor_key: "BitChute", + webpage_url: PAGE, + formats: [{ format_id: "0", ext: "mp4", url: FILE }], + url: FILE, + }); + assert.equal(s.platform, "bitchute"); + assert.equal(s.id, ID); + assert.equal(s.slug, `example-channel/${ID}`); + assert.equal(s.webpageUrl, PAGE); + assert.equal(s.mediaUrl, FILE); + // No page URL on record: the canonical one. + const bare = summarize("example-channel", ID, { id: ID, upload_date: "20240102", extractor_key: "BitChute" }); + assert.equal(bare.webpageUrl, PAGE); + assert.equal(bare.mediaUrl, undefined); +}); + +test("a moment on BitChute links the bare page: its watch page takes no start time", () => { + assert.equal(platformMomentUrl(PAGE, "bitchute", 125), PAGE); + assert.equal(platformMomentUrl(PAGE, null, 125), PAGE); + assert.equal(platformMomentBaseUrl(PAGE, "bitchute"), null); +}); + +test('"auto" takes BitChute\'s one file', () => { + // BitChute offers one format (an mp4 with no codec fields): `bestaudio` + // matches nothing and `worst` takes that file. + assert.equal(resolveDownloadFormatSelector("auto", "bitchute"), "bestaudio/worst"); +}); diff --git a/common/lib/bitchute.ts b/common/lib/bitchute.ts @@ -0,0 +1,40 @@ +// BitChute records — the pure rules that are BitChute's own. +// +// A BitChute video is fetched by yt-dlp's BitChute extractor, which offers ONE +// format: the uploaded mp4, served as a plain file from a `seedNNN.bitchute.com` +// media host (`https://seed123.bitchute.com/<channel hash>/<id>.mp4`). That file +// is what the viewer plays: BitChute's embed (bitchute.com/embed/<id>/) takes +// no start time it honours, and a native <video> over the file seeks +// (common/components/FilePlayer.tsx), as archive.org records do. + +type FormatLike = { url?: unknown; ext?: unknown; protocol?: unknown }; + +const MEDIA_HOST_RE = /^https:\/\/[a-z0-9-]+\.bitchute\.com\//i; +const PLAYABLE_EXT_RE = /\.(mp4|webm|m4v)(?:[?#]|$)/i; + +function playable(url: unknown, ext: unknown, protocol?: unknown): url is string { + if (typeof url !== "string" || !MEDIA_HOST_RE.test(url)) return false; + // An HLS manifest carries ext "mp4" too; it is a playlist, not a file. + if (/\.m3u8(?:[?#]|$)/i.test(url)) return false; + if (typeof protocol === "string" && protocol.startsWith("m3u8")) return false; + if (typeof ext === "string" && /^(mp4|webm|m4v)$/i.test(ext)) return true; + return PLAYABLE_EXT_RE.test(url); +} + +// The file a BitChute record's player plays: the first offered format that is +// a plain mp4/webm on a bitchute.com host, else the record's top-level `url` +// (yt-dlp writes the chosen format's there). An HLS manifest (a live stream) +// is not a file; undefined when there is none. +export function bitchutePlayableUrl(meta: { + formats?: unknown; + url?: unknown; + ext?: unknown; + protocol?: unknown; +}): string | undefined { + const formats = Array.isArray(meta.formats) ? (meta.formats as FormatLike[]) : []; + for (const f of formats) { + if (f && playable(f.url, f.ext, f.protocol)) return f.url as string; + } + if (playable(meta.url, meta.ext, meta.protocol)) return meta.url as string; + return undefined; +} diff --git a/common/lib/detectPlatform.mjs b/common/lib/detectPlatform.mjs @@ -54,6 +54,9 @@ export function detectPlatform(url) { // archive.org ITEMS only. Not web.archive.org (the Wayback Machine's page // captures are a different kind of record) — see lib/archiveOrgId.ts. if (host === "archive.org" || host === "www.archive.org") return "archiveorg"; + // bitchute.com, www. and old. (the pages), api. and the seedNNN. media + // hosts — every host BitChute serves from is the one platform's. + if (host === "bitchute.com" || host.endsWith(".bitchute.com")) return "bitchute"; if (XENFORO_HOSTS.some((h) => host === h || host.endsWith(`.${h}`))) { return "xenforo"; } diff --git a/common/lib/duplicates.ts b/common/lib/duplicates.ts @@ -263,6 +263,7 @@ const PLATFORM_PREFERENCE: ReadonlyArray<string> = [ "odysee", "kick", "twitch", + "bitchute", ]; function platformRank(platform: string): number { diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts @@ -44,7 +44,8 @@ function twitchTime(totalSeconds: number): string { // Best-effort deep link into a platform's own watch page at `seconds`. Mirrors // the per-platform time params the in-app players build. Platforms whose watch -// page has no reliable start param (Rumble, Kick) get the bare `webpageUrl`. +// page has no reliable start param (Rumble, Kick, archive.org, BitChute) get +// the bare `webpageUrl`. // Returns null only when there is no `webpageUrl` to work from. export function platformMomentUrl( webpageUrl: string | null | undefined, @@ -71,10 +72,11 @@ export function platformMomentUrl( case "twitch": u.searchParams.set("t", twitchTime(secs)); return u.toString(); - // Rumble / Kick / archive.org / unknown: the watch page has no dependable - // start param — return the plain webpage URL rather than an invalid seek. - // (archive.org's own player takes none we can rely on; the archive's - // viewer plays the file itself and seeks it — PlayerProvider.) + // Rumble / Kick / archive.org / BitChute / unknown: the watch page has no + // dependable start param — return the plain webpage URL rather than an + // invalid seek. (archive.org's and BitChute's own players take none we can + // rely on; the archive's viewer plays the file itself and seeks it — + // PlayerProvider.) default: return webpageUrl; } @@ -166,7 +168,7 @@ export function viewerMomentBaseUrl( // webpage URL), a base MUST be appendable — so only platforms whose time param // takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs` // token, so appending an integer would be an invalid seek); Rumble/Kick/ -// archive.org/unknown have no dependable start param at all. Null in every +// archive.org/BitChute/unknown have no dependable start param at all. Null in every // non-appendable case. export function platformMomentBaseUrl( webpageUrl: string | null | undefined, diff --git a/common/lib/platform.ts b/common/lib/platform.ts @@ -11,6 +11,8 @@ export type Platform = // archive.org items (lib/archiveOrgId.ts): a whole item, or one file inside // a multi-file item. | "archiveorg" + // BitChute videos (bitchute.com/video/<id>/) and channels. + | "bitchute" | "twitter" | "bluesky" | "xenforo"; @@ -22,6 +24,7 @@ export const PLATFORM_VALUES: ReadonlyArray<Platform> = [ "twitch", "kick", "archiveorg", + "bitchute", "twitter", "bluesky", "xenforo", @@ -63,6 +66,8 @@ export function defaultWebpageUrl(platform: Platform, id: string): string { const file = /^(.+?)__.*-[0-9a-f]{8}$/.exec(id); return `https://archive.org/details/${file ? file[1] : id}`; } + // BitChute's canonical id is its video id, which is also yt-dlp's. + if (platform === "bitchute") return `https://www.bitchute.com/video/${id}/`; // Social posts: /i/status/<id> resolves without knowing the handle. Bluesky // has no handle-free permalink, so this is only a last-resort fallback — // every archived post carries its own canonical `url` (see postPermalink). diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -3,6 +3,7 @@ import path from "node:path"; import { formatDate, formatDuration } from "./format"; import { defaultWebpageUrl, detectPlatform } from "./platform"; import { archiveOrgPlayableUrl } from "./archiveOrg"; +import { bitchutePlayableUrl } from "./bitchute"; import { archiveOrgVideoIdFromNativeId } from "./archiveOrgId"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { MediaType, VideoStat, VideoStatus } from "./stats"; @@ -41,10 +42,14 @@ export type RawMetadata = { // yt-dlp's coarse kind: "video" | "livestream" | "short" (YouTube). Absent on // platforms that don't distinguish, where we fall back to is/was_live. media_type?: string; - // Every format the extractor offered. Read only for archive.org, where each - // is a plain download URL and one of them is what the player plays - // (lib/archiveOrg.ts archiveOrgPlayableUrl). + // Every format the extractor offered. Read only for archive.org and + // BitChute, where each is a plain download URL and one of them is what the + // player plays (lib/archiveOrg.ts archiveOrgPlayableUrl, lib/bitchute.ts). formats?: unknown; + // The chosen format's URL, which yt-dlp writes at the top level. Read only + // for BitChute, as the playable file's fallback. + url?: unknown; + ext?: unknown; }; // Read and parse a video's metadata.info.json into the typed RawMetadata @@ -86,9 +91,11 @@ export function platformFromMetadata(meta: RawMetadata): Platform { if (/^twitch/i.test(key)) return "twitch"; if (/^kick/i.test(key)) return "kick"; if (/^archive\.?org$/i.test(key)) return "archiveorg"; + // `BitChute` (a video) and `BitChuteChannel` (a listing). + if (/^bitchute/i.test(key)) return "bitchute"; if (/^youtube/i.test(key)) return "youtube"; const fromPage = detectPlatform(meta.webpage_url); - if (fromPage === "archiveorg") return fromPage; + if (fromPage === "archiveorg" || fromPage === "bitchute") return fromPage; return "youtube"; } @@ -141,10 +148,13 @@ export function summarize( webpageUrl: meta.webpage_url ?? defaultWebpageUrl(platform, id), // Kick VODs play from a persisted HLS manifest (no iframe embed exists). hlsUrl: platform === "kick" ? meta.manifest_url : undefined, - // archive.org plays the file itself in a native <video>, which seeks. + // archive.org and BitChute play the file itself in a native <video>, + // which seeks. ...(platform === "archiveorg" ? { mediaUrl: archiveOrgPlayableUrl(meta) } - : {}), + : platform === "bitchute" + ? { mediaUrl: bitchutePlayableUrl(meta) } + : {}), }; } diff --git a/common/lib/videoId.ts b/common/lib/videoId.ts @@ -44,6 +44,12 @@ export function extractVideoId(url: string): string | null { if (lastColon > 0) return decoded.slice(lastColon + 1); return null; } + if (host === "bitchute.com" || host.endsWith(".bitchute.com")) { + // A video page or its embed: /video/<id>/, /embed/<id>/ (www. or old.). + // A channel, playlist or profile page names no video. + const segs = u.pathname.split("/").filter(Boolean); + return (segs[0] === "video" || segs[0] === "embed") && segs[1] ? segs[1] : null; + } if (host.endsWith("twitch.tv")) { // VODs: /videos/<id> or legacy /<channel>/v/<id>; clips: // clips.twitch.tv/<slug> or /<channel>/clip/<slug>. Grab the segment diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts @@ -5,7 +5,10 @@ import { channelPaceSeconds, pacedPlatformArgs, platformArgs, + platformArgsForUrl, PLATFORM_ARGS, + PLATFORM_MIN_GAP_SECONDS, + platformMinGapSeconds, staticSleepRequestsSeconds, withSleepRequests, } from "./channelArgs"; @@ -164,3 +167,21 @@ test("archive.org is paced and backs off, and keeps the fetched page as webpage_ // No parallel transfer is ever asked for. assert.equal(args.some((a) => /concurrent|downloader|^-N$/.test(a)), false); }); + +test("BitChute is paced hardest: 3 s between requests, exponential retry sleeps, a 60 s floor between videos", () => { + const args = platformArgs("bitchute"); + assert.equal(staticSleepRequestsSeconds("bitchute"), 3); + assert.deepEqual(args.slice(0, 2), ["--sleep-requests", "3"]); + assert.ok(args.includes("http:exp=2:120")); + assert.ok(args.includes("extractor:exp=2:120")); + assert.deepEqual(platformArgsForUrl("https://www.bitchute.com/video/Zq3xVb7Kp2Lm/"), args); + // No impersonation, and no parallel transfer is ever asked for. + assert.equal(args.includes("--impersonate"), false); + assert.equal(args.some((a) => /concurrent|downloader|^-N$/.test(a)), false); + assert.equal(platformMinGapSeconds("bitchute"), 60); + assert.equal(PLATFORM_MIN_GAP_SECONDS.bitchute, 60); + // Every other platform keeps the operator's gap. + for (const p of ["youtube", "rumble", "odysee", "archiveorg", "unknown", null]) { + assert.equal(platformMinGapSeconds(p), 0, String(p)); + } +}); diff --git a/common/ytdlp/channelArgs.ts b/common/ytdlp/channelArgs.ts @@ -39,6 +39,8 @@ export { PLATFORM_ARGS, platformArgs, platformArgsForUrl, + PLATFORM_MIN_GAP_SECONDS, + platformMinGapSeconds, staticSleepRequestsSeconds, withSleepRequests, } from "./platformArgs.mjs"; diff --git a/common/ytdlp/downloadFormat.ts b/common/ytdlp/downloadFormat.ts @@ -6,7 +6,8 @@ import type { Platform } from "../lib/platform"; // `original` format (every HLS rung is CDN-truncated to a few minutes), so auto // prefers `original` there, archive.org gets the uploader's original file // (ARCHIVE_ORG_AUTO_FORMAT_SELECTOR), and the historical `bestaudio/worst` -// everywhere else. +// everywhere else — BitChute included, whose one format (an mp4 with no codec +// fields) `bestaudio` never matches and `worst` takes. export type DownloadFormatPreset = | "auto" | "original" diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs @@ -48,6 +48,16 @@ import { detectPlatform } from "../lib/detectPlatform.mjs"; // (reconcileVideoDirs.ts) — the file's record would be merged into the item's. // The import always fetches by the canonical file page, so the two agree; the // regex only takes an archive.org URL, and leaves any other untouched. +// +// bitchute: BitChute rate-limits (its API answered HTTP 429 on 2026-10-06), so +// it is paced harder than any other platform: `--sleep-requests 3` between +// the extractor's requests (two API calls per video, then a HEAD per media +// host it tries), and the same exponential `--retry-sleep` as archive.org so a +// refused or dropped request waits 2 s doubling to 120 s instead of retrying +// at once. A video is one plain mp4 over one HTTP stream — no fragments, no +// parallel ranges, and nothing here passes `-N` or an external downloader. No +// `--impersonate`: the extractor's API answers without a browser fingerprint. +// Between two VIDEOS the floor is PLATFORM_MIN_GAP_SECONDS below. /** @type {Readonly<Partial<Record<Platform, readonly string[]>>>} */ export const PLATFORM_ARGS = Object.freeze({ rumble: Object.freeze(["--impersonate", "chrome", "--sleep-requests", "1"]), @@ -62,9 +72,40 @@ export const PLATFORM_ARGS = Object.freeze({ "--parse-metadata", "original_url:(?P<webpage_url>https://archive\\.org/(?:details|embed|download)/.+)", ]), + bitchute: Object.freeze([ + "--sleep-requests", + "3", + "--retry-sleep", + "http:exp=2:120", + "--retry-sleep", + "extractor:exp=2:120", + ]), +}); + +// THE FLOOR UNDER THE GAP BETWEEN TWO VIDEOS on a platform, in seconds — for a +// platform that must be asked less often than the operator's +// `sleepBetweenDownloadsSeconds` would ask it. Every gap a batch download, a +// sync, a persist or the auto-download lane waits on that platform is at least +// this, plus up to half again at random so a run never settles into a fixed +// beat (jobs/platformBackoff.ts, downloadGapMs). A platform absent here keeps +// the operator's gap exactly, unjittered. +/** @type {Readonly<Partial<Record<Platform, number>>>} */ +export const PLATFORM_MIN_GAP_SECONDS = Object.freeze({ + bitchute: 60, }); /** + * @param {string | null | undefined} platform + * @returns {number} + */ +export function platformMinGapSeconds(platform) { + if (!platform) return 0; + /** @type {Record<string, number | undefined>} */ + const table = PLATFORM_MIN_GAP_SECONDS; + return table[platform] ?? 0; +} + +/** * @param {Platform | null | undefined} platform * @returns {string[]} */ diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -19,6 +19,7 @@ import { channelPaceSeconds, channelPlatform, pacedPlatformArgs, + platformMinGapSeconds, staticSleepRequestsSeconds, } from "./channelArgs"; import { isRealAudioFile } from "../lib/videoStatus"; @@ -1077,6 +1078,9 @@ export async function runManagedDownloads( // rate limit doubled the platform's pace, a batch spaces its videos further // apart too. Read per gap, so a 429 in this batch slows the rest of it. const basePace = staticSleepRequestsSeconds(channelPlatform(effectiveChannelConfig)); + // A platform's floor under the gap (BitChute's 60 s, jittered) — see + // PLATFORM_MIN_GAP_SECONDS in platformArgs.mjs. + const minGap = platformMinGapSeconds(channelPlatform(effectiveChannelConfig)); const paceNow = deps.paceSeconds ?? (() => channelPaceSeconds(effectiveChannelConfig)); const recordSubs = deps.recordSubtitleDeferral ?? @@ -1223,7 +1227,9 @@ export async function runManagedDownloads( // declined slept 30 s after nothing but its metadata prefetch. Every // other outcome still sleeps — a real fetch, success or failure, and // every failure, per-video ones included (see declinedWithoutMediaFetch). - const gapMs = downloadGapMs(sleepSeconds, paceNow(), basePace); + const gapMs = downloadGapMs(sleepSeconds, paceNow(), basePace, { + minSeconds: minGap, + }); if ( gapMs > 0 && !isLast && @@ -1465,6 +1471,7 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { opts.channelConfig.sleepBetweenDownloadsSeconds ?? getSettings().sleepBetweenDownloadsSeconds; const subsBasePace = staticSleepRequestsSeconds(channelPlatform(opts.channelConfig)); + const subsMinGap = platformMinGapSeconds(channelPlatform(opts.channelConfig)); await Promise.all( tofetch.map((url, index) => limit(async () => { @@ -1475,6 +1482,7 @@ async function downloadMissingSubs(opts: RunYtdlpOpts): Promise<void> { subsSleepSeconds, channelPaceSeconds(opts.channelConfig), subsBasePace, + { minSeconds: subsMinGap }, ); if (gapMs > 0) { opts.onLog(`Sleeping ${gapMs / 1000}s before the next video...\n`);