// The per-platform yt-dlp args table — THE one copy. // // Plain JS (with JSDoc types) so umtool's report-to-video scripts, which are // `.mjs` run by bare `node` (20 in the runtime image — no type stripping), can // import it: they spawn yt-dlp for clip fetches and availability checks and hit // the same Cloudflare wall. `ytdlp/channelArgs.ts` re-exports it for every TS // caller. Never copy these args into another file. import { detectPlatform } from "../lib/detectPlatform.mjs"; /** @typedef {import("../lib/platform").Platform} Platform */ // Args a platform needs on EVERY yt-dlp spawn, whatever the job. Code, not a // setting: per-channel `ytdlpExtraArgs` is the tweak surface, and it comes // AFTER these, so a channel override wins (yt-dlp is last-flag-wins for // `--sleep-requests` and `--impersonate`). // // Every entry paces at the REQUEST level with `--sleep-requests 1`: one // second between the HTTP requests a single yt-dlp process makes (listing // pages, the prefetch, subtitle fetches, retries). That is a different layer // from `sleepBetweenDownloadsSeconds`, which only spaces whole per-video // downloads apart. A spawn that already passes its own `--sleep-requests 1` // (metadataScan.ts, fetchWindowManaged.ts) now carries it twice — harmless, // yt-dlp keeps the last. // // rumble: every request 403s at Cloudflare without a browser TLS fingerprint // (yt-dlp #17496); a probe on 2026-09-24 got 200 with `--impersonate chrome` // and 403 without. `--sleep-requests 1` paces the listing walk — the // the-quartering-rumble full sweep 429'd at page 155 unpaced the same day. // // youtube: the 2026-09-25 429 investigation // (~/reports/release-7/data/q-429-report.md, finding 1) found YouTube had NO // request-level pacing at all — each video attempt fires 2–4 requests // (prefetch, subtitles, retries) back to back, and the two channels that 429'd // were the two most-downloaded. Mirrors rumble's pace; a channel's own // `ytdlpExtraArgs` still wins because it comes after. // // archiveorg: be polite to archive.org (a non-profit serving files from its // own disks). `--sleep-requests 2` spaces the extractor's requests (the embed // page, then the metadata API); `--retry-sleep` turns yt-dlp's immediate // retries of a refused request or a dropped transfer into an exponential wait // (2 s doubling to 120 s), and the stock retry counts bound how many. yt-dlp // downloads an archive.org file as ONE plain HTTP stream (no fragments, no // parallel ranges), and nothing here passes `-N` or an external downloader. // `--parse-metadata` makes the record's `webpage_url` the URL it was fetched // by: for one file of a multi-file item yt-dlp writes the ITEM's page there, // and the snapshot renames every video dir to `extractVideoId(webpage_url)` // (reconcileVideoDirs.ts) — the file's record would be merged into the item's. // The import always fetches by the canonical file page, so the two agree; the // regex only takes an archive.org URL, and leaves any other untouched. // // bitchute: BitChute rate-limits (its API answered HTTP 429 on 2026-10-06), so // it is paced harder than any other platform: `--sleep-requests 3` between // the extractor's requests (two API calls per video, then a HEAD per media // host it tries), and the same exponential `--retry-sleep` as archive.org so a // refused or dropped request waits 2 s doubling to 120 s instead of retrying // at once. A video is one plain mp4 over one HTTP stream — no fragments, no // parallel ranges, and nothing here passes `-N` or an external downloader. No // `--impersonate`: the extractor's API answers without a browser fingerprint. // Between two VIDEOS the floor is PLATFORM_MIN_GAP_SECONDS below. /** @type {Readonly>>} */ export const PLATFORM_ARGS = Object.freeze({ rumble: Object.freeze(["--impersonate", "chrome", "--sleep-requests", "1"]), youtube: Object.freeze(["--sleep-requests", "1"]), archiveorg: Object.freeze([ "--sleep-requests", "2", "--retry-sleep", "http:exp=2:120", "--retry-sleep", "extractor:exp=2:120", "--parse-metadata", "original_url:(?Phttps://archive\\.org/(?:details|embed|download)/.+)", ]), bitchute: Object.freeze([ "--sleep-requests", "3", "--retry-sleep", "http:exp=2:120", "--retry-sleep", "extractor:exp=2:120", ]), }); // THE FLOOR UNDER THE GAP BETWEEN TWO VIDEOS on a platform, in seconds — for a // platform that must be asked less often than the operator's // `sleepBetweenDownloadsSeconds` would ask it. Every gap a batch download, a // sync, a persist or the auto-download lane waits on that platform is at least // this, plus up to half again at random so a run never settles into a fixed // beat (jobs/platformBackoff.ts, downloadGapMs). A platform absent here keeps // the operator's gap exactly, unjittered. /** @type {Readonly>>} */ export const PLATFORM_MIN_GAP_SECONDS = Object.freeze({ bitchute: 60, }); // THE FLOOR UNDER THE GAP BETWEEN TWO ONE-OFF IMPORTS on a platform, in // seconds, where it must be higher than the batch floor above. Each import is // its own job, so nothing else spaces them: Odysee's API answered HTTP 429 to // imports queued 10–50 s apart (2026-10-06). An import waits the larger of // this and PLATFORM_MIN_GAP_SECONDS after the last import there, plus up to // half again at random (jobs/platformGap.ts). Batch downloads and syncs of an // Odysee channel keep the operator's gap. /** @type {Readonly>>} */ export const PLATFORM_IMPORT_MIN_GAP_SECONDS = Object.freeze({ odysee: 60, }); /** * @param {string | null | undefined} platform * @returns {number} */ export function platformMinGapSeconds(platform) { if (!platform) return 0; /** @type {Record} */ const table = PLATFORM_MIN_GAP_SECONDS; return table[platform] ?? 0; } /** * @param {string | null | undefined} platform * @returns {number} */ export function platformImportMinGapSeconds(platform) { if (!platform) return 0; /** @type {Record} */ const table = PLATFORM_IMPORT_MIN_GAP_SECONDS; return Math.max(platformMinGapSeconds(platform), table[platform] ?? 0); } /** * @param {Platform | null | undefined} platform * @returns {string[]} */ export function platformArgs(platform) { if (!platform) return []; return [...(PLATFORM_ARGS[platform] ?? [])]; } /** * The args for whatever platform `url` is on — for a spawn that has a URL and * no channel config (umtool's clip fetch and availability check). * @param {string | undefined | null} url * @returns {string[]} */ export function platformArgsForUrl(url) { return platformArgs(detectPlatform(url)); } // THE STATIC PACE a platform's spawns carry (`--sleep-requests` in its // PLATFORM_ARGS entry), or 0 for a platform with none. The adaptive pace // (common/jobs/platformBackoff.ts, release 17 slice RL) starts here, doubles on // a rate limit and decays back to it. /** * @param {string | null | undefined} platform * @returns {number} */ export function staticSleepRequestsSeconds(platform) { if (!platform) return 0; /** @type {Record} */ const table = PLATFORM_ARGS; const args = table[platform]; if (!args) return 0; const i = args.indexOf("--sleep-requests"); const v = i >= 0 ? Number(args[i + 1]) : 0; return Number.isFinite(v) && v > 0 ? v : 0; } // `args` with its `--sleep-requests` raised to `seconds` — replaced when the // args carry a lower one, appended when they carry none. Never LOWERS a pace: // a value below what the args already say is ignored. Returns a copy. /** * @param {readonly string[]} args * @param {number | undefined | null} seconds * @returns {string[]} */ export function withSleepRequests(args, seconds) { const out = [...args]; if (seconds == null || !Number.isFinite(seconds) || seconds <= 0) return out; const i = out.indexOf("--sleep-requests"); if (i < 0) { out.push("--sleep-requests", String(seconds)); return out; } const cur = Number(out[i + 1]); if (!Number.isFinite(cur) || seconds > cur) out[i + 1] = String(seconds); return out; }