commit db6c5b798b9e2ac78295fa7a1207252d1bdbdbaa
parent 6198bcebf6756650febb7eb5f4d848158bee0e00
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 11:50:29 -0400
Merge sources/odysee-gap (a one-off Odysee or BitChute import waits a floor after the last one there; the BitChute import spec's refresh race)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
8 files changed, 264 insertions(+), 18 deletions(-)
diff --git a/common/jobs/platformGap.test.ts b/common/jobs/platformGap.test.ts
@@ -0,0 +1,72 @@
+import { test, beforeEach } from "node:test";
+import assert from "node:assert/strict";
+import {
+ clearPlatformGaps,
+ notePlatformGap,
+ platformGapRemainingMs,
+ waitForPlatformGap,
+} from "./platformGap";
+
+beforeEach(() => clearPlatformGaps());
+
+test("a platform with no import yet is asked at once", () => {
+ assert.equal(platformGapRemainingMs("odysee", 1_000), 0);
+});
+
+test("the next import waits the gap from when the last one settled, per platform", () => {
+ notePlatformGap("odysee", 60_000, 1_000);
+ assert.equal(platformGapRemainingMs("odysee", 1_000), 60_000);
+ assert.equal(platformGapRemainingMs("odysee", 31_000), 30_000);
+ assert.equal(platformGapRemainingMs("odysee", 61_000), 0);
+ assert.equal(platformGapRemainingMs("bitchute", 1_000), 0);
+});
+
+test("a later start is never shortened by a shorter gap", () => {
+ notePlatformGap("odysee", 90_000, 0);
+ notePlatformGap("odysee", 10_000, 5_000);
+ assert.equal(platformGapRemainingMs("odysee", 5_000), 85_000);
+});
+
+test("the gap lives on globalThis, so every bundle shares it", () => {
+ notePlatformGap("odysee", 1_000, 0);
+ assert.equal(globalThis.__yttPlatformGap__?.get("odysee"), 1_000);
+});
+
+test("waiting re-reads the remaining time after each sleep and logs each wait", async () => {
+ const left = [45_000, 2_000, 0];
+ const slept: number[] = [];
+ const lines: string[] = [];
+ const waited = await waitForPlatformGap({
+ label: "Odysee",
+ remainingMs: () => left.shift() ?? 0,
+ onLog: (l) => lines.push(l),
+ sleep: async (ms) => {
+ slept.push(ms);
+ },
+ });
+ assert.deepEqual(slept, [45_000, 2_000]);
+ assert.equal(waited, 47_000);
+ assert.deepEqual(lines, [
+ "Odysee asks for a gap between videos: waiting 45s.",
+ "Odysee asks for a gap between videos: waiting 2s.",
+ ]);
+});
+
+test("nothing to wait: no sleep, no line", async () => {
+ const lines: string[] = [];
+ const waited = await waitForPlatformGap({
+ label: "Odysee",
+ remainingMs: async () => 0,
+ onLog: (l) => lines.push(l),
+ sleep: async () => assert.fail("must not sleep"),
+ });
+ assert.equal(waited, 0);
+ assert.deepEqual(lines, []);
+});
+
+test("an aborted job stops waiting", async () => {
+ const ac = new AbortController();
+ const p = waitForPlatformGap({ label: "Odysee", remainingMs: () => 60_000, signal: ac.signal });
+ ac.abort();
+ await assert.rejects(p, { name: "AbortError" });
+});
diff --git a/common/jobs/platformGap.ts b/common/jobs/platformGap.ts
@@ -0,0 +1,82 @@
+// THE GAP BETWEEN TWO ONE-OFF IMPORTS ON ONE PLATFORM. A batch download, a
+// sync, a persist and the auto-download lane each wait their platform's floor
+// between two videos — but every import is its own job, so ten imports queued
+// on `platform:odysee` used to ask Odysee as fast as each job could finish
+// (Odysee answered 429 at 10–50 s apart, 2026-10-06). An import on a platform
+// with an import floor (platformImportMinGapSeconds, ytdlp/platformArgs.mjs)
+// waits here first, and sets the platform's next start when it settles.
+//
+// Process-wide (on globalThis, as the job registry is: a route bundle and the
+// job runner must share one view) and in memory: a restart is itself a gap.
+
+declare global {
+ // eslint-disable-next-line no-var
+ var __yttPlatformGap__: Map<string, number> | undefined;
+}
+
+function nextStartAt(): Map<string, number> {
+ if (!globalThis.__yttPlatformGap__) globalThis.__yttPlatformGap__ = new Map();
+ return globalThis.__yttPlatformGap__;
+}
+
+// Milliseconds before `platform` may be asked again (0 = now).
+export function platformGapRemainingMs(platform: string, now: number = Date.now()): number {
+ const at = nextStartAt().get(platform);
+ return at !== undefined && at > now ? at - now : 0;
+}
+
+// A unit on `platform` settled at `now`: the next one starts no sooner than
+// `gapMs` later. A later start already set is kept (never shortened).
+export function notePlatformGap(platform: string, gapMs: number, now: number = Date.now()): void {
+ if (gapMs <= 0) return;
+ const at = now + gapMs;
+ const map = nextStartAt();
+ if ((map.get(platform) ?? 0) < at) map.set(platform, at);
+}
+
+export function clearPlatformGaps(): void {
+ nextStartAt().clear();
+}
+
+// Wait out `remainingMs()` — re-read after each wait, since another import or
+// a 429 may have moved it — logging once per wait. Rejects with an AbortError
+// when `signal` aborts.
+export async function waitForPlatformGap(opts: {
+ label: string;
+ remainingMs: () => Promise<number> | number;
+ signal?: AbortSignal;
+ onLog?: (line: string) => void;
+ sleep?: (ms: number, signal?: AbortSignal) => Promise<void>;
+}): Promise<number> {
+ const sleep = opts.sleep ?? abortableSleep;
+ let waited = 0;
+ for (;;) {
+ if (opts.signal?.aborted) throw abortError();
+ const ms = await opts.remainingMs();
+ if (ms <= 0) return waited;
+ opts.onLog?.(`${opts.label} asks for a gap between videos: waiting ${Math.ceil(ms / 1000)}s.`);
+ await sleep(ms, opts.signal);
+ waited += ms;
+ }
+}
+
+function abortError(): Error {
+ const err = new Error("aborted while waiting for the platform gap");
+ err.name = "AbortError";
+ return err;
+}
+
+function abortableSleep(ms: number, signal?: AbortSignal): Promise<void> {
+ return new Promise((resolve, reject) => {
+ if (signal?.aborted) return reject(abortError());
+ const t = setTimeout(() => {
+ signal?.removeEventListener("abort", onAbort);
+ resolve();
+ }, ms);
+ const onAbort = () => {
+ clearTimeout(t);
+ reject(abortError());
+ };
+ signal?.addEventListener("abort", onAbort, { once: true });
+ });
+}
diff --git a/common/ytdlp/channelArgs.test.ts b/common/ytdlp/channelArgs.test.ts
@@ -6,6 +6,8 @@ import {
pacedPlatformArgs,
platformArgs,
platformArgsForUrl,
+ platformImportMinGapSeconds,
+ PLATFORM_IMPORT_MIN_GAP_SECONDS,
PLATFORM_ARGS,
PLATFORM_MIN_GAP_SECONDS,
platformMinGapSeconds,
@@ -185,3 +187,14 @@ test("BitChute is paced hardest: 3 s between requests, exponential retry sleeps,
assert.equal(platformMinGapSeconds(p), 0, String(p));
}
});
+
+test("one-off imports have their own floor: Odysee's 60 s, BitChute's batch floor, none elsewhere", () => {
+ assert.equal(PLATFORM_IMPORT_MIN_GAP_SECONDS.odysee, 60);
+ assert.equal(platformImportMinGapSeconds("odysee"), 60);
+ assert.equal(platformImportMinGapSeconds("bitchute"), 60);
+ // Odysee's batch gap is untouched: the import floor is imports only.
+ assert.equal(platformMinGapSeconds("odysee"), 0);
+ for (const p of ["youtube", "rumble", "archiveorg", "unknown", null]) {
+ assert.equal(platformImportMinGapSeconds(p), 0, String(p));
+ }
+});
diff --git a/common/ytdlp/channelArgs.ts b/common/ytdlp/channelArgs.ts
@@ -41,6 +41,8 @@ export {
platformArgsForUrl,
PLATFORM_MIN_GAP_SECONDS,
platformMinGapSeconds,
+ PLATFORM_IMPORT_MIN_GAP_SECONDS,
+ platformImportMinGapSeconds,
staticSleepRequestsSeconds,
withSleepRequests,
} from "./platformArgs.mjs";
diff --git a/common/ytdlp/platformArgs.mjs b/common/ytdlp/platformArgs.mjs
@@ -94,6 +94,18 @@ export const PLATFORM_MIN_GAP_SECONDS = Object.freeze({
bitchute: 60,
});
+// THE FLOOR UNDER THE GAP BETWEEN TWO ONE-OFF IMPORTS on a platform, in
+// seconds, where it must be higher than the batch floor above. Each import is
+// its own job, so nothing else spaces them: Odysee's API answered HTTP 429 to
+// imports queued 10–50 s apart (2026-10-06). An import waits the larger of
+// this and PLATFORM_MIN_GAP_SECONDS after the last import there, plus up to
+// half again at random (jobs/platformGap.ts). Batch downloads and syncs of an
+// Odysee channel keep the operator's gap.
+/** @type {Readonly<Partial<Record<Platform, number>>>} */
+export const PLATFORM_IMPORT_MIN_GAP_SECONDS = Object.freeze({
+ odysee: 60,
+});
+
/**
* @param {string | null | undefined} platform
* @returns {number}
@@ -106,6 +118,17 @@ export function platformMinGapSeconds(platform) {
}
/**
+ * @param {string | null | undefined} platform
+ * @returns {number}
+ */
+export function platformImportMinGapSeconds(platform) {
+ if (!platform) return 0;
+ /** @type {Record<string, number | undefined>} */
+ const table = PLATFORM_IMPORT_MIN_GAP_SECONDS;
+ return Math.max(platformMinGapSeconds(platform), table[platform] ?? 0);
+}
+
+/**
* @param {Platform | null | undefined} platform
* @returns {string[]}
*/
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -4,6 +4,7 @@
- **A video's other English tracks are readable and searchable where their words differ.** Uploaded captions are not always a transcript of what was said, so the tracks beside the transcript stay: the served `en` beside `en-orig`, a regional or auto-translated track, and the captions a local transcription replaced. One is kept where its words differ from the transcript's and from every track kept before it; identical tracks, most of them, add nothing. The index keeps them in an `alts` sub-DB and writes `track` and `altTracks` onto the transcript record only then, so every other record's page is what it was. A search hit in a word only an alternate holds names the track; one every track says is found once, in the transcript. The video page's **Transcript** card reads the transcript and switches tracks ("Track: original audio captions ▾"); switching changes nothing on disk, and **Set as transcript** stays the way the transcript itself changes. English VTTs are no longer shipped as subtitle tracks. One notion of a track — ids, plain labels, which are kept, how a hit across them is found — lives in `common/lib/captionTracks.ts`.
- **The next index build reads the alternate tracks once.** Every record that can hold one — two or more English VTTs, or a transcription beside captions — is re-read from disk, and nothing else; the log says `Alternate tracks v1: N record(s) re-read.` and how many hold a track whose words differ. The version is recorded only when no channel is held. A transcribed video's captions now count toward its change time, so a later caption fetch reaches the index.
- **The MCP reads every English track.** `search_transcripts` and the query-tree tools match a record's alternate tracks and tag a snippet from one (`[in uploaded captions 1:30]`); `get_transcript` names a video's tracks in its header and reads another with `track`; `get_transcripts` windows a match only an alternate holds, under its name; `get_video_metadata` lists the other tracks without their cues. The sweep plan says what such a hit is before it is quoted.
+- **One-off imports from Odysee and BitChute are spaced, as a batch download is.** Each import is its own job, so nothing used to space them: imports queued on Odysee went out as fast as each finished, and Odysee answered 429. An imported Odysee or BitChute URL now waits at least 60 s after the last import from the same platform, plus up to half again at random; the job's log says "Odysee asks for a gap between videos: waiting Ns." An Odysee import, as a BitChute one already did, runs on that platform's own queue whatever the channel's platform, is refused while the platform is held or in a rate-limit cooldown, waits out a cooldown that began after it was queued, and records what the platform answered, so a 429 backs every path on that platform off. Batch downloads and syncs of an Odysee channel keep the operator's gap. The wait is held in memory, and a restart clears it. Needs a restart of the editor.
- **A Wayback Machine capture is a copy, and says of what.** A capture URL (`web.archive.org/web/<timestamp>[id_|im_|…]/<original>`) names its record by what it is a capture of: an archived YouTube page by its YouTube id (no longer `watch`), a JW Player file by its media id (no longer `<id>-<rendition>.mp4`). Every download of a capture writes `wayback.json` (the original URL, the capture's timestamp, the capture page and its raw bytes); the video page says "Archived copy (Wayback Machine, <date>) of <original>"; a citation links the original, marked as possibly gone, and the Wayback copy, and its moment link is the capture, which plays (a capture URL never takes a time param). An existing record whose page is a capture is renamed to its id by the next snapshot.
- **`archilyzer wayback refresh <slug> [--titles <file>] [--dry-run]`** brings a channel's Wayback copies up to that offline: `wayback.json`, the dir renamed through the snapshot's own reconcile pass with its roster entry moved, and with `--titles` (`id → {title, upload_date}`) the title and date of a raw file that has none, recorded in the metadata history as `wayback-provenance`. A record a live job holds is skipped and named; a second run changes nothing.
- **A video's captions are read from its original-audio track first, and a track with no text never hides one that has it.** Where YouTube serves both, `transcript.en-orig.vtt` (the captions of the original audio) is read before `transcript.en.vtt`, whose text can be a rewrite of what was said; then regional tracks (`en-US`, `en-GB`, …), then auto-translated `en-en-*` ones. The transcript is the first track in that order that has cues. One rule (`englishVttsByPreference` / `readEnglishVttCues` in `common/lib/videoStatus.ts`) serves the index, normalize and report compose, so search, the export, the MCP and report videos read the same words. **Set as transcript** on a video's page copies the chosen track to `transcript.en.vtt` and pins it there with `transcript-pin.json`, which ranks it first; deleting `transcript-pin.json` returns the video to the automatic pick. The served `en` track stays readable and searchable as an alternate track where its words differ (below). Videos with a human-made `en` track are not counted as auto-captions-only, so the replace-auto-captions lane still leaves them alone.
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -30,6 +30,13 @@ import {
recordDownloadBackoff,
recordPlatformClean,
} from "yt-dlp-transcript-common/jobs/downloadBackoff";
+import { downloadGapMs } from "yt-dlp-transcript-common/jobs/platformBackoff";
+import {
+ notePlatformGap,
+ platformGapRemainingMs,
+ waitForPlatformGap,
+} from "yt-dlp-transcript-common/jobs/platformGap";
+import { platformImportMinGapSeconds } from "yt-dlp-transcript-common/ytdlp/channelArgs";
import {
countNotYetDownloaded,
readChannelConfig,
@@ -531,6 +538,15 @@ export async function importVideoAction(
// again when on disk — and what BitChute answers is recorded on its pacing
// state below.
const bitchute = urlPlatform === "bitchute";
+ // ANY URL ON A PLATFORM WITH AN IMPORT FLOOR (platformImportMinGapSeconds:
+ // BitChute, Odysee) is paced though each import is its own job: it runs
+ // on the platform's own queue with the platform's args, is refused while the
+ // platform is held or cooling down, waits out the floor after the last import
+ // there (jobs/platformGap.ts) and any cooldown that began since it was
+ // queued, and records what the platform answered.
+ const pacedPlatform =
+ urlPlatform && platformImportMinGapSeconds(urlPlatform) > 0 ? urlPlatform : null;
+ const pacedLabel = pacedPlatform ? (PACED_LABELS[pacedPlatform] ?? pacedPlatform) : "";
let downloadConfig = channelConfig;
if (archiveOrg) {
const refused = await archiveOrgRefusal(paths, "The import");
@@ -555,9 +571,14 @@ export async function importVideoAction(
downloadConfig = { ...channelConfig, platform: "archiveorg" };
}
}
- if (bitchute) {
- const refused = await platformRefusal(paths, "bitchute", "BitChute", "The import");
+ if (pacedPlatform) {
+ const refused = await platformRefusal(paths, pacedPlatform, pacedLabel, "The import");
if (refused) return { ok: false, info: true, error: refused };
+ if (channelConfig.platform !== pacedPlatform) {
+ downloadConfig = { ...channelConfig, platform: pacedPlatform };
+ }
+ }
+ if (bitchute) {
const resolved = resolveBitchuteImportUrl(videoUrl);
if (!resolved.ok) return { ok: false, error: resolved.error };
videoUrl = resolved.url;
@@ -574,9 +595,6 @@ export async function importVideoAction(
error: `Already downloaded: data/${resolved.id}/ — BitChute is not asked for it again.`,
};
}
- if (channelConfig.platform !== "bitchute") {
- downloadConfig = { ...channelConfig, platform: "bitchute" };
- }
}
// Best-effort canonical id: used only for revalidation/labels. When null,
// downloadOneManaged falls back to %(id)s and the reconcile pass repairs the
@@ -589,7 +607,7 @@ export async function importVideoAction(
// queue control starts at the CHANNEL's queue, which is therefore read as
// "no choice made"; any other queue the operator picks still wins.
const channelQueue = downloadQueueKey(channelConfig);
- const ownQueue = archiveOrg || bitchute ? platformQueueKey(urlPlatform) : null;
+ const ownQueue = archiveOrg || pacedPlatform ? platformQueueKey(urlPlatform) : null;
const override =
ownQueue && queueKey !== undefined && queueKey.trim() === channelQueue
? undefined
@@ -607,6 +625,18 @@ export async function importVideoAction(
kind: "download",
});
try {
+ if (pacedPlatform) {
+ await waitForPlatformGap({
+ label: pacedLabel,
+ remainingMs: async () =>
+ Math.max(
+ platformGapRemainingMs(pacedPlatform),
+ await platformCooldownRemainingMs(pacedPlatform, paths).catch(() => 0),
+ ),
+ signal,
+ onLog: task.onLog,
+ });
+ }
const outcome = await downloadOneManaged({
channelSlug: slug,
channelConfig: downloadConfig,
@@ -619,16 +649,23 @@ export async function importVideoAction(
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
});
- if (bitchute) {
- // BitChute's shared pacing state learns what it said: a 429 backs
- // every BitChute path off (the lane, a sync, the next import).
- // Best-effort, as the batch downloads' bookkeeping is.
+ if (pacedPlatform) {
+ // The platform's shared pacing state learns what it said: a 429
+ // backs every path on it off (the lane, a sync, the next import).
+ // Best-effort, as the batch downloads' bookkeeping is. The next
+ // import there waits the floor from now, whatever the answer.
+ notePlatformGap(
+ pacedPlatform,
+ downloadGapMs(settings.sleepBetweenDownloadsSeconds, 0, 0, {
+ minSeconds: platformImportMinGapSeconds(pacedPlatform),
+ }),
+ );
const answer = importPlatformSignal(outcome);
try {
if (answer === "rate_limit" || answer === "network") {
- await recordDownloadBackoff("bitchute", paths, answer);
+ await recordDownloadBackoff(pacedPlatform, paths, answer);
} else if (answer === "clean") {
- const line = await recordPlatformClean("bitchute", paths);
+ const line = await recordPlatformClean(pacedPlatform, paths);
if (line) task.onLog(line);
}
} catch {
@@ -660,6 +697,9 @@ export async function importVideoAction(
});
}
+// How an import names a paced platform in what it says.
+const PACED_LABELS: Partial<Record<string, string>> = { bitchute: "BitChute", odysee: "Odysee" };
+
// A platform held, or in a rate-limit cooldown: a sentence, else null. An
// import asks a platform nothing while it has asked us to wait.
async function platformRefusal(
diff --git a/editor/e2e/import-video.spec.ts b/editor/e2e/import-video.spec.ts
@@ -113,11 +113,24 @@ test("a BitChute import runs on BitChute's queue and pace, and is never fetched
expect(queues).toEqual(["platform:bitchute"]);
}).toPass({ timeout: 15_000 });
- // Imported again: refused as already downloaded, nothing fetched.
+ // Imported again: refused as already downloaded, nothing fetched. The first
+ // run's end refreshes the page, and a click that lands during that refresh
+ // never reaches the button's handler — so click until the notice shows. An
+ // import that was NOT refused never shows it, and still fails here.
await expect(importBtn).toBeEnabled({ timeout: 30_000 });
- await importBtn.click();
- await expect(page.getByLabel("Import video notice")).toContainText(
- "Already downloaded",
- { timeout: 10_000 },
- );
+ await expect(async () => {
+ await importBtn.click();
+ await expect(page.getByLabel("Import video notice")).toContainText(
+ "Already downloaded",
+ { timeout: 3_000 },
+ );
+ }).toPass({ timeout: 30_000 });
+ // However many clicks that took, BitChute was asked once.
+ const imports: string[] = [];
+ for (const name of await readdir(resolvePath("test-transcripts/.jobs"))) {
+ if (!name.endsWith(".meta.json")) continue;
+ const m = JSON.parse(await readFile(resolvePath(`test-transcripts/.jobs/${name}`), "utf8"));
+ if (m.kind === "import-one" && m.videoId === id) imports.push(m.id);
+ }
+ expect(imports).toHaveLength(1);
});