commit b686a43e0336096a119003f5698448376e0968ca
parent 2ddc49784e4a347db52aa13e64ee733836498787
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 3 Oct 2026 15:48:03 -0400
common: X reads wait a random 4–10 s per request; older-posts windows a random 45–120 s apart
gallery-dl's default request gap for X is 0. Every read now passes
sleep-request=4.0-10.0 (gallery-dl's uniform range) and ratelimit=wait;
the search walk's fixed 15 s between windows becomes a fresh random draw.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 60 insertions(+), 6 deletions(-)
diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts
@@ -5,7 +5,14 @@
import { test } from "node:test";
import assert from "node:assert/strict";
-import { buildGalleryDlArgs, galleryDlCookieChoice } from "./xGalleryDlFetcher";
+import {
+ buildGalleryDlArgs,
+ galleryDlCookieChoice,
+ OLDER_WINDOW_PAUSE_MAX_MS,
+ OLDER_WINDOW_PAUSE_MIN_MS,
+ olderWindowPauseMs,
+ X_SLEEP_REQUEST,
+} from "./xGalleryDlFetcher";
const ACCOUNT = "https://x.com/someaccount";
const JAR = "/corpus/.x-session/cookies.txt";
@@ -280,6 +287,8 @@ test("search argv: the timeline's flags and login, latest-first max_id paging, n
"extractor.twitter.replies=true",
"extractor.twitter.search-results=latest",
"extractor.twitter.search-pagination=max_id",
+ "extractor.twitter.sleep-request=4.0-10.0",
+ "extractor.twitter.ratelimit=wait",
]) {
assert.ok(argv.includes(opt), opt);
}
@@ -307,3 +316,23 @@ test("stream: tracks the oldest post id read (any post, archived or not) and the
assert.equal(s.takePending().length, 3);
assert.equal(s.pendingCount(), 0);
});
+
+test("every X read paces itself: a random 4–10 s before each request, rate limits waited out", () => {
+ for (const argv of [
+ buildGalleryDlArgs({ accountUrl: "https://x.com/example_user" }),
+ buildGalleryDlSearchArgs({ handle: "example_user", position: { since: "2020-12-11", until: "2021-03-11" } }),
+ ]) {
+ assert.ok(argv.includes(`extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`));
+ assert.ok(argv.includes("extractor.twitter.ratelimit=wait"));
+ }
+ const [lo, hi] = X_SLEEP_REQUEST.split("-").map(Number);
+ assert.ok(lo >= 4 && hi > lo, "a range, not a fixed gap");
+});
+
+test("the pause between search windows is a fresh random draw in 45–120 s", () => {
+ assert.equal(olderWindowPauseMs(() => 0), OLDER_WINDOW_PAUSE_MIN_MS);
+ assert.equal(olderWindowPauseMs(() => 1), OLDER_WINDOW_PAUSE_MAX_MS);
+ const mid = olderWindowPauseMs(() => 0.5);
+ assert.ok(mid > OLDER_WINDOW_PAUSE_MIN_MS && mid < OLDER_WINDOW_PAUSE_MAX_MS);
+ assert.ok(OLDER_WINDOW_PAUSE_MIN_MS >= 45_000);
+});
diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts
@@ -115,6 +115,9 @@ export function buildGalleryDlArgs(
return args;
}
+// The random gap gallery-dl waits before each X request, in seconds ("min-max").
+export const X_SLEEP_REQUEST = "4.0-10.0";
+
// The flags every gallery-dl read shares: metadata only, streamed, text tweets
// with their retweets and replies, the login, and the --range cap.
function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] {
@@ -145,6 +148,17 @@ function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] {
"extractor.twitter.videos=false",
"-o",
"extractor.twitter.cards=false",
+ // PACING. gallery-dl's default gap between X API requests is 0: it pages as
+ // fast as X answers and only slows down when X's rate-limit headers say to.
+ // An account-history walk is hundreds of requests, so every read waits a
+ // RANDOM 4–10 s before each one (gallery-dl's own "a-b" range, a uniform
+ // draw per request): no fixed rhythm, and well under the rate a person
+ // scrolling would make. Running into the limit anyway is waited out, never
+ // pushed through ("wait" is gallery-dl's default; stated so it stays).
+ "-o",
+ `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`,
+ "-o",
+ "extractor.twitter.ratelimit=wait",
];
// Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no
// account at all (verified: 500 tweets over ~7.5 months for a public
@@ -255,8 +269,16 @@ function olderId(a: string | undefined, b: string): string {
// Pause between two windows' searches. Each window is its own gallery-dl run
// (a user lookup, then the search pages), so a stretch of empty windows would
-// otherwise be a burst of requests a second apart.
-export const OLDER_WINDOW_PAUSE_MS = 15_000;
+// otherwise be a burst of requests a second apart. RANDOM, 45–120 s, drawn per
+// window, for the same reason the per-request gap is: no fixed rhythm.
+export const OLDER_WINDOW_PAUSE_MIN_MS = 45_000;
+export const OLDER_WINDOW_PAUSE_MAX_MS = 120_000;
+export function olderWindowPauseMs(rand: () => number = Math.random): number {
+ return Math.round(
+ OLDER_WINDOW_PAUSE_MIN_MS +
+ rand() * (OLDER_WINDOW_PAUSE_MAX_MS - OLDER_WINDOW_PAUSE_MIN_MS),
+ );
+}
function pause(ms: number, signal: AbortSignal): Promise<void> {
if (ms <= 0 || signal.aborted) return Promise.resolve();
@@ -638,7 +660,9 @@ export async function fetchOlderViaSearch(
}
const deadline = Date.now() + BACKFILL_BUDGET_MS;
- const pauseMs = input.windowPauseMs ?? OLDER_WINDOW_PAUSE_MS;
+ // A fixed pause when the caller names one (tests pass 0); otherwise a fresh
+ // random draw before every window.
+ const pauseFor = () => input.windowPauseMs ?? olderWindowPauseMs();
let position: OlderBackfillPosition = { ...input.position };
let accountCreatedAt = input.accountCreatedAt;
// Posts not handed to a checkpoint, returned for the final write.
@@ -818,7 +842,7 @@ export async function fetchOlderViaSearch(
onLog?.("Cancelled; the next run resumes where this one left off.");
return stopped(position);
}
- await pause(pauseMs, signal);
+ await pause(pauseFor(), signal);
if (signal.aborted) {
onLog?.("Cancelled; the next run resumes where this one left off.");
return stopped(position);
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **X posts are fetched more slowly, with random gaps.** Every read of X now waits a random 4 to 10 seconds before each request to X, where it used to page as fast as X answered, and always waits out a rate limit rather than pushing through. When fetching older posts, the pause between one three-month window and the next is a random 45 to 120 seconds instead of a fixed 15. A deep walk of an account's history takes longer; a routine fetch of new posts takes a few seconds more.
- **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched.
- **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history.
- **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete.
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -8546,7 +8546,7 @@ phase deletes from the destination.
- **The older-posts backfill** (`fetchPosts({ older })` → `SocialFetcher.fetchOlder`; for X
`fetchOlderViaSearch` in `xGalleryDlFetcher.ts`, window arithmetic in `social/olderBackfill.ts`).
Windows of 3 months in UTC days, `from:<h> since:<d> until:<d> include:nativeretweets`, the first
- ending the day after the oldest archived post; one gallery-dl run per window, 15 s apart, one
+ ending the day after the oldest archived post; one gallery-dl run per window, a random 45–120 s apart (every X request itself waits a random 4–10 s: `sleep-request=4.0-10.0`, `ratelimit=wait`), one
3-hour budget across them. Its resume point is its own: the oldest post id read in the window,
sent back as `max_id:<id>` (inclusive; gallery-dl's rewrite replaces it). State is
`posts-state.json` `older` (`OlderBackfillState`), apart from the timeline `cursor`; a normal fetch