Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit b686a43e0336096a119003f5698448376e0968ca
parent 2ddc49784e4a347db52aa13e64ee733836498787
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  3 Oct 2026 15:48:03 -0400

common: X reads wait a random 4–10 s per request; older-posts windows a random 45–120 s apart

gallery-dl's default request gap for X is 0. Every read now passes
sleep-request=4.0-10.0 (gallery-dl's uniform range) and ratelimit=wait;
the search walk's fixed 15 s between windows becomes a fresh random draw.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Mcommon/social/xGalleryDlFetcher.test.ts | 31++++++++++++++++++++++++++++++-
Mcommon/social/xGalleryDlFetcher.ts | 32++++++++++++++++++++++++++++----
Meditor/CHANGELOG.md | 1+
Mplans/FACTS.md | 2+-
4 files changed, 60 insertions(+), 6 deletions(-)

diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts @@ -5,7 +5,14 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { buildGalleryDlArgs, galleryDlCookieChoice } from "./xGalleryDlFetcher"; +import { + buildGalleryDlArgs, + galleryDlCookieChoice, + OLDER_WINDOW_PAUSE_MAX_MS, + OLDER_WINDOW_PAUSE_MIN_MS, + olderWindowPauseMs, + X_SLEEP_REQUEST, +} from "./xGalleryDlFetcher"; const ACCOUNT = "https://x.com/someaccount"; const JAR = "/corpus/.x-session/cookies.txt"; @@ -280,6 +287,8 @@ test("search argv: the timeline's flags and login, latest-first max_id paging, n "extractor.twitter.replies=true", "extractor.twitter.search-results=latest", "extractor.twitter.search-pagination=max_id", + "extractor.twitter.sleep-request=4.0-10.0", + "extractor.twitter.ratelimit=wait", ]) { assert.ok(argv.includes(opt), opt); } @@ -307,3 +316,23 @@ test("stream: tracks the oldest post id read (any post, archived or not) and the assert.equal(s.takePending().length, 3); assert.equal(s.pendingCount(), 0); }); + +test("every X read paces itself: a random 4–10 s before each request, rate limits waited out", () => { + for (const argv of [ + buildGalleryDlArgs({ accountUrl: "https://x.com/example_user" }), + buildGalleryDlSearchArgs({ handle: "example_user", position: { since: "2020-12-11", until: "2021-03-11" } }), + ]) { + assert.ok(argv.includes(`extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`)); + assert.ok(argv.includes("extractor.twitter.ratelimit=wait")); + } + const [lo, hi] = X_SLEEP_REQUEST.split("-").map(Number); + assert.ok(lo >= 4 && hi > lo, "a range, not a fixed gap"); +}); + +test("the pause between search windows is a fresh random draw in 45–120 s", () => { + assert.equal(olderWindowPauseMs(() => 0), OLDER_WINDOW_PAUSE_MIN_MS); + assert.equal(olderWindowPauseMs(() => 1), OLDER_WINDOW_PAUSE_MAX_MS); + const mid = olderWindowPauseMs(() => 0.5); + assert.ok(mid > OLDER_WINDOW_PAUSE_MIN_MS && mid < OLDER_WINDOW_PAUSE_MAX_MS); + assert.ok(OLDER_WINDOW_PAUSE_MIN_MS >= 45_000); +}); diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -115,6 +115,9 @@ export function buildGalleryDlArgs( return args; } +// The random gap gallery-dl waits before each X request, in seconds ("min-max"). +export const X_SLEEP_REQUEST = "4.0-10.0"; + // The flags every gallery-dl read shares: metadata only, streamed, text tweets // with their retweets and replies, the login, and the --range cap. function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { @@ -145,6 +148,17 @@ function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { "extractor.twitter.videos=false", "-o", "extractor.twitter.cards=false", + // PACING. gallery-dl's default gap between X API requests is 0: it pages as + // fast as X answers and only slows down when X's rate-limit headers say to. + // An account-history walk is hundreds of requests, so every read waits a + // RANDOM 4–10 s before each one (gallery-dl's own "a-b" range, a uniform + // draw per request): no fixed rhythm, and well under the rate a person + // scrolling would make. Running into the limit anyway is waited out, never + // pushed through ("wait" is gallery-dl's default; stated so it stays). + "-o", + `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`, + "-o", + "extractor.twitter.ratelimit=wait", ]; // Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no // account at all (verified: 500 tweets over ~7.5 months for a public @@ -255,8 +269,16 @@ function olderId(a: string | undefined, b: string): string { // Pause between two windows' searches. Each window is its own gallery-dl run // (a user lookup, then the search pages), so a stretch of empty windows would -// otherwise be a burst of requests a second apart. -export const OLDER_WINDOW_PAUSE_MS = 15_000; +// otherwise be a burst of requests a second apart. RANDOM, 45–120 s, drawn per +// window, for the same reason the per-request gap is: no fixed rhythm. +export const OLDER_WINDOW_PAUSE_MIN_MS = 45_000; +export const OLDER_WINDOW_PAUSE_MAX_MS = 120_000; +export function olderWindowPauseMs(rand: () => number = Math.random): number { + return Math.round( + OLDER_WINDOW_PAUSE_MIN_MS + + rand() * (OLDER_WINDOW_PAUSE_MAX_MS - OLDER_WINDOW_PAUSE_MIN_MS), + ); +} function pause(ms: number, signal: AbortSignal): Promise<void> { if (ms <= 0 || signal.aborted) return Promise.resolve(); @@ -638,7 +660,9 @@ export async function fetchOlderViaSearch( } const deadline = Date.now() + BACKFILL_BUDGET_MS; - const pauseMs = input.windowPauseMs ?? OLDER_WINDOW_PAUSE_MS; + // A fixed pause when the caller names one (tests pass 0); otherwise a fresh + // random draw before every window. + const pauseFor = () => input.windowPauseMs ?? olderWindowPauseMs(); let position: OlderBackfillPosition = { ...input.position }; let accountCreatedAt = input.accountCreatedAt; // Posts not handed to a checkpoint, returned for the final write. @@ -818,7 +842,7 @@ export async function fetchOlderViaSearch( onLog?.("Cancelled; the next run resumes where this one left off."); return stopped(position); } - await pause(pauseMs, signal); + await pause(pauseFor(), signal); if (signal.aborted) { onLog?.("Cancelled; the next run resumes where this one left off."); return stopped(position); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **X posts are fetched more slowly, with random gaps.** Every read of X now waits a random 4 to 10 seconds before each request to X, where it used to page as fast as X answered, and always waits out a rate limit rather than pushing through. When fetching older posts, the pause between one three-month window and the next is a random 45 to 120 seconds instead of a fixed 15. A deep walk of an account's history takes longer; a routine fetch of new posts takes a few seconds more. - **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched. - **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history. - **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete. diff --git a/plans/FACTS.md b/plans/FACTS.md @@ -8546,7 +8546,7 @@ phase deletes from the destination. - **The older-posts backfill** (`fetchPosts({ older })` → `SocialFetcher.fetchOlder`; for X `fetchOlderViaSearch` in `xGalleryDlFetcher.ts`, window arithmetic in `social/olderBackfill.ts`). Windows of 3 months in UTC days, `from:<h> since:<d> until:<d> include:nativeretweets`, the first - ending the day after the oldest archived post; one gallery-dl run per window, 15 s apart, one + ending the day after the oldest archived post; one gallery-dl run per window, a random 45–120 s apart (every X request itself waits a random 4–10 s: `sleep-request=4.0-10.0`, `ratelimit=wait`), one 3-hour budget across them. Its resume point is its own: the oldest post id read in the window, sent back as `max_id:<id>` (inclusive; gallery-dl's rewrite replaces it). State is `posts-state.json` `older` (`OlderBackfillState`), apart from the timeline `cursor`; a normal fetch