commit 2ddc49784e4a347db52aa13e64ee733836498787
parent e0bbb6a7266dd272bddd7b43e4664fa6b1926a7a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 3 Oct 2026 13:37:47 -0400
Merge x-search-backfill (X posts older than the timeline reaches: a search walk in three-month windows from the oldest archived post, resumable via max_id, its own state beside the timeline cursor; Fetch older posts button, fetch-posts ops route, --older/--floor)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
23 files changed, 1793 insertions(+), 41 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -254,6 +254,7 @@ pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"downl
pnpm ops lane --json '{"lane":"download","held":true}'
pnpm ops refresh-report --json '{"all":true}'
pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}'
+pnpm ops fetch-posts --json '{"slug":"example-x","older":true}' --wait
pnpm ops get channel the-quartering
pnpm ops list # every action name
```
diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts
@@ -252,7 +252,7 @@ export const COMMANDS: Command[] = [
script(["duplicates"], "duplicate-shorts.ts",
"[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192),
script(["posts", "fetch"], "fetch-posts.ts",
- "--slug <channel> [--full] [--limit N] fetch a social channel's posts into its posts corpus"),
+ "--slug <channel> [--full | --older [--floor YYYY-MM-DD]] [--limit N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post)"),
script(["posts", "check"], "check-post-availability.ts",
"--slug <channel> [--mode stale|unchecked|all] [--limit N] which archived posts were deleted at the source"),
script(["diarize", "backfill"], "diarize-backfill.ts",
diff --git a/common/bin/fetch-posts.ts b/common/bin/fetch-posts.ts
@@ -2,11 +2,16 @@
// Fetch social posts for one channel into its on-disk posts corpus.
//
// pnpm --filter yt-dlp-transcript-common exec tsx bin/fetch-posts.ts \
-// --slug <channel-slug> [--full] [--limit N]
+// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD]] [--limit N]
//
// --full re-walks the account's whole history instead of stopping at the
// stored watermark. New posts are still deduped against the posts-archive, so
// it repairs a gap rather than creating duplicates.
+//
+// --older walks the account's history backwards from the oldest archived post
+// (X: search windows), below what the timeline reaches; --floor is the date it
+// stops at. Its position is saved apart from the timeline's, so a later run
+// continues it and a normal fetch is unaffected.
import { getPaths } from "../lib/paths";
import { getSettings } from "../lib/settings";
@@ -16,7 +21,9 @@ import { parseFlags } from "./_parseFlags";
const flags = parseFlags(process.argv.slice(2));
const slug = flags.slug;
if (!slug) {
- console.error("Usage: fetch-posts.ts --slug <channel-slug> [--full] [--limit N]");
+ console.error(
+ "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD]] [--limit N]",
+ );
process.exit(2);
}
@@ -31,6 +38,8 @@ fetchPosts({
slug,
settings: getSettings(),
full: flags.full === "true",
+ older: flags.older === "true",
+ floor: flags.floor,
limit,
onLog: (line) => console.log(line),
})
diff --git a/common/controller/fetchOlderPosts.test.ts b/common/controller/fetchOlderPosts.test.ts
@@ -0,0 +1,398 @@
+// fetchPosts({ older: true }) over the gallery-dl X fetcher, end to end
+// against a fake binary that answers X SEARCH URLs from a fixed set of posts:
+// the walk steps back a window at a time, writes through the same shard
+// writer, records its own position apart from the timeline cursor, stops at a
+// floor / the account's creation date / a run of empty windows, and a normal
+// fetch never resumes backwards from it.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/fetchOlderPosts.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { chmod, mkdir, mkdtemp, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "fetcholder-"));
+const BIN = path.join(ROOT, "fake-gallery-dl.mjs");
+const ARGS_LOG = path.join(ROOT, "argv.jsonl");
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+process.env.GALLERY_DL_BIN = BIN;
+process.env.FAKE_ARGS_LOG = ARGS_LOG;
+
+// The account's posts, by date. An id is X's: the creation time in its high
+// bits, so ids order as dates do.
+const POST_DATES = [
+ "2021-03-10 12:00:00", // the oldest post already archived
+ "2021-02-01 09:00:00",
+ "2020-12-15 18:30:00",
+ "2020-06-01 08:00:00",
+ "2019-11-20 22:15:00",
+];
+
+// The fake: a timeline URL reads nothing (the timeline is exhausted); a search
+// URL answers its query's from:/since:/until:/max_id: from POST_DATES, newest
+// first, honouring --range. FAKE_ACCOUNT_DATE is the account's creation date
+// carried on each record's `user`. FAKE_MODE=stall prints the window's first
+// post, then sleeps on a "rate limit" until killed.
+await writeFile(
+ BIN,
+ `#!/usr/bin/env node
+import { appendFileSync } from "node:fs";
+const args = process.argv.slice(2);
+appendFileSync(process.env.FAKE_ARGS_LOG, JSON.stringify(args) + "\\n");
+const url = new URL(args[args.length - 1]);
+if (!url.pathname.startsWith("/search")) process.exit(0);
+const q = url.searchParams.get("q");
+const term = (k) => (q.split(" ").find((t) => t.startsWith(k + ":")) || "").slice(k.length + 1) || undefined;
+const from = term("from"), since = term("since"), until = term("until"), maxId = term("max_id");
+const range = args[args.indexOf("--range") + 1];
+const cap = args.includes("--range") ? Number(range.split("-")[1]) : Infinity;
+const dates = ${JSON.stringify(POST_DATES)};
+const idOf = (d) => (BigInt(Date.parse(d.replace(" ", "T") + "Z") - 1288834974657) << 22n);
+const hits = dates
+ .filter((d) => d.slice(0, 10) >= since && d.slice(0, 10) < until)
+ .filter((d) => !maxId || idOf(d) <= BigInt(maxId))
+ .sort().reverse();
+const user = { name: from, nick: "Example" };
+if (process.env.FAKE_ACCOUNT_DATE) user.date = process.env.FAKE_ACCOUNT_DATE;
+const line = (d) => '[2, {"tweet_id": ' + idOf(d) + ', "date": "' + d + '", "content": "post at ' + d +
+ '", "author": ' + JSON.stringify(user) + ', "user": ' + JSON.stringify(user) + '}]\\n';
+if (process.env.FAKE_MODE === "stall" && hits.length) {
+ process.stdout.write(line(hits[0]));
+ process.stderr.write("[twitter][info] Waiting for 14 minutes until 15:11:22 (rate limit)\\n");
+ setTimeout(() => process.exit(0), 60_000);
+} else {
+ hits.slice(0, cap).forEach((d) => process.stdout.write(line(d)));
+ process.exit(0);
+}
+`,
+);
+await chmod(BIN, 0o755);
+
+const HANDLE = "example_user";
+const channelsDir = path.join(ROOT, "transcripts", "channels");
+
+async function makeChannel(slug: string): Promise<string> {
+ const channelRoot = path.join(channelsDir, slug);
+ await mkdir(channelRoot, { recursive: true });
+ await writeFile(
+ path.join(channelRoot, "config.json"),
+ JSON.stringify({
+ handling: "transcribe",
+ sourceKind: "social",
+ postFetcher: "x-gallery-dl",
+ socialHandle: HANDLE,
+ platform: "twitter",
+ name: "Example (X)",
+ url: `https://x.com/${HANDLE}`,
+ }),
+ );
+ return channelRoot;
+}
+
+const { fetchPosts, FULL_AND_OLDER_REFUSAL, olderPostsProblem } = await import("./fetchPosts");
+const {
+ readPostFetchState,
+ readSeenPostIds,
+ writePostFetchState,
+ writePosts,
+ readAllPosts,
+} = await import("../lib/posts-server");
+const { normalizeXTweet } = await import("../social/xNormalize");
+const { getPaths } = await import("../lib/paths");
+
+const LOGIN = { cookiesFromBrowser: "firefox" };
+
+function idOf(d: string): string {
+ return String(
+ BigInt(Date.parse(d.replace(" ", "T") + "Z") - 1288834974657) << 22n,
+ );
+}
+
+// The archive as a normal fetch would have left it: the oldest post, plus a
+// timeline resume point.
+async function seed(channelRoot: string, slug: string): Promise<void> {
+ const post = normalizeXTweet(
+ {
+ tweet_id: idOf(POST_DATES[0]),
+ date: POST_DATES[0],
+ content: "archived",
+ author: { name: HANDLE },
+ user: { name: HANDLE },
+ },
+ slug,
+ );
+ assert.ok(post);
+ await writePosts(channelRoot, [post]);
+ await writePostFetchState(channelRoot, {
+ lastFetchedAt: "2026-01-01T00:00:00.000Z",
+ cursor: "1/TIMELINE-RESUME",
+ });
+}
+
+async function argvLines(): Promise<string[][]> {
+ try {
+ return (await readFile(ARGS_LOG, "utf8"))
+ .trim()
+ .split("\n")
+ .filter(Boolean)
+ .map((l) => JSON.parse(l) as string[]);
+ } catch {
+ return [];
+ }
+}
+
+const queryOf = (argv: string[]) =>
+ new URL(argv[argv.length - 1]).searchParams.get("q") ?? "";
+
+test("an older fetch walks back window by window and stops after four empty windows", async () => {
+ const slug = "older-walk";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ const before = (await argvLines()).length;
+ const log: string[] = [];
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ olderWindowPauseMs: 0,
+ onLog: (l) => log.push(l),
+ });
+ assert.equal(result.ok, true, log.join("\n"));
+ assert.equal(result.complete, true);
+ assert.equal(result.written, 4);
+ assert.equal((await readSeenPostIds(channelRoot)).size, 5);
+
+ const runs = (await argvLines()).slice(before);
+ const queries = runs.map(queryOf);
+ // Windows tile backwards from the day after the oldest archived post.
+ assert.deepEqual(queries.slice(0, 3), [
+ `from:${HANDLE} since:2020-12-11 until:2021-03-11 include:nativeretweets`,
+ `from:${HANDLE} since:2020-09-11 until:2020-12-11 include:nativeretweets`,
+ `from:${HANDLE} since:2020-06-11 until:2020-09-11 include:nativeretweets`,
+ ]);
+ // Ten windows: three with posts, the run of four empty ones at the end, and
+ // the empty ones between that a later post reset.
+ assert.equal(runs.length, 10, queries.join("\n"));
+ assert.match(queries[9], /since:2018-09-11 until:2018-12-11/);
+
+ const state = await readPostFetchState(channelRoot);
+ assert.equal(state?.older?.complete, true);
+ assert.match(state?.older?.completeReason ?? "", /4 windows of 3 months in a row held no posts/);
+ // The timeline's own state is exactly as the last normal fetch left it.
+ assert.equal(state?.cursor, "1/TIMELINE-RESUME");
+ assert.equal(state?.lastFetchedAt, "2026-01-01T00:00:00.000Z");
+ // Not a sync: the channel's sync stamp is not touched.
+ const config = JSON.parse(await readFile(path.join(channelRoot, "config.json"), "utf8"));
+ assert.equal(config.lastSyncedAt, undefined);
+
+ // Re-running says so, and searches nothing.
+ const again: string[] = [];
+ const rerun = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ olderWindowPauseMs: 0,
+ onLog: (l) => again.push(l),
+ });
+ assert.equal(rerun.ok, true);
+ assert.equal(rerun.written, 0);
+ assert.equal((await argvLines()).length, before + 10);
+ assert.ok(again.some((l) => /already complete/.test(l)), again.join("\n"));
+});
+
+test("a normal fetch ignores the older walk's state, resumes the TIMELINE, and keeps the older state", async () => {
+ const slug = "older-walk";
+ const channelRoot = path.join(channelsDir, slug);
+ const olderBefore = (await readPostFetchState(channelRoot))?.older;
+ assert.ok(olderBefore);
+ const result = await fetchPosts({ paths: getPaths(), slug, settings: LOGIN });
+ assert.equal(result.ok, true);
+ const argv = (await argvLines()).at(-1)!;
+ assert.equal(argv[argv.length - 1], `https://x.com/${HANDLE}/timeline`);
+ assert.ok(argv.includes("extractor.twitter.cursor=1/TIMELINE-RESUME"));
+ assert.equal(argv.some((a) => a.includes("/search")), false);
+ const state = await readPostFetchState(channelRoot);
+ assert.deepEqual(state?.older, olderBefore);
+});
+
+test("a normal fetch with a half-walked older state still reads new posts from the top", async () => {
+ const slug = "older-half";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ await writePostFetchState(channelRoot, {
+ older: { since: "2019-01-01", until: "2019-04-01", maxId: "123", emptyWindows: 1 },
+ });
+ await fetchPosts({ paths: getPaths(), slug, settings: LOGIN });
+ const argv = (await argvLines()).at(-1)!;
+ assert.equal(argv[argv.length - 1], `https://x.com/${HANDLE}/timeline`);
+ assert.equal(argv.some((a) => a.startsWith("extractor.twitter.cursor=")), false);
+ assert.deepEqual((await readPostFetchState(channelRoot))?.older, {
+ since: "2019-01-01",
+ until: "2019-04-01",
+ maxId: "123",
+ emptyWindows: 1,
+ });
+});
+
+test("a floor date ends the walk at that date, and posts below it are not fetched", async () => {
+ const slug = "older-floor";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ floor: "2020-07-01",
+ olderWindowPauseMs: 0,
+ });
+ assert.equal(result.ok, true);
+ assert.equal(result.complete, true);
+ assert.equal(result.written, 2);
+ const queries = (await argvLines()).map(queryOf);
+ assert.match(queries.at(-1)!, /since:2020-07-01 until:2020-09-11/);
+ const state = await readPostFetchState(channelRoot);
+ assert.equal(state?.older?.floor, "2020-07-01");
+ assert.match(state?.older?.completeReason ?? "", /reached 2020-07-01/);
+ const dates = (await readAllPosts(channelRoot)).map((p) => p.createdAt.slice(0, 10));
+ assert.equal(dates.includes("2020-06-01"), false);
+});
+
+test("the account's creation date, once a post carries it, is the floor", async () => {
+ const slug = "older-created";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ process.env.FAKE_ACCOUNT_DATE = "2019-10-05 10:00:00";
+ try {
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ olderWindowPauseMs: 0,
+ });
+ assert.equal(result.complete, true);
+ assert.equal(result.written, 4);
+ } finally {
+ delete process.env.FAKE_ACCOUNT_DATE;
+ }
+ const state = await readPostFetchState(channelRoot);
+ assert.equal(state?.older?.accountCreatedAt, "2019-10-05T10:00:00.000Z");
+ assert.match(state?.older?.completeReason ?? "", /reached 2019-10-05/);
+ // It stopped at the window holding the creation date, not four windows on.
+ assert.match((await argvLines()).map(queryOf).at(-1)!, /since:2019-10-05 until:2019-12-11/);
+});
+
+test("a cancelled window keeps what it read and resumes INSIDE the window, below the oldest post read", async () => {
+ const slug = "older-cancel";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ const controller = new AbortController();
+ process.env.FAKE_MODE = "stall";
+ try {
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ olderWindowPauseMs: 0,
+ signal: controller.signal,
+ onLog: (l) => {
+ if (/rate limit/.test(l)) controller.abort();
+ },
+ });
+ assert.equal(result.ok, true);
+ assert.equal(result.complete, false);
+ } finally {
+ delete process.env.FAKE_MODE;
+ }
+ // The stall printed the window's newest post: the archived one.
+ const state = await readPostFetchState(channelRoot);
+ assert.deepEqual(
+ { since: state?.older?.since, until: state?.older?.until, maxId: state?.older?.maxId },
+ { since: "2020-12-11", until: "2021-03-11", maxId: idOf(POST_DATES[0]) },
+ );
+ assert.equal(state?.cursor, "1/TIMELINE-RESUME");
+
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ olderWindowPauseMs: 0,
+ });
+ assert.equal(result.complete, true);
+ assert.equal(result.written, 4);
+ const resumed = (await argvLines()).map(queryOf).find((q) => q.includes("max_id:"));
+ assert.equal(
+ resumed,
+ `from:${HANDLE} since:2020-12-11 until:2021-03-11 include:nativeretweets max_id:${idOf(POST_DATES[0])}`,
+ );
+});
+
+test("a limit stops the walk part-way with a resume point", async () => {
+ const slug = "older-limit";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ const result = await fetchPosts({
+ paths: getPaths(),
+ slug,
+ settings: LOGIN,
+ older: true,
+ limit: 2,
+ olderWindowPauseMs: 0,
+ });
+ assert.equal(result.complete, false);
+ assert.equal(result.written, 1);
+ const argv = (await argvLines()).at(-1)!;
+ assert.equal(argv[argv.indexOf("--range") + 1], "1-2");
+ const older = (await readPostFetchState(channelRoot))?.older;
+ assert.equal(older?.maxId, idOf(POST_DATES[1]));
+ assert.equal(older?.complete, undefined);
+});
+
+test("with no login an older fetch needs cookies and spawns nothing", async () => {
+ const slug = "older-guest";
+ const channelRoot = await makeChannel(slug);
+ await seed(channelRoot, slug);
+ const before = (await argvLines()).length;
+ const result = await fetchPosts({ paths: getPaths(), slug, settings: {}, older: true });
+ assert.equal(result.ok, false);
+ assert.equal(result.needsCookies, true);
+ assert.match(result.error ?? "", /X search needs a logged-in session/);
+ assert.equal((await argvLines()).length, before);
+ const state = await readPostFetchState(channelRoot);
+ assert.equal(state?.needsCookies, true);
+ assert.equal(state?.cursor, "1/TIMELINE-RESUME");
+ assert.equal(state?.older?.complete, undefined);
+});
+
+test("full and older together are refused, and so is a fetcher with no older walk", async () => {
+ const both = await fetchPosts({
+ paths: getPaths(),
+ slug: "older-walk",
+ settings: LOGIN,
+ full: true,
+ older: true,
+ });
+ assert.equal(both.ok, false);
+ assert.equal(both.error, FULL_AND_OLDER_REFUSAL);
+
+ const badFloor = await fetchPosts({
+ paths: getPaths(),
+ slug: "older-walk",
+ settings: LOGIN,
+ older: true,
+ floor: "2020-02-30",
+ });
+ assert.equal(badFloor.ok, false);
+ assert.match(badFloor.error ?? "", /not a date/);
+
+ assert.equal(olderPostsProblem({ label: "X", fetchOlder: async () => ({ posts: [], complete: true, position: { since: "", until: "", emptyWindows: 0 } }) }), null);
+ assert.equal(olderPostsProblem({ label: "Bluesky" }), "Bluesky cannot fetch older posts.");
+ assert.equal(olderPostsProblem(undefined), "This channel has no post fetcher.");
+});
diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts
@@ -21,16 +21,25 @@ import {
} from "../lib/cookiePolicy";
import {
latestPostCreatedAt,
+ oldestPostCreatedAt,
readPostFetchState,
readSeenPostIds,
writePostFetchState,
writePosts,
+ type OlderBackfillPosition,
+ type OlderBackfillState,
type PostFetchState,
} from "../lib/posts-server";
import {
handleFromAccountUrl,
resolveSocialFetcher,
+ type SocialFetcher,
} from "../social/fetchers";
+import {
+ firstOlderWindow,
+ isUtcDay,
+ olderFloorDay,
+} from "../social/olderBackfill";
// Registering the built-in fetchers is a side effect of importing them. Keep
// this list here (rather than inside fetchers.ts) so the registry module stays
// free of imports from the heavier fetchers.
@@ -42,6 +51,7 @@ import "../social/xGalleryDlFetcher";
// only ever used when a channel opts into it via postFetcher: "x-playwright".
import "../social/xPlaywrightFetcher";
import "../social/xNitterFetcher";
+import type { XCookieSource } from "../social/xCookieSource";
import {
resolveXCookieSourceFor,
type XLoginSettings,
@@ -57,11 +67,37 @@ export type FetchPostsOptions = {
// posts are still deduped against the archive, so this is a safe repair
// operation rather than a duplicate-maker.
full?: boolean;
+ // Walk the account's history BACKWARDS from the oldest archived post, with
+ // the fetcher's `fetchOlder` (X: search windows), instead of fetching new
+ // posts. Its position is kept in posts-state.json's `older`, apart from the
+ // timeline cursor. Not with `full`.
+ older?: boolean;
+ // The older walk's floor date (YYYY-MM-DD): it stops there. Remembered for
+ // later runs of the walk. Default: none (it stops after a run of empty
+ // windows, or at the account's creation date).
+ floor?: string;
+ // The older walk's pause between two windows. Default: the fetcher's own.
+ olderWindowPauseMs?: number;
limit?: number;
onLog?: (line: string) => void;
signal?: AbortSignal;
};
+// The refusal for both walks at once, shared with the server action so the
+// button, the ops route and this controller say the same thing.
+export const FULL_AND_OLDER_REFUSAL =
+ "A full re-fetch and an older-posts fetch are different walks — run one at a time.";
+
+// Why this fetcher cannot fetch older posts, or null when it can.
+export function olderPostsProblem(
+ fetcher: Pick<SocialFetcher, "label" | "fetchOlder"> | undefined,
+): string | null {
+ if (!fetcher) return "This channel has no post fetcher.";
+ return typeof fetcher.fetchOlder === "function"
+ ? null
+ : `${fetcher.label} cannot fetch older posts.`;
+}
+
export type FetchPostsResult = {
ok: boolean;
written: number;
@@ -78,6 +114,19 @@ export async function fetchPosts(
const log = (line: string) => onLog?.(line);
const channelRoot = path.join(paths.channelsDir, slug);
+ if (opts.full && opts.older) {
+ return { ok: false, written: 0, skipped: 0, complete: false, error: FULL_AND_OLDER_REFUSAL };
+ }
+ if (opts.floor !== undefined && !isUtcDay(opts.floor)) {
+ return {
+ ok: false,
+ written: 0,
+ skipped: 0,
+ complete: false,
+ error: `"${opts.floor}" is not a date (YYYY-MM-DD)`,
+ };
+ }
+
const config = await readChannelConfig(paths, slug);
if (!config) {
return { ok: false, written: 0, skipped: 0, complete: false, error: `No such channel: ${slug}` };
@@ -118,6 +167,30 @@ export async function fetchPosts(
const seenIds = await readSeenPostIds(channelRoot);
const priorState = await readPostFetchState(channelRoot);
+
+ if (opts.older) {
+ const problem = olderPostsProblem(fetcher);
+ if (problem) {
+ return { ok: false, written: 0, skipped: 0, complete: false, error: problem };
+ }
+ const policy = resolveCookiePolicy(settings, config);
+ const xLogin =
+ fetcher.platform === "twitter"
+ ? await resolveXCookieSourceFor(paths, settings, policy.cookies)
+ : undefined;
+ return fetchOlderPosts({
+ opts,
+ fetcher,
+ channelRoot,
+ handle,
+ accountUrl,
+ seenIds,
+ priorState,
+ cookies: alwaysCookies(policy),
+ cookieSource: xLogin?.source,
+ browserCookies: xLogin?.browserSpec,
+ });
+ }
// A stored cursor means the previous run stopped early (hit its limit, or was
// cancelled). Resume the backfill from there rather than applying the
// watermark, which would otherwise leave that history permanently missing.
@@ -156,6 +229,9 @@ export async function fetchPosts(
const state: PostFetchState = {
lastFetchedAt: new Date().toISOString(),
};
+ // The older walk's position rides along untouched: a normal fetch neither
+ // reads it nor drops it.
+ if (priorState?.older) state.older = priorState.older;
// Progress a long run saves as it goes (a fetcher that streams calls this
// between pages): the posts so far, then the resume point they are safe
@@ -258,3 +334,169 @@ export async function fetchPosts(
error: result.error,
};
}
+
+// The older-posts walk (`FetchPostsOptions.older`). Its own position in
+// posts-state.json's `older`; every other field of the state is carried as it
+// was, so the timeline cursor and watermark are exactly what the last normal
+// fetch left. It does not stamp the channel's lastSyncedAt: it is not a sync,
+// and the scheduler's next new-posts fetch stays where it was.
+async function fetchOlderPosts(ctx: {
+ opts: FetchPostsOptions;
+ fetcher: SocialFetcher;
+ channelRoot: string;
+ handle: string;
+ accountUrl: string;
+ seenIds: ReadonlySet<string>;
+ priorState: PostFetchState | null;
+ cookies?: string;
+ cookieSource?: XCookieSource;
+ browserCookies?: string;
+}): Promise<FetchPostsResult> {
+ const { opts, fetcher, channelRoot, handle, priorState } = ctx;
+ const { slug } = opts;
+ const log = (line: string) => opts.onLog?.(line);
+ const fetchOlder = fetcher.fetchOlder!;
+ const prior = priorState?.older;
+
+ if (prior?.complete) {
+ log(
+ `Older posts for ${slug} are already complete` +
+ (prior.completedAt ? ` (since ${prior.completedAt})` : "") +
+ (prior.completeReason ? `: ${prior.completeReason}` : "") +
+ ". Nothing to walk.",
+ );
+ return { ok: true, written: 0, skipped: 0, complete: true };
+ }
+
+ const floor = opts.floor ?? prior?.floor;
+ let accountCreatedAt = prior?.accountCreatedAt;
+ const startedAt = new Date().toISOString();
+ let position: OlderBackfillPosition;
+ if (prior) {
+ position = {
+ since: prior.since,
+ until: prior.until,
+ emptyWindows: prior.emptyWindows ?? 0,
+ ...(prior.maxId ? { maxId: prior.maxId } : {}),
+ };
+ } else {
+ const oldest = await oldestPostCreatedAt(channelRoot);
+ position = firstOlderWindow({
+ oldestArchivedAt: oldest ?? undefined,
+ floor: olderFloorDay(floor, accountCreatedAt),
+ });
+ }
+
+ log(
+ `Fetching older posts for ${slug} via ${fetcher.label} (@${handle}): ` +
+ (prior ? "resuming at " : "starting at ") +
+ `${position.since} – ${position.until}` +
+ (floor ? `, down to ${floor}` : "") +
+ `; ${ctx.seenIds.size} already archived.`,
+ );
+
+ let written = 0;
+ let skipped = 0;
+ const shards = new Set<string>();
+ const olderState = (
+ at: OlderBackfillPosition,
+ extra: Partial<OlderBackfillState> = {},
+ ): OlderBackfillState => ({
+ since: at.since,
+ until: at.until,
+ emptyWindows: at.emptyWindows,
+ ...(at.maxId ? { maxId: at.maxId } : {}),
+ ...(floor ? { floor } : {}),
+ ...(accountCreatedAt ? { accountCreatedAt } : {}),
+ lastRunAt: startedAt,
+ lastWritten: written,
+ ...extra,
+ });
+ // Everything but `older` is carried from the state as it was.
+ const save = (older: OlderBackfillState, needsCookies?: boolean) =>
+ writePostFetchState(channelRoot, {
+ ...(priorState ?? {}),
+ ...(needsCookies ? { needsCookies: true } : {}),
+ older,
+ });
+
+ const onCheckpoint = async (cp: {
+ posts: Post[];
+ position: OlderBackfillPosition;
+ accountCreatedAt?: string;
+ }) => {
+ const w = await writePosts(channelRoot, cp.posts);
+ written += w.written;
+ skipped += w.skipped;
+ for (const shard of w.shards) shards.add(shard);
+ accountCreatedAt ??= cp.accountCreatedAt;
+ await save(olderState(cp.position));
+ if (w.written > 0) {
+ log(`Saved ${w.written} older post(s) (${written} so far this run); resume point recorded.`);
+ }
+ };
+
+ const controller = new AbortController();
+ let result;
+ try {
+ result = await fetchOlder({
+ accountUrl: ctx.accountUrl,
+ handle,
+ channelSlug: slug,
+ seenIds: ctx.seenIds,
+ cookies: ctx.cookies,
+ cookieSource: ctx.cookieSource,
+ browserCookies: ctx.browserCookies,
+ limit: opts.limit,
+ signal: opts.signal ?? controller.signal,
+ onLog: log,
+ position,
+ floor,
+ accountCreatedAt,
+ windowPauseMs: opts.olderWindowPauseMs,
+ onCheckpoint,
+ });
+ } catch (err) {
+ const message = (err as Error).message;
+ log(`[error] ${message}`);
+ // Whatever was checkpointed is on disk with its position; keep that one.
+ const now = (await readPostFetchState(channelRoot))?.older;
+ await save({ ...(now ?? olderState(position)), lastError: message });
+ return { ok: false, written, skipped, complete: false, error: message };
+ }
+
+ accountCreatedAt ??= result.accountCreatedAt;
+ const final = await writePosts(channelRoot, result.posts);
+ written += final.written;
+ skipped += final.skipped;
+ for (const shard of final.shards) shards.add(shard);
+ log(
+ `Wrote ${written} older post(s), skipped ${skipped} already archived` +
+ (shards.size > 0 ? ` (shards: ${[...shards].join(", ")})` : "") +
+ (result.complete ? " — older posts complete." : " — more history remains."),
+ );
+ if (result.error) log(`[error] ${result.error}`);
+ await save(
+ olderState(
+ result.position,
+ result.complete
+ ? {
+ complete: true,
+ completedAt: new Date().toISOString(),
+ ...(result.completeReason ? { completeReason: result.completeReason } : {}),
+ }
+ : result.error
+ ? { lastError: result.error }
+ : {},
+ ),
+ result.needsCookies,
+ );
+ return {
+ ok: !result.error,
+ written,
+ skipped,
+ complete: result.complete,
+ needsCookies: result.needsCookies,
+ error: result.error,
+ };
+}
diff --git a/common/lib/posts-server.ts b/common/lib/posts-server.ts
@@ -211,6 +211,24 @@ export async function latestPostCreatedAt(
return null;
}
+// The oldest archived post's createdAt: where the older-posts backfill starts.
+// Only the oldest non-empty shard is read (listPostShards sorts oldest first).
+export async function oldestPostCreatedAt(
+ channelRoot: string,
+): Promise<string | null> {
+ const shards = await listPostShards(channelRoot);
+ for (const shard of shards) {
+ const posts = await readPostShard(channelRoot, shard);
+ if (posts.length === 0) continue;
+ let oldest = posts[0].createdAt;
+ for (const post of posts) {
+ if (post.createdAt < oldest) oldest = post.createdAt;
+ }
+ return oldest;
+ }
+ return null;
+}
+
// ---------------------------------------------------------------------------
// Availability sidecar ("is this post still up?")
// ---------------------------------------------------------------------------
@@ -310,7 +328,42 @@ export type PostFetchState = {
// the post-corpus analogue of the video pipeline's needs_auth outcome.
needsCookies?: boolean;
lastFetchedCount?: number;
+ // The TIMELINE walk's resume point (gallery-dl's cursor). Only a normal fetch
+ // reads it; the older-posts backfill keeps its own position in `older`.
cursor?: string;
+ // The older-posts backfill (a fetcher's `fetchOlder`), which walks the
+ // account's history backwards through search windows. Kept apart from
+ // `cursor` so a normal fetch never resumes backwards, and carried through
+ // every write a normal fetch makes.
+ older?: OlderBackfillState;
+};
+
+// Where the older-posts backfill stands. The window is in UTC days, as X's
+// `since:`/`until:` search operators take them: posts created on or after
+// `since` and before `until`.
+export type OlderBackfillPosition = {
+ since: string; // YYYY-MM-DD
+ until: string; // YYYY-MM-DD
+ // Resume INSIDE the window: only posts with an id at or below this one (the
+ // oldest post the walk has read in this window).
+ maxId?: string;
+ // Windows in a row, ending with the last one walked, that held no posts.
+ emptyWindows: number;
+};
+
+export type OlderBackfillState = OlderBackfillPosition & {
+ // The date the walk stops at (YYYY-MM-DD), when one was given.
+ floor?: string;
+ // The account's creation time, once a fetched post has carried it: the walk
+ // never goes below it.
+ accountCreatedAt?: string;
+ complete?: boolean;
+ completedAt?: string;
+ // Why the walk ended, as a sentence ("reached the floor date 2020-01-01").
+ completeReason?: string;
+ lastRunAt?: string;
+ lastWritten?: number;
+ lastError?: string;
};
export const POST_FETCH_STATE_FILENAME = "posts-state.json";
diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts
@@ -14,6 +14,7 @@
// in the client bundle.
import type { Post, PostAvailability, PostPlatform } from "../lib/posts";
+import type { OlderBackfillPosition } from "../lib/posts-server";
import type { XCookieSource } from "./xCookieSource";
export type PostFetchInput = {
@@ -80,6 +81,54 @@ export type PostFetchResult = {
error?: string;
};
+// Input for the older-posts backfill (`SocialFetcher.fetchOlder`): walk the
+// account's history BACKWARDS from `position`, a window at a time, below what
+// the timeline walk can reach. Login, limit, signal and log are the same as a
+// normal fetch's; there is no watermark and no timeline cursor.
+export type OlderPostFetchInput = Pick<
+ PostFetchInput,
+ | "accountUrl"
+ | "handle"
+ | "channelSlug"
+ | "seenIds"
+ | "cookies"
+ | "cookieSource"
+ | "browserCookies"
+ | "limit"
+ | "signal"
+ | "onLog"
+> & {
+ // Where to start: the stored position of an unfinished walk, or the first
+ // window below the oldest archived post (olderBackfill.ts, firstOlderWindow).
+ position: OlderBackfillPosition;
+ // The date the walk stops at (YYYY-MM-DD), when one is set.
+ floor?: string;
+ // The account's creation time, when an earlier run learned it.
+ accountCreatedAt?: string;
+ // Pause between two windows' searches, so a run of quiet windows is not a
+ // burst of requests. Default: the fetcher's own.
+ windowPauseMs?: number;
+ // Save progress during the walk: posts not yet handed back, and the position
+ // they are safe under. Posts passed here are NOT returned again.
+ onCheckpoint?: (checkpoint: {
+ posts: Post[];
+ position: OlderBackfillPosition;
+ accountCreatedAt?: string;
+ }) => Promise<void>;
+};
+
+export type OlderPostFetchResult = {
+ posts: Post[];
+ // True when the walk has reached its end (`completeReason` says which).
+ complete: boolean;
+ completeReason?: string;
+ // Where the next run resumes, when the walk is not complete.
+ position: OlderBackfillPosition;
+ accountCreatedAt?: string;
+ needsCookies?: boolean;
+ error?: string;
+};
+
export type SocialFetcherProbe = {
ok: boolean;
// Display name for the account, used to autofill the channel form.
@@ -132,6 +181,10 @@ export type SocialFetcher = {
checkAvailability?(
input: PostAvailabilityInput,
): Promise<Map<string, PostAvailability>>;
+ // Optional: walk the account's history backwards below what `fetch` can
+ // reach (X: search windows). A fetcher without it has no such backfill, and
+ // the channel page offers no "Fetch older posts" for it.
+ fetchOlder?(input: OlderPostFetchInput): Promise<OlderPostFetchResult>;
};
// Populated by registerSocialFetcher() from each fetcher module. Indirection
diff --git a/common/social/olderBackfill.test.ts b/common/social/olderBackfill.test.ts
@@ -0,0 +1,116 @@
+// The older-posts backfill's window arithmetic (olderBackfill.ts).
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/olderBackfill.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ addUtcDays,
+ addUtcMonths,
+ describeOlderBackfill,
+ firstOlderWindow,
+ isUtcDay,
+ olderFloorDay,
+ stepOlderWindow,
+} from "./olderBackfill";
+
+test("day arithmetic is UTC and clamps a month to its last day", () => {
+ assert.equal(addUtcDays("2020-12-31", 1), "2021-01-01");
+ assert.equal(addUtcDays("2020-03-01", -1), "2020-02-29");
+ assert.equal(addUtcMonths("2024-05-31", -3), "2024-02-29");
+ assert.equal(addUtcMonths("2021-01-15", -3), "2020-10-15");
+ assert.equal(isUtcDay("2020-02-29"), true);
+ assert.equal(isUtcDay("2021-02-29"), false);
+ assert.equal(isUtcDay("2021-2-1"), false);
+});
+
+test("the first window ends the day after the oldest archived post", () => {
+ assert.deepEqual(firstOlderWindow({ oldestArchivedAt: "2021-03-10T23:59:00.000Z" }), {
+ since: "2020-12-11",
+ until: "2021-03-11",
+ emptyWindows: 0,
+ });
+ // Nothing archived: it ends tomorrow.
+ assert.equal(
+ firstOlderWindow({ now: new Date("2026-10-03T12:00:00Z") }).until,
+ "2026-10-04",
+ );
+ // A floor inside the first window clamps its start.
+ assert.equal(
+ firstOlderWindow({ oldestArchivedAt: "2021-03-10T00:00:00Z", floor: "2021-01-01" }).since,
+ "2021-01-01",
+ );
+});
+
+test("windows step back with no gap, count empty ones, and end after a run of them", () => {
+ let pos = firstOlderWindow({ oldestArchivedAt: "2021-03-10T00:00:00Z" });
+ const seen: string[] = [];
+ const held = new Set(["2020-12-11"]); // only the first window holds posts
+ for (let i = 0; i < 10; i++) {
+ seen.push(`${pos.since}..${pos.until}`);
+ const step = stepOlderWindow(pos, { hadPosts: held.has(pos.since) });
+ if (step.complete) {
+ assert.equal(step.emptyWindows, 4);
+ assert.match(step.reason, /4 windows of 3 months in a row held no posts, back to 2019-12-11/);
+ break;
+ }
+ assert.equal(step.next.until, pos.since);
+ assert.equal(step.next.maxId, undefined);
+ pos = step.next;
+ }
+ assert.deepEqual(seen, [
+ "2020-12-11..2021-03-11",
+ "2020-09-11..2020-12-11",
+ "2020-06-11..2020-09-11",
+ "2020-03-11..2020-06-11",
+ "2019-12-11..2020-03-11",
+ ]);
+});
+
+test("a window with posts resets the empty count", () => {
+ const step = stepOlderWindow(
+ { since: "2020-01-01", until: "2020-04-01", emptyWindows: 3 },
+ { hadPosts: true },
+ );
+ assert.equal(step.complete, false);
+ if (!step.complete) assert.equal(step.next.emptyWindows, 0);
+});
+
+test("the floor clamps the last window and ends the walk after it", () => {
+ const step = stepOlderWindow(
+ { since: "2020-04-01", until: "2020-07-01", emptyWindows: 0 },
+ { hadPosts: true, floor: "2020-02-15" },
+ );
+ assert.equal(step.complete, false);
+ if (step.complete) return;
+ assert.deepEqual(step.next, { since: "2020-02-15", until: "2020-04-01", emptyWindows: 0 });
+ const last = stepOlderWindow(step.next, { hadPosts: false, floor: "2020-02-15" });
+ assert.equal(last.complete, true);
+ if (last.complete) assert.match(last.reason, /reached 2020-02-15/);
+});
+
+test("the floor is the later of the given date and the account's creation day", () => {
+ assert.equal(olderFloorDay(undefined, undefined), undefined);
+ assert.equal(olderFloorDay("2015-01-01", undefined), "2015-01-01");
+ assert.equal(olderFloorDay(undefined, "2012-06-30T22:00:00.000Z"), "2012-06-30");
+ assert.equal(olderFloorDay("2015-01-01", "2012-06-30T22:00:00.000Z"), "2015-01-01");
+ assert.equal(olderFloorDay("2010-01-01", "2012-06-30T22:00:00.000Z"), "2012-06-30");
+});
+
+test("the channel page's one line", () => {
+ assert.equal(describeOlderBackfill(undefined), "not started");
+ assert.equal(
+ describeOlderBackfill({ since: "2019-01-01", until: "2019-04-01", emptyWindows: 0 }),
+ "walked back to 2019-04-01; next 2019-01-01 – 2019-04-01",
+ );
+ assert.equal(
+ describeOlderBackfill({
+ since: "2019-01-01",
+ until: "2019-04-01",
+ emptyWindows: 4,
+ complete: true,
+ completeReason: "reached 2019-01-01, the earliest date to walk to",
+ }),
+ "complete — reached 2019-01-01, the earliest date to walk to",
+ );
+});
diff --git a/common/social/olderBackfill.ts b/common/social/olderBackfill.ts
@@ -0,0 +1,152 @@
+// The older-posts backfill's window arithmetic, pure (no I/O) so every step of
+// the walk is testable without a fetcher.
+//
+// The walk goes BACKWARDS from the oldest archived post in fixed windows of
+// UTC days — the unit X's `since:`/`until:` search operators take. Each window
+// is one bounded search; the next window ends where the last one began, so the
+// windows tile the history with no gap and no overlap. The walk ends at a floor
+// date (given, or the account's creation date once a post has carried it), or
+// after a run of windows in a row that held no posts.
+
+import type {
+ OlderBackfillPosition,
+ OlderBackfillState,
+} from "../lib/posts-server";
+
+// One window's span. Three months keeps one search bounded (a run that stops
+// mid-window resumes inside it) without spending a request per quiet week.
+export const OLDER_WINDOW_MONTHS = 3;
+
+// Windows in a row with no posts before the walk calls the history done: four
+// three-month windows, a year with nothing.
+export const OLDER_MAX_EMPTY_WINDOWS = 4;
+
+const DAY_RE = /^\d{4}-\d{2}-\d{2}$/;
+
+// A real YYYY-MM-DD calendar day (2021-02-30 is not one).
+export function isUtcDay(value: string): boolean {
+ if (!DAY_RE.test(value)) return false;
+ const ms = Date.parse(`${value}T00:00:00Z`);
+ return Number.isFinite(ms) && new Date(ms).toISOString().slice(0, 10) === value;
+}
+
+// The UTC day an instant falls on.
+export function utcDayOf(instant: string | Date): string {
+ const d = typeof instant === "string" ? new Date(instant) : instant;
+ return d.toISOString().slice(0, 10);
+}
+
+export function addUtcDays(day: string, days: number): string {
+ const [y, m, d] = day.split("-").map(Number);
+ return new Date(Date.UTC(y, m - 1, d + days)).toISOString().slice(0, 10);
+}
+
+// Calendar months, clamped to the target month's last day (2024-05-31 minus
+// three months is 2024-02-29, not March 2nd).
+export function addUtcMonths(day: string, months: number): string {
+ const [y, m, d] = day.split("-").map(Number);
+ const target = new Date(Date.UTC(y, m - 1 + months, 1));
+ const lastDay = new Date(
+ Date.UTC(target.getUTCFullYear(), target.getUTCMonth() + 1, 0),
+ ).getUTCDate();
+ target.setUTCDate(Math.min(d, lastDay));
+ return target.toISOString().slice(0, 10);
+}
+
+const later = (a: string, b: string | undefined): string =>
+ b !== undefined && b > a ? b : a;
+
+// Where the walk may not go below: the given floor date, or the day the
+// account was created, whichever is later.
+export function olderFloorDay(
+ floor: string | undefined,
+ accountCreatedAt: string | undefined,
+): string | undefined {
+ const created = accountCreatedAt ? utcDayOf(accountCreatedAt) : undefined;
+ if (floor && created) return later(floor, created);
+ return floor ?? created;
+}
+
+// The first window: it ends the day AFTER the oldest archived post (`until` is
+// exclusive), so the rest of that day is read too — the posts already on disk
+// are skipped as duplicates. With nothing archived it ends tomorrow.
+export function firstOlderWindow(opts: {
+ oldestArchivedAt?: string;
+ now?: Date;
+ months?: number;
+ floor?: string;
+}): OlderBackfillPosition {
+ const months = opts.months ?? OLDER_WINDOW_MONTHS;
+ const until = addUtcDays(
+ utcDayOf(opts.oldestArchivedAt ?? opts.now ?? new Date()),
+ 1,
+ );
+ return {
+ since: later(addUtcMonths(until, -months), opts.floor),
+ until,
+ emptyWindows: 0,
+ };
+}
+
+// True when the window starts at or below the floor: it is the walk's last.
+export function isAtFloor(
+ position: OlderBackfillPosition,
+ floor: string | undefined,
+): boolean {
+ return floor !== undefined && position.since <= floor;
+}
+
+export type OlderWindowStep =
+ | { complete: true; reason: string; emptyWindows: number }
+ | { complete: false; next: OlderBackfillPosition };
+
+// What follows a window walked to its end. `hadPosts` counts any post the
+// window held, archived already or not — a window full of known posts is not
+// an empty one.
+export function stepOlderWindow(
+ position: OlderBackfillPosition,
+ opts: {
+ hadPosts: boolean;
+ floor?: string;
+ months?: number;
+ maxEmptyWindows?: number;
+ },
+): OlderWindowStep {
+ const months = opts.months ?? OLDER_WINDOW_MONTHS;
+ const maxEmpty = opts.maxEmptyWindows ?? OLDER_MAX_EMPTY_WINDOWS;
+ const emptyWindows = opts.hadPosts ? 0 : position.emptyWindows + 1;
+ if (isAtFloor(position, opts.floor)) {
+ return {
+ complete: true,
+ reason: `reached ${opts.floor}, the earliest date to walk to`,
+ emptyWindows,
+ };
+ }
+ if (emptyWindows >= maxEmpty) {
+ return {
+ complete: true,
+ reason: `${emptyWindows} windows of ${months} months in a row held no posts, back to ${position.since}`,
+ emptyWindows,
+ };
+ }
+ const until = position.since;
+ return {
+ complete: false,
+ next: {
+ since: later(addUtcMonths(until, -months), opts.floor),
+ until,
+ emptyWindows,
+ },
+ };
+}
+
+// One line for the channel page.
+export function describeOlderBackfill(
+ state: OlderBackfillState | undefined,
+): string {
+ if (!state) return "not started";
+ if (state.complete) {
+ return `complete${state.completeReason ? ` — ${state.completeReason}` : ""}`;
+ }
+ return `walked back to ${state.until}; next ${state.since} – ${state.until}`;
+}
diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts
@@ -209,3 +209,101 @@ test("stream: a gallery-dl that ignores output.jsonl (one pretty array at exit)
test("restart-from-top cursor is the timeline extractor's first state", () => {
assert.equal(RESTART_FROM_TOP_CURSOR, "1/");
});
+
+// ---------------------------------------------------------------------------
+// Search: the older-posts backfill's argv
+// ---------------------------------------------------------------------------
+
+import {
+ buildGalleryDlSearchArgs,
+ xSearchHandle,
+ xSearchQuery,
+ xSearchUrl,
+} from "./xGalleryDlFetcher";
+
+// What gallery-dl's search extractor does with the URL's q: "+" to a space,
+// THEN unquote (extractor/twitter.py, TwitterSearchExtractor.tweets).
+function galleryDlQueryOf(url: string): string {
+ const m = /\/search\/?\?(?:[^&#]+&)*q=([^&#]+)/.exec(url);
+ assert.ok(m, url);
+ return decodeURIComponent(m[1].replace(/\+/g, " "));
+}
+
+test("search query: from/since/until, reposts included, max_id only when resuming", () => {
+ const window = { since: "2020-12-11", until: "2021-03-11" };
+ assert.equal(
+ xSearchQuery("example_user", window),
+ "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets",
+ );
+ assert.equal(
+ xSearchQuery("@example_user", { ...window, maxId: "1899000000000000001" }),
+ "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets max_id:1899000000000000001",
+ );
+ assert.throws(() => xSearchQuery("example_user", { ...window, maxId: "12 OR x" }), /not a post id/);
+});
+
+test("search handle: anything search would read as more than a name is refused", () => {
+ assert.equal(xSearchHandle(" @Example_User1 "), "Example_User1");
+ for (const bad of ["example user", "example:user", 'ex"ample', "example.user", "", "a-b"]) {
+ assert.throws(() => xSearchHandle(bad), /not an X handle/, bad);
+ }
+});
+
+test("search URL: the query survives gallery-dl's decoding exactly, spaces and colons included", () => {
+ const query = xSearchQuery("example_user", {
+ since: "2020-12-11",
+ until: "2021-03-11",
+ maxId: "1899000000000000001",
+ });
+ const url = xSearchUrl(query);
+ assert.match(url, /^https:\/\/x\.com\/search\?q=from%3Aexample_user%20since%3A2020-12-11%20/);
+ assert.ok(url.endsWith("&f=live"));
+ assert.equal(url.includes(" "), false);
+ assert.equal(galleryDlQueryOf(url), query);
+ // An encoded "+" would stay a "+", not become a space.
+ assert.equal(galleryDlQueryOf(xSearchUrl("a+b c")), "a+b c");
+});
+
+test("search argv: the timeline's flags and login, latest-first max_id paging, no cursor", () => {
+ const argv = buildGalleryDlSearchArgs({
+ handle: "example_user",
+ position: { since: "2020-12-11", until: "2021-03-11" },
+ cookieFile: JAR,
+ limit: 50,
+ });
+ assert.equal(flagValue(argv, "--cookies"), JAR);
+ assert.equal(flagValue(argv, "--range"), "1-50");
+ for (const opt of [
+ "output.jsonl=true",
+ "extractor.twitter.text-tweets=true",
+ "extractor.twitter.retweets=true",
+ "extractor.twitter.replies=true",
+ "extractor.twitter.search-results=latest",
+ "extractor.twitter.search-pagination=max_id",
+ ]) {
+ assert.ok(argv.includes(opt), opt);
+ }
+ assert.equal(argv.some((a) => a.startsWith("extractor.twitter.cursor=")), false);
+ assert.equal(argv.includes("--no-download"), true);
+ assert.equal(
+ galleryDlQueryOf(argv[argv.length - 1]),
+ "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets",
+ );
+});
+
+test("stream: tracks the oldest post id read (any post, archived or not) and the account's creation time", () => {
+ const known = String(1899000000000000000n + 1n);
+ const s = newStream({ seenIds: new Set([`${known}`]), stopAtKnown: false, accountHandle: "FakeTester" });
+ s.pushStdout(jsonl(5));
+ s.pushStdout(jsonl(1)); // archived already: still the oldest read
+ s.pushStdout(jsonl(3));
+ assert.equal(s.oldestId, known);
+ assert.equal(s.accountCreatedAt, undefined);
+ s.pushStdout(
+ JSON.stringify([2, { ...tweet(2), user: { name: "faketester", date: "2009-03-10 18:00:00" } }])
+ .replace(/"tweet_id":"(\d+)"/, '"tweet_id":$1'),
+ );
+ assert.equal(s.accountCreatedAt, "2009-03-10T18:00:00.000Z");
+ assert.equal(s.takePending().length, 3);
+ assert.equal(s.pendingCount(), 0);
+});
diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts
@@ -29,11 +29,15 @@ import type { Readable } from "node:stream";
import { execa } from "execa";
import { getPaths } from "../lib/paths";
import type { Post } from "../lib/posts";
-import { normalizeXTweet, type XTweetRaw } from "./xNormalize";
+import type { OlderBackfillPosition } from "../lib/posts-server";
+import { normalizeXTweet, xCreatedAt, type XTweetRaw } from "./xNormalize";
+import { olderFloorDay, stepOlderWindow } from "./olderBackfill";
import { readXSessionStatus, xCookieFile } from "./xSessionBroker";
import type { XCookieSource } from "./xCookieSource";
import {
registerSocialFetcher,
+ type OlderPostFetchInput,
+ type OlderPostFetchResult,
type PostFetchInput,
type PostFetchResult,
type SocialFetcher,
@@ -81,22 +85,39 @@ export function timelineUrlFor(accountUrl: string): string {
}
}
-export function buildGalleryDlArgs(opts: {
- accountUrl: string;
+type GalleryDlLoginAndCap = {
cookies?: string;
// A cookies.txt exported by the Playwright session broker
// (xSessionBroker.ts). PREFERRED over `cookies`: a live browser profile keeps
// the session fresh, which is the fix for X cookies expiring in days.
cookieFile?: string;
limit?: number;
- // gallery-dl's own resume point (`extractor.twitter.cursor`), as logged by a
- // previous streamed run — see GalleryDlStream.
- cursor?: string;
- // A long fetch, read as it runs: log at debug level so gallery-dl prints its
- // "Cursor: …" line after every page. Off for the probe, whose error message
- // is the tail of stderr and must not be buried in debug noise.
- stream?: boolean;
-}): string[] {
+};
+
+export function buildGalleryDlArgs(
+ opts: GalleryDlLoginAndCap & {
+ accountUrl: string;
+ // gallery-dl's own resume point (`extractor.twitter.cursor`), as logged by
+ // a previous streamed run — see GalleryDlStream.
+ cursor?: string;
+ // A long fetch, read as it runs: log at debug level so gallery-dl prints
+ // its "Cursor: …" line after every page. Off for the probe, whose error
+ // message is the tail of stderr and must not be buried in debug noise.
+ stream?: boolean;
+ },
+): string[] {
+ const args = galleryDlCommonArgs(opts);
+ if (opts.cursor) {
+ args.push("-o", `extractor.twitter.cursor=${opts.cursor}`);
+ }
+ if (opts.stream) args.push("--verbose");
+ args.push(timelineUrlFor(opts.accountUrl));
+ return args;
+}
+
+// The flags every gallery-dl read shares: metadata only, streamed, text tweets
+// with their retweets and replies, the login, and the --range cap.
+function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] {
const args = [
// One JSON object per item on stdout — we want metadata, never files.
"--dump-json",
@@ -141,14 +162,115 @@ export function buildGalleryDlArgs(opts: {
if (opts.limit && opts.limit > 0) {
args.push("--range", `1-${Math.floor(opts.limit)}`);
}
- if (opts.cursor) {
- args.push("-o", `extractor.twitter.cursor=${opts.cursor}`);
+ return args;
+}
+
+// ---------------------------------------------------------------------------
+// Search: the older-posts backfill
+// ---------------------------------------------------------------------------
+//
+// X's timeline only pages back so far; below that, an account's posts are
+// reachable through search. gallery-dl's `twitter:search` extractor takes a
+// search URL (`https://x.com/search?q=<query>`), reads the query's `from:` user
+// for the records' `user`, and pages with `max_id:` (its default
+// `search-pagination`): after each page it rewrites the query's `max_id:` to
+// just below the oldest tweet it has read. That pagination logs no cursor, so
+// this walk's resume point is its own: the oldest post id read in the window,
+// passed back as `max_id:` — which gallery-dl's rewrite then replaces, rather
+// than adding a second one.
+//
+// Search needs a logged-in session; a guest run is refused before spawning.
+
+// An X handle as search's `from:` takes it. Anything else (a space, a quote, a
+// colon) would change the query's meaning rather than name an account.
+const X_HANDLE_RE = /^[A-Za-z0-9_]{1,50}$/;
+
+export function xSearchHandle(handle: string): string {
+ const bare = handle.trim().replace(/^@/, "");
+ if (!X_HANDLE_RE.test(bare)) {
+ throw new Error(
+ `"${handle}" is not an X handle search can take (letters, digits and "_")`,
+ );
}
- if (opts.stream) args.push("--verbose");
- args.push(timelineUrlFor(opts.accountUrl));
+ return bare;
+}
+
+// The query for one window. `include:nativeretweets` keeps the account's
+// reposts, as the timeline walk does (search leaves them out by default).
+export function xSearchQuery(
+ handle: string,
+ position: Pick<OlderBackfillPosition, "since" | "until" | "maxId">,
+): string {
+ const terms = [
+ `from:${xSearchHandle(handle)}`,
+ `since:${position.since}`,
+ `until:${position.until}`,
+ "include:nativeretweets",
+ ];
+ if (position.maxId) {
+ if (!/^\d+$/.test(position.maxId)) {
+ throw new Error(`"${position.maxId}" is not a post id`);
+ }
+ terms.push(`max_id:${position.maxId}`);
+ }
+ return terms.join(" ");
+}
+
+// The search URL gallery-dl's search extractor matches. The query is
+// percent-encoded whole (a space is %20, a colon %3A): gallery-dl turns "+"
+// into a space BEFORE it unquotes, so an encoded "+" survives as itself.
+// `f=live` is X's own "Latest" tab, which is also what gallery-dl asks for.
+export function xSearchUrl(query: string): string {
+ return `https://x.com/search?q=${encodeURIComponent(query)}&f=live`;
+}
+
+export function buildGalleryDlSearchArgs(
+ opts: GalleryDlLoginAndCap & {
+ handle: string;
+ position: Pick<OlderBackfillPosition, "since" | "until" | "maxId">;
+ },
+): string[] {
+ const args = galleryDlCommonArgs(opts);
+ args.push(
+ // Newest first, paged by max_id — the order the resume point relies on.
+ "-o",
+ "extractor.twitter.search-results=latest",
+ "-o",
+ "extractor.twitter.search-pagination=max_id",
+ );
+ args.push(xSearchUrl(xSearchQuery(opts.handle, opts.position)));
return args;
}
+// Inclusive: the post at the resume point is read again (and skipped as a
+// duplicate) rather than risk an off-by-one past it.
+function olderId(a: string | undefined, b: string): string {
+ if (a === undefined) return b;
+ try {
+ return BigInt(b) < BigInt(a) ? b : a;
+ } catch {
+ return a;
+ }
+}
+
+// Pause between two windows' searches. Each window is its own gallery-dl run
+// (a user lookup, then the search pages), so a stretch of empty windows would
+// otherwise be a burst of requests a second apart.
+export const OLDER_WINDOW_PAUSE_MS = 15_000;
+
+function pause(ms: number, signal: AbortSignal): Promise<void> {
+ if (ms <= 0 || signal.aborted) return Promise.resolve();
+ return new Promise((resolve) => {
+ const t = setTimeout(done, ms);
+ function done() {
+ clearTimeout(t);
+ signal.removeEventListener("abort", done);
+ resolve();
+ }
+ signal.addEventListener("abort", done, { once: true });
+ });
+}
+
// The timeline extractor's cursor is `<state>[_<tweet id>]/<API cursor>`, and
// "1/" is its first state with no API cursor: walk from the top. A run that
// stopped before gallery-dl logged any cursor resumes from here rather than
@@ -202,6 +324,11 @@ export class GalleryDlStream {
rateLimited = false;
stopReached = false;
cursor?: string;
+ // The oldest post id read this run (any post, archived or not) — a search
+ // walk's resume point.
+ oldestId?: string;
+ // The account's creation time, from a record whose `user` is the account.
+ accountCreatedAt?: string;
private previousCursor?: string;
private savedCursor?: string;
private pending: Post[] = [];
@@ -216,6 +343,8 @@ export class GalleryDlStream {
seenIds: ReadonlySet<string>;
since?: string;
stopAtKnown: boolean;
+ // The account's handle, to recognise its `user` record.
+ accountHandle?: string;
onLog?: (line: string) => void;
},
) {}
@@ -267,6 +396,14 @@ export class GalleryDlStream {
return { posts, cursor };
}
+ // Every post not yet handed over, whatever the cursor — for a walk whose
+ // resume point comes from the records themselves (oldestId), not stderr.
+ takePending(): Post[] {
+ const posts = this.pending;
+ this.pending = [];
+ return posts;
+ }
+
// A checkpoint that could not be written goes back into the queue, so the
// final write retries it.
requeue(posts: Post[]): void {
@@ -294,6 +431,18 @@ export class GalleryDlStream {
if (!post || this.ids.has(post.id)) return;
this.ids.add(post.id);
this.distinct++;
+ if (/^\d+$/.test(post.id)) this.oldestId = olderId(this.oldestId, post.id);
+ if (!this.accountCreatedAt && this.opts.accountHandle) {
+ const user = rec.user as Record<string, unknown> | undefined;
+ if (
+ user &&
+ typeof user.name === "string" &&
+ user.name.toLowerCase() === this.opts.accountHandle.toLowerCase() &&
+ typeof user.date === "string"
+ ) {
+ this.accountCreatedAt = xCreatedAt({ date: user.date });
+ }
+ }
const isKnown = this.opts.seenIds.has(post.id);
const isOlder =
!isKnown && Boolean(this.opts.since && post.createdAt <= this.opts.since);
@@ -416,6 +565,28 @@ function flattenRecords(items: unknown[]): XTweetRaw[] {
return out;
}
+// The login source (galleryDlCookieChoice) for one run, logged: the operator's
+// browser, or the session broker's exported jar — kept fresh by a live browser
+// profile, so it survives the cookie expiry that otherwise breaks gallery-dl
+// within days. One resolution for the timeline walk and the search walk.
+async function galleryDlLoginFor(
+ input: Pick<PostFetchInput, "cookieSource" | "browserCookies" | "cookies" | "onLog">,
+): Promise<{ cookies?: string; cookieFile?: string }> {
+ const paths = getPaths();
+ const status = await readXSessionStatus(paths);
+ const choice = galleryDlCookieChoice({
+ source: input.cookieSource,
+ browserCookies: input.browserCookies,
+ jarFile:
+ status.hasCookies && status.looksAuthenticated
+ ? xCookieFile(paths)
+ : undefined,
+ alwaysCookies: input.cookies,
+ });
+ input.onLog?.(choice.note);
+ return { cookies: choice.cookies, cookieFile: choice.cookieFile };
+}
+
// Feed a pipe to `onLine` a line at a time; resolves once it has closed.
function readLines(
input: Readable | null | undefined,
@@ -429,6 +600,232 @@ function readLines(
});
}
+const SEARCH_NEEDS_LOGIN =
+ "X search needs a logged-in session: connect an X account on /settings, or " +
+ "set cookiesFromBrowser for this channel (or globally), then run it again.";
+
+// The older-posts backfill: search windows walked backwards from `position`
+// (olderBackfill.ts does the window arithmetic). One gallery-dl run per
+// window, a pause between windows, the same per-run budget as a timeline
+// backfill across all of them. Progress is saved mid-window (the oldest post
+// read so far is the resume point) and at every window boundary.
+export async function fetchOlderViaSearch(
+ input: OlderPostFetchInput,
+): Promise<OlderPostFetchResult> {
+ const { channelSlug, seenIds, signal, onLog } = input;
+ const bin = getPaths().galleryDlBin;
+ let handle: string;
+ try {
+ handle = xSearchHandle(input.handle);
+ } catch (err) {
+ return {
+ posts: [],
+ complete: false,
+ position: input.position,
+ error: (err as Error).message,
+ };
+ }
+ const { cookies, cookieFile } = await galleryDlLoginFor(input);
+ if (!cookies && !cookieFile) {
+ onLog?.(`[auth] ${SEARCH_NEEDS_LOGIN}`);
+ return {
+ posts: [],
+ complete: false,
+ position: input.position,
+ needsCookies: true,
+ error: SEARCH_NEEDS_LOGIN,
+ };
+ }
+
+ const deadline = Date.now() + BACKFILL_BUDGET_MS;
+ const pauseMs = input.windowPauseMs ?? OLDER_WINDOW_PAUSE_MS;
+ let position: OlderBackfillPosition = { ...input.position };
+ let accountCreatedAt = input.accountCreatedAt;
+ // Posts not handed to a checkpoint, returned for the final write.
+ const carried: Post[] = [];
+ let checkpointsOn = Boolean(input.onCheckpoint);
+ let read = 0;
+
+ const save = async (posts: Post[], at: OlderBackfillPosition) => {
+ if (!checkpointsOn || !input.onCheckpoint) {
+ carried.push(...posts);
+ return;
+ }
+ try {
+ await input.onCheckpoint({ posts, position: at, accountCreatedAt });
+ } catch (err) {
+ // Keep the posts for the final write and stop checkpointing, so a saved
+ // position never runs ahead of what is on disk.
+ checkpointsOn = false;
+ carried.push(...posts);
+ onLog?.(`[warn] could not save progress mid-run: ${(err as Error).message}`);
+ }
+ };
+ const stopped = (
+ at: OlderBackfillPosition,
+ extra: Partial<OlderPostFetchResult> = {},
+ ): OlderPostFetchResult => ({
+ posts: carried,
+ complete: false,
+ position: at,
+ accountCreatedAt,
+ ...extra,
+ });
+
+ onLog?.(
+ `Searching @${handle}'s posts backwards, ${position.since} – ${position.until}` +
+ (position.maxId ? " from the saved resume point" : "") +
+ ` (up to ${Math.round(BACKFILL_BUDGET_MS / 60_000)} min this run; progress is saved as it goes)`,
+ );
+
+ for (;;) {
+ const floor = olderFloorDay(input.floor, accountCreatedAt);
+ if (floor !== undefined && position.until <= floor) {
+ return {
+ posts: carried,
+ complete: true,
+ completeReason: `the archive already reaches back to ${floor}, the earliest date to walk to`,
+ position,
+ accountCreatedAt,
+ };
+ }
+ const remainingMs = deadline - Date.now();
+ if (remainingMs <= 0) {
+ onLog?.(`Stopped at this run's ${Math.round(BACKFILL_BUDGET_MS / 60_000)}-minute budget; the next run resumes at ${position.since} – ${position.until}.`);
+ return stopped(position);
+ }
+ const remainingLimit = input.limit ? input.limit - read : undefined;
+ if (remainingLimit !== undefined && remainingLimit <= 0) {
+ return stopped(position);
+ }
+
+ const windowStart = position;
+ const stream = new GalleryDlStream({
+ channelSlug,
+ seenIds,
+ stopAtKnown: false,
+ accountHandle: handle,
+ onLog,
+ });
+ const args = buildGalleryDlSearchArgs({
+ handle,
+ position: windowStart,
+ cookies,
+ cookieFile,
+ limit: remainingLimit,
+ });
+ const subprocess = execa(bin, args, {
+ reject: false,
+ buffer: false,
+ stdin: "ignore",
+ timeout: remainingMs,
+ cancelSignal: signal,
+ env: { PYTHONUNBUFFERED: "1" },
+ });
+
+ // Mid-window checkpoints, one at a time and in order. The resume point is
+ // the oldest post read so far, which every post taken here is at or above.
+ let checkpoints: Promise<void> = Promise.resolve();
+ let lastCheckpointAt = Date.now();
+ const maybeCheckpoint = () => {
+ if (!checkpointsOn) return;
+ const due =
+ stream.pendingCount() >= CHECKPOINT_EVERY_POSTS ||
+ (stream.pendingCount() > 0 &&
+ Date.now() - lastCheckpointAt >= CHECKPOINT_EVERY_MS);
+ if (!due || !stream.oldestId) return;
+ lastCheckpointAt = Date.now();
+ const posts = stream.takePending();
+ const at = { ...windowStart, maxId: stream.oldestId };
+ checkpoints = checkpoints.then(() => save(posts, at));
+ };
+ const stdoutDone = readLines(subprocess.stdout, (line) => {
+ stream.pushStdout(line);
+ maybeCheckpoint();
+ });
+ const stderrDone = readLines(subprocess.stderr, (line) => {
+ stream.pushStderr(line);
+ maybeCheckpoint();
+ });
+ const res = await subprocess;
+ await Promise.all([stdoutDone, stderrDone]);
+ await checkpoints;
+
+ const spawnError = `${res.code ?? ""} ${res.message ?? ""}`;
+ if (res.failed && /ENOENT/.test(spawnError) && stream.distinct === 0) {
+ throw new Error(
+ `gallery-dl not found (looked for "${bin}"; set GALLERY_DL_BIN)`,
+ );
+ }
+
+ read += stream.distinct;
+ accountCreatedAt ??= stream.accountCreatedAt;
+ carried.push(...stream.finish());
+ const here: OlderBackfillPosition = {
+ ...windowStart,
+ maxId: stream.oldestId ?? windowStart.maxId,
+ };
+ if (stream.rateLimited) {
+ onLog?.("[rate-limit] X throttled this search — gallery-dl paused between pages.");
+ }
+ onLog?.(
+ `${windowStart.since} – ${windowStart.until}: read ${stream.distinct} post(s), ${stream.newPosts} new` +
+ (stream.known ? `, ${stream.known} already archived` : ""),
+ );
+
+ if (res.isCanceled || signal.aborted || res.timedOut) {
+ onLog?.(
+ res.timedOut
+ ? `Stopped at this run's ${Math.round(BACKFILL_BUDGET_MS / 60_000)}-minute budget; the next run resumes where this one left off.`
+ : "Cancelled; the next run resumes where this one left off.",
+ );
+ return stopped(here);
+ }
+ if (res.exitCode !== 0) {
+ const tail = stream.stderrTail();
+ if (looksLikeAuthFailure(tail)) {
+ onLog?.(`[auth] gallery-dl could not authenticate to X:\n${tail}`);
+ return stopped(here, { needsCookies: true, error: SEARCH_NEEDS_LOGIN });
+ }
+ return stopped(here, {
+ error: `gallery-dl exited ${res.exitCode ?? res.signal ?? "abnormally"}: ${tail || "(no output)"}`,
+ });
+ }
+ if (remainingLimit !== undefined && stream.distinct >= remainingLimit) {
+ return stopped(here);
+ }
+
+ // The window is walked to its end.
+ const step = stepOlderWindow(windowStart, {
+ // A window resumed below a saved point held posts above it.
+ hadPosts: stream.distinct > 0 || Boolean(windowStart.maxId),
+ floor: olderFloorDay(input.floor, accountCreatedAt),
+ });
+ if (step.complete) {
+ onLog?.(`Older-posts backfill complete: ${step.reason}.`);
+ return {
+ posts: carried,
+ complete: true,
+ completeReason: step.reason,
+ position: { since: windowStart.since, until: windowStart.until, emptyWindows: step.emptyWindows },
+ accountCreatedAt,
+ };
+ }
+ position = step.next;
+ // Every window boundary is saved: the posts so far, and the next window.
+ await save(carried.splice(0), position);
+ if (signal.aborted) {
+ onLog?.("Cancelled; the next run resumes where this one left off.");
+ return stopped(position);
+ }
+ await pause(pauseMs, signal);
+ if (signal.aborted) {
+ onLog?.("Cancelled; the next run resumes where this one left off.");
+ return stopped(position);
+ }
+ }
+}
+
export const xGalleryDlFetcher: SocialFetcher = {
id: "x-gallery-dl",
label: "X / Twitter (gallery-dl)",
@@ -498,23 +895,8 @@ export const xGalleryDlFetcher: SocialFetcher = {
async fetch(input: PostFetchInput): Promise<PostFetchResult> {
const { accountUrl, channelSlug, since, seenIds, limit, signal, onLog } =
input;
- const paths = getPaths();
- const bin = paths.galleryDlBin;
- // The login source (galleryDlCookieChoice): the operator's browser, or the
- // session broker's exported jar — kept fresh by a live browser profile, so
- // it survives the cookie expiry that otherwise breaks gallery-dl within days.
- const status = await readXSessionStatus(paths);
- const choice = galleryDlCookieChoice({
- source: input.cookieSource,
- browserCookies: input.browserCookies,
- jarFile:
- status.hasCookies && status.looksAuthenticated
- ? xCookieFile(paths)
- : undefined,
- alwaysCookies: input.cookies,
- });
- onLog?.(choice.note);
- const { cookies, cookieFile } = choice;
+ const bin = getPaths().galleryDlBin;
+ const { cookies, cookieFile } = await galleryDlLoginFor(input);
const args = buildGalleryDlArgs({
accountUrl,
cookies,
@@ -673,6 +1055,8 @@ export const xGalleryDlFetcher: SocialFetcher = {
? { posts, complete: false, cursor: resumeAt }
: { posts, complete: true };
},
+
+ fetchOlder: fetchOlderViaSearch,
};
registerSocialFetcher(xGalleryDlFetcher);
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -2,6 +2,7 @@
## [Unreleased]
- **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched.
+- **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history.
- **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete.
- **`pnpm ops retry-bucket` can run part of a bucket.** Its body takes `"ids"`, a list of video ids, and runs only those, as ticking them on the bucket's card does. Every id must be in the named bucket: one that is not is refused with a 400 naming it, and nothing runs. A job started with `ids` is not replayable, like a checkbox selection in the UI.
- **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen.
diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts
@@ -1,6 +1,6 @@
import test from "node:test";
import assert from "node:assert/strict";
-import { OpsInputError, optSubset } from "./_lib";
+import { OpsInputError, optPositiveInt, optSubset } from "./_lib";
// Run with:
// pnpm -C editor exec tsx --test "app/**/*.test.ts"
@@ -36,3 +36,19 @@ test("optSubset: an empty or non-string list is refused, never read as 'all'", (
);
}
});
+
+test("optPositiveInt: absent is undefined; a whole number above zero comes back", () => {
+ assert.equal(optPositiveInt({}, "limit"), undefined);
+ assert.equal(optPositiveInt({ limit: 200 }, "limit"), 200);
+});
+
+test("optPositiveInt: zero, negatives, fractions and strings are refused", () => {
+ for (const limit of [0, -5, 2.5, "10", null, Number.NaN]) {
+ assert.throws(
+ () => optPositiveInt({ limit }, "limit"),
+ (e: unknown) =>
+ e instanceof OpsInputError && /"limit" must be a whole number above zero/.test(e.message),
+ JSON.stringify(limit),
+ );
+ }
+});
diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts
@@ -146,6 +146,17 @@ export function optBool(body: OpsBody, key: string): boolean | undefined {
return v;
}
+// A whole number above zero, or undefined when absent. A type check, not a
+// rule: what the number means is the action's business.
+export function optPositiveInt(body: OpsBody, key: string): number | undefined {
+ const v = body[key];
+ if (v === undefined) return undefined;
+ if (typeof v !== "number" || !Number.isInteger(v) || v <= 0) {
+ throw new OpsInputError(`"${key}" must be a whole number above zero`);
+ }
+ return v;
+}
+
export function reqStringArray(body: OpsBody, key: string): string[] {
const v = body[key];
if (
diff --git a/editor/app/api/ops/fetch-posts/route.ts b/editor/app/api/ops/fetch-posts/route.ts
@@ -0,0 +1,41 @@
+import { fetchPostsAction } from "../../../channels/[slug]/socialActions";
+import {
+ jobResponse,
+ ops,
+ optBool,
+ optPositiveInt,
+ optString,
+ reqSlug,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, full?, older?, floor?, limit?, queueKey? } -> { ok: true, jobId }
+//
+// A social channel's "Fetch posts" button, over HTTP — and its "Re-fetch full
+// history" (`full`) and "Fetch older posts" (`older`, with an optional `floor`
+// date, YYYY-MM-DD, that the walk stops at). The job runs on the platform's
+// queue, as the buttons' does.
+//
+// Every refusal is the action's own sentence: a channel that is not a social
+// one, `full` and `older` together, a fetcher with no older walk, a floor
+// without `older` or that is not a date.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["slug", "full", "older", "floor", "limit", "queueKey"],
+ async (body) => {
+ const slug = reqSlug(body, "slug");
+ return jobResponse(
+ await fetchPostsAction(
+ slug,
+ optString(body, "queueKey"),
+ optBool(body, "full"),
+ optPositiveInt(body, "limit"),
+ optBool(body, "older"),
+ optString(body, "floor"),
+ ),
+ );
+ },
+ );
+}
diff --git a/editor/app/channels/[slug]/components/SocialChannelPanel.tsx b/editor/app/channels/[slug]/components/SocialChannelPanel.tsx
@@ -30,6 +30,10 @@ export type SocialChannelState = {
deletedCount: number;
checkedCount: number;
canCheckAvailability: boolean;
+ // The older-posts backfill (a fetcher's `fetchOlder`): whether this fetcher
+ // has one, and where it stands, as one line.
+ canFetchOlder: boolean;
+ olderStatus: string;
};
export function SocialChannelPanel({
@@ -63,9 +67,15 @@ export function SocialChannelPanel({
if (res.ok) void res.stream.cancel();
});
- const run = (full: boolean) =>
+ const run = (full: boolean, older = false) =>
startTransition(async () => {
- const res = await fetchPostsAction(slug, undefined, full);
+ const res = await fetchPostsAction(
+ slug,
+ undefined,
+ full,
+ undefined,
+ older || undefined,
+ );
// The job streams to the jobs page; nothing to consume here, but the
// stream must be released or its buffered chunks leak.
if (res.ok) void res.stream.cancel();
@@ -111,6 +121,9 @@ export function SocialChannelPanel({
: String(state.lastFetchedCount)
}
/>
+ {state.canFetchOlder && (
+ <Stat label="Older posts" value={state.olderStatus} />
+ )}
</dl>
{state.availableFetchers.length > 1 && (
@@ -181,6 +194,18 @@ export function SocialChannelPanel({
>
Re-fetch full history
</button>
+ {state.canFetchOlder && (
+ <button
+ type="button"
+ onClick={() => run(false, true)}
+ disabled={pending}
+ aria-label="fetch older posts"
+ title="Walk back from the oldest archived post through search, a few months at a time, for posts older than the timeline reaches. Needs a login. Progress is saved as it goes; a later run continues it, and it says so once the history is done."
+ className="rounded-md border border-border px-3 py-1 text-sm hover:bg-muted disabled:opacity-50"
+ >
+ Fetch older posts
+ </button>
+ )}
</div>
{/* The two failure surfaces a social channel actually has. */}
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -11,6 +11,7 @@ import {
} from "yt-dlp-transcript-common/lib/posts-server";
import { isPostGone } from "yt-dlp-transcript-common/lib/posts";
import { resolveSocialFetcher } from "yt-dlp-transcript-common/social/fetchers";
+import { describeOlderBackfill } from "yt-dlp-transcript-common/social/olderBackfill";
import "yt-dlp-transcript-common/social/blueskyFetcher";
import "yt-dlp-transcript-common/social/xGalleryDlFetcher";
import "yt-dlp-transcript-common/social/xPlaywrightFetcher";
@@ -186,6 +187,8 @@ export default async function ChannelDetailPage({
checkedCount: availabilityRecords.length,
canCheckAvailability:
typeof fetcher?.checkAvailability === "function",
+ canFetchOlder: typeof fetcher?.fetchOlder === "function",
+ olderStatus: describeOlderBackfill(fetchState?.older),
handle: config.socialHandle ?? slug,
accountUrl: config.url,
}}
diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts
@@ -19,7 +19,12 @@ import {
readChannelConfig,
} from "yt-dlp-transcript-common/controller/channels";
import { isSocialChannel } from "yt-dlp-transcript-common/lib/channelConfig";
-import { fetchPosts } from "yt-dlp-transcript-common/controller/fetchPosts";
+import {
+ fetchPosts,
+ FULL_AND_OLDER_REFUSAL,
+ olderPostsProblem,
+} from "yt-dlp-transcript-common/controller/fetchPosts";
+import { isUtcDay } from "yt-dlp-transcript-common/social/olderBackfill";
import {
checkPostAvailability,
type CheckPostAvailabilityMode,
@@ -27,6 +32,7 @@ import {
import {
getSocialFetcher,
listSocialFetchers,
+ resolveSocialFetcher,
} from "yt-dlp-transcript-common/social/fetchers";
import {
runManagedFunction,
@@ -121,18 +127,39 @@ export async function checkPostAvailabilityAction(
});
}
+// `older` walks the account's history backwards from the oldest archived post
+// (the fetcher's `fetchOlder` — X: search windows) instead of fetching new
+// posts; `floor` (YYYY-MM-DD) is the date that walk stops at. Both are refused
+// HERE, before a job exists, when they cannot run: with `full`, on a fetcher
+// that has no older walk, or with a floor that is not a date.
export async function fetchPostsAction(
slug: string,
queueKey?: string,
full?: boolean,
limit?: number,
+ older?: boolean,
+ floor?: string,
): Promise<StreamActionResult> {
+ if (full && older) return { ok: false, error: FULL_AND_OLDER_REFUSAL };
+ if (floor !== undefined && !older) {
+ return { ok: false, error: "A floor date applies only to an older-posts fetch." };
+ }
+ if (floor !== undefined && !isUtcDay(floor)) {
+ return { ok: false, error: `"${floor}" is not a date (YYYY-MM-DD).` };
+ }
const paths = getPaths();
const config = await readChannelConfig(paths, slug);
if (!config) return { ok: false, error: `No such channel: ${slug}` };
if (!isSocialChannel(config)) {
return { ok: false, error: `${slug} is not a social channel.` };
}
+ if (older) {
+ await registerBuiltinSocialFetchers();
+ const problem = olderPostsProblem(
+ resolveSocialFetcher(config.postFetcher, config.url),
+ );
+ if (problem) return { ok: false, error: problem };
+ }
// queueKeyForUrl() already routes x.com / bsky.app to platform:x.com /
// platform:bsky.app, so per-platform serialization and the existing 429
@@ -144,13 +171,19 @@ export async function fetchPostsAction(
queueKey: key,
paths,
channelSlug: slug,
- spec: { kind: "fetch-posts", slug, params: { queueKey, full, limit } },
+ spec: {
+ kind: "fetch-posts",
+ slug,
+ params: { queueKey, full, limit, older, floor },
+ },
fn: async (onLog, signal) => {
const result = await fetchPosts({
paths,
slug,
settings: getSettings(),
full,
+ older,
+ floor,
limit,
onLog,
signal,
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -246,7 +246,14 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
},
"fetch-posts": (spec) => {
const { p, queueKey } = params(spec);
- return fetchPostsAction(spec.slug, queueKey, bool(p.full), num(p.limit));
+ return fetchPostsAction(
+ spec.slug,
+ queueKey,
+ bool(p.full),
+ num(p.limit),
+ bool(p.older),
+ str(p.floor),
+ );
},
"download-missing-subs": (spec) => {
const { p, queueKey } = params(spec);
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -192,6 +192,7 @@ test("a traversing slug is refused at the door, on every route that takes one",
["channel-priority", { slugs: ["../../escape"], tier: "paused" }],
["relocate", { slugs: ["../../escape"], root: "/tmp/ops-api-never" }],
["relocate-back", { slugs: ["../../escape"] }],
+ ["fetch-posts", { slug: "../../escape", older: true }],
];
for (const [action, data] of cases) {
const { status, body } = await ops(request, action, data);
@@ -267,6 +268,73 @@ test("retry-bucket and transcribe-bucket ids must be in the bucket: a stray is r
expect(await listJobIds()).toEqual(before);
});
+test("fetch-posts refuses a channel that is not social, full with older, and an older fetch its fetcher cannot do — before any job", async ({
+ request,
+}) => {
+ await resetData("title-filter-channel");
+ await settings();
+ // A social channel whose fetcher has no older-posts walk. Nothing below
+ // reaches a fetch: every case is refused before a job exists.
+ await writeChannelConfig("example-bsky", {
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "bluesky",
+ postFetcher: "bluesky-atproto",
+ socialHandle: "example.bsky.social",
+ name: "Example (Bluesky)",
+ url: "https://bsky.app/profile/example.bsky.social",
+ });
+ const before = await listJobIds();
+
+ // The sentence the action gives for a video channel.
+ const video = await ops(request, "fetch-posts", { slug: "test-filter" });
+ expect(video.status).toBe(400);
+ expect(video.body.error).toBe("test-filter is not a social channel.");
+ const videoOlder = await ops(request, "fetch-posts", { slug: "test-filter", older: true });
+ expect(videoOlder.status).toBe(400);
+ expect(videoOlder.body.error).toBe("test-filter is not a social channel.");
+
+ // Two walks at once, on any channel.
+ const both = await ops(request, "fetch-posts", {
+ slug: "example-bsky",
+ full: true,
+ older: true,
+ });
+ expect(both.status).toBe(400);
+ expect(both.body.error).toMatch(/different walks — run one at a time/);
+ const bothVideo = await ops(request, "fetch-posts", {
+ slug: "test-filter",
+ full: true,
+ older: true,
+ });
+ expect(bothVideo.status).toBe(400);
+ expect(bothVideo.body.error).toMatch(/different walks/);
+
+ // A fetcher with no older walk, a floor that is not a date, a floor
+ // without older, and a limit that is not a count.
+ const bsky = await ops(request, "fetch-posts", { slug: "example-bsky", older: true });
+ expect(bsky.status).toBe(400);
+ expect(bsky.body.error).toMatch(/cannot fetch older posts/);
+ const floor = await ops(request, "fetch-posts", {
+ slug: "example-bsky",
+ older: true,
+ floor: "2020-13-01",
+ });
+ expect(floor.status).toBe(400);
+ expect(floor.body.error).toMatch(/is not a date/);
+ const floorAlone = await ops(request, "fetch-posts", {
+ slug: "example-bsky",
+ floor: "2020-01-01",
+ });
+ expect(floorAlone.status).toBe(400);
+ expect(floorAlone.body.error).toMatch(/only to an older-posts fetch/);
+ const limit = await ops(request, "fetch-posts", { slug: "example-bsky", limit: 0 });
+ expect(limit.status).toBe(400);
+ expect(limit.body.error).toMatch(/"limit" must be a whole number above zero/);
+
+ expect(await listJobIds()).toEqual(before);
+});
+
test("channel-config round-trips a download filter and refuses a bad regex", async ({
page,
request,
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -8534,6 +8534,26 @@ phase deletes from the destination.
`expiry` is seconds (read as ms when too large for seconds); `lastAccessed` is PRTime (µs). The
Chromium family is NOT read here (its values are encrypted with a keyring key): the readers say so
and gallery-dl reads it.
+- **gallery-dl 1.32.9's `twitter:search`** (`extractor/twitter.py`, `TwitterSearchExtractor`)
+ matches `…/search?…q=<query>` (q may follow other params; `&f=live` after it is fine) and decodes q
+ as `unquote(q.replace("+", " "))` — so the query is percent-encoded whole and an encoded `+`
+ survives. It reads the query's single `from:<user>` to fill each record's `user` (with the
+ account's `date`, its creation time). Product defaults to "Latest"; `search-pagination` defaults
+ to `max_id`: after each page it REWRITES the query's `max_id:` (adding one if absent) to just below
+ the oldest tweet read and sets the API cursor to None — so **a search run logs no
+ `Cursor:` line and cannot be resumed with `extractor.twitter.cursor`**. It stops after
+ `search-stop` (default 3) empty pages. Quoted tweets are skipped unless `quoted` is set.
+- **The older-posts backfill** (`fetchPosts({ older })` → `SocialFetcher.fetchOlder`; for X
+ `fetchOlderViaSearch` in `xGalleryDlFetcher.ts`, window arithmetic in `social/olderBackfill.ts`).
+ Windows of 3 months in UTC days, `from:<h> since:<d> until:<d> include:nativeretweets`, the first
+ ending the day after the oldest archived post; one gallery-dl run per window, 15 s apart, one
+ 3-hour budget across them. Its resume point is its own: the oldest post id read in the window,
+ sent back as `max_id:<id>` (inclusive; gallery-dl's rewrite replaces it). State is
+ `posts-state.json` `older` (`OlderBackfillState`), apart from the timeline `cursor`; a normal fetch
+ copies `older` through untouched and an older run copies every other field through. Ends at the
+ later of the `floor` and the account's creation day, or after 4 empty windows in a row; then
+ `older.complete` and a re-run fetches nothing. No login = refused before spawning
+ (`needsCookies`). It does not stamp `lastSyncedAt`.
- **`node:sqlite`** (Node 22.23 here; unflagged since 22.13, an ExperimentalWarning once per process)
is loaded at run time through an assembled specifier with `webpackIgnore`
(`common/social/nodeSqlite.ts`), like `playwrightRuntime.ts`: the editor's Turbopack build has
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -100,6 +100,9 @@ const ACTIONS = [
// The TRANSCRIBE half of a channel's "downloaded, not transcribed" bucket
// (retry-bucket is the download retry and skips audio already on disk).
"transcribe-bucket",
+ // A social channel's post fetch: new posts, the full re-walk ("full"), or
+ // the walk back below the oldest archived post ("older").
+ "fetch-posts",
"build-index",
"build-deploy",
"build-site",
@@ -317,6 +320,12 @@ export function usage() {
' bucket on the transcription queue, as the channel page\'s Transcribe button',
' does: {"slug"}. "ids": [...] narrows it the same way as on retry-bucket.',
"",
+ 'fetch-posts fetches a social channel\'s new posts: {"slug"}. "full": true',
+ ' re-walks the whole timeline; "older": true walks back from the oldest',
+ " archived post through search (X; needs a login), saving its place for",
+ ' the next run, down to "floor": "YYYY-MM-DD" when given. "limit": N caps',
+ ' the posts one run reads. "full" and "older" together are refused.',
+ "",
'"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare',
" Pages PREVIEW instead of production: the same bundle goes to a branch",
" alias, https://<branch>.<project>.pages.dev, and the live site is left",
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -378,3 +378,15 @@ test("build-homepage and deploy-homepage are POSTs to their own routes, named in
assert.match(usage(), /build-homepage builds the homepage package into homepage\/out/);
assert.match(usage(), /project archilyzer \(https:\/\/archilyzer\.pages\.dev\)/);
});
+
+// A social channel's post fetch: a POST to its route, the body passed through
+// untouched — the route and the action judge it ("full" with "older" included).
+test("fetch-posts is a POST to its route, named in the usage", () => {
+ const p = parseArgs(["fetch-posts", "--json", '{"slug":"example-x","older":true}', "--wait"]);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/fetch-posts");
+ assert.deepEqual(p.body, { slug: "example-x", older: true });
+ assert.equal(p.wait, true);
+ assert.match(usage(), /Actions:.*transcribe-bucket, fetch-posts/);
+ assert.match(usage(), /"older": true walks back from the oldest/);
+});