Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 2ddc49784e4a347db52aa13e64ee733836498787
parent e0bbb6a7266dd272bddd7b43e4664fa6b1926a7a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  3 Oct 2026 13:37:47 -0400

Merge x-search-backfill (X posts older than the timeline reaches: a search walk in three-month windows from the oldest archived post, resumable via max_id, its own state beside the timeline cursor; Fetch older posts button, fetch-posts ops route, --older/--floor)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
MRUNNING_IN_DOCKER.md | 1+
Mcommon/bin/archilyzer.ts | 2+-
Mcommon/bin/fetch-posts.ts | 13+++++++++++--
Acommon/controller/fetchOlderPosts.test.ts | 398+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/fetchPosts.ts | 242+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/posts-server.ts | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/fetchers.ts | 53+++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/olderBackfill.test.ts | 116+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/olderBackfill.ts | 152+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.test.ts | 98+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.ts | 448+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Meditor/CHANGELOG.md | 1+
Meditor/app/api/ops/_lib.test.ts | 18+++++++++++++++++-
Meditor/app/api/ops/_lib.ts | 11+++++++++++
Aeditor/app/api/ops/fetch-posts/route.ts | 41+++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/SocialChannelPanel.tsx | 29+++++++++++++++++++++++++++--
Meditor/app/channels/[slug]/page.tsx | 3+++
Meditor/app/channels/[slug]/socialActions.ts | 37+++++++++++++++++++++++++++++++++++--
Meditor/app/jobs/jobReplayRegistry.ts | 9++++++++-
Meditor/e2e/ops-api.spec.ts | 68++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mplans/FACTS.md | 20++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 9+++++++++
Mscripts/archilyzer-ops.test.mjs | 12++++++++++++
23 files changed, 1793 insertions(+), 41 deletions(-)

diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md @@ -254,6 +254,7 @@ pnpm ops channel-priority --json '{"slugs":["the-quartering"],"operation":"downl pnpm ops lane --json '{"lane":"download","held":true}' pnpm ops refresh-report --json '{"all":true}' pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}' +pnpm ops fetch-posts --json '{"slug":"example-x","older":true}' --wait pnpm ops get channel the-quartering pnpm ops list # every action name ``` diff --git a/common/bin/archilyzer.ts b/common/bin/archilyzer.ts @@ -252,7 +252,7 @@ export const COMMANDS: Command[] = [ script(["duplicates"], "duplicate-shorts.ts", "[--threshold N] [--all-durations] [--blocking title|duration|both] [--near F] [--tolerance N] … on-demand duplicate detection (after index + stats)", 8192), script(["posts", "fetch"], "fetch-posts.ts", - "--slug <channel> [--full] [--limit N] fetch a social channel's posts into its posts corpus"), + "--slug <channel> [--full | --older [--floor YYYY-MM-DD]] [--limit N] fetch a social channel's posts into its posts corpus (--older: walk back below the oldest archived post)"), script(["posts", "check"], "check-post-availability.ts", "--slug <channel> [--mode stale|unchecked|all] [--limit N] which archived posts were deleted at the source"), script(["diarize", "backfill"], "diarize-backfill.ts", diff --git a/common/bin/fetch-posts.ts b/common/bin/fetch-posts.ts @@ -2,11 +2,16 @@ // Fetch social posts for one channel into its on-disk posts corpus. // // pnpm --filter yt-dlp-transcript-common exec tsx bin/fetch-posts.ts \ -// --slug <channel-slug> [--full] [--limit N] +// --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD]] [--limit N] // // --full re-walks the account's whole history instead of stopping at the // stored watermark. New posts are still deduped against the posts-archive, so // it repairs a gap rather than creating duplicates. +// +// --older walks the account's history backwards from the oldest archived post +// (X: search windows), below what the timeline reaches; --floor is the date it +// stops at. Its position is saved apart from the timeline's, so a later run +// continues it and a normal fetch is unaffected. import { getPaths } from "../lib/paths"; import { getSettings } from "../lib/settings"; @@ -16,7 +21,9 @@ import { parseFlags } from "./_parseFlags"; const flags = parseFlags(process.argv.slice(2)); const slug = flags.slug; if (!slug) { - console.error("Usage: fetch-posts.ts --slug <channel-slug> [--full] [--limit N]"); + console.error( + "Usage: fetch-posts.ts --slug <channel-slug> [--full | --older [--floor YYYY-MM-DD]] [--limit N]", + ); process.exit(2); } @@ -31,6 +38,8 @@ fetchPosts({ slug, settings: getSettings(), full: flags.full === "true", + older: flags.older === "true", + floor: flags.floor, limit, onLog: (line) => console.log(line), }) diff --git a/common/controller/fetchOlderPosts.test.ts b/common/controller/fetchOlderPosts.test.ts @@ -0,0 +1,398 @@ +// fetchPosts({ older: true }) over the gallery-dl X fetcher, end to end +// against a fake binary that answers X SEARCH URLs from a fixed set of posts: +// the walk steps back a window at a time, writes through the same shard +// writer, records its own position apart from the timeline cursor, stops at a +// floor / the account's creation date / a run of empty windows, and a normal +// fetch never resumes backwards from it. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/fetchOlderPosts.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { chmod, mkdir, mkdtemp, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "fetcholder-")); +const BIN = path.join(ROOT, "fake-gallery-dl.mjs"); +const ARGS_LOG = path.join(ROOT, "argv.jsonl"); +process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); +process.env.GALLERY_DL_BIN = BIN; +process.env.FAKE_ARGS_LOG = ARGS_LOG; + +// The account's posts, by date. An id is X's: the creation time in its high +// bits, so ids order as dates do. +const POST_DATES = [ + "2021-03-10 12:00:00", // the oldest post already archived + "2021-02-01 09:00:00", + "2020-12-15 18:30:00", + "2020-06-01 08:00:00", + "2019-11-20 22:15:00", +]; + +// The fake: a timeline URL reads nothing (the timeline is exhausted); a search +// URL answers its query's from:/since:/until:/max_id: from POST_DATES, newest +// first, honouring --range. FAKE_ACCOUNT_DATE is the account's creation date +// carried on each record's `user`. FAKE_MODE=stall prints the window's first +// post, then sleeps on a "rate limit" until killed. +await writeFile( + BIN, + `#!/usr/bin/env node +import { appendFileSync } from "node:fs"; +const args = process.argv.slice(2); +appendFileSync(process.env.FAKE_ARGS_LOG, JSON.stringify(args) + "\\n"); +const url = new URL(args[args.length - 1]); +if (!url.pathname.startsWith("/search")) process.exit(0); +const q = url.searchParams.get("q"); +const term = (k) => (q.split(" ").find((t) => t.startsWith(k + ":")) || "").slice(k.length + 1) || undefined; +const from = term("from"), since = term("since"), until = term("until"), maxId = term("max_id"); +const range = args[args.indexOf("--range") + 1]; +const cap = args.includes("--range") ? Number(range.split("-")[1]) : Infinity; +const dates = ${JSON.stringify(POST_DATES)}; +const idOf = (d) => (BigInt(Date.parse(d.replace(" ", "T") + "Z") - 1288834974657) << 22n); +const hits = dates + .filter((d) => d.slice(0, 10) >= since && d.slice(0, 10) < until) + .filter((d) => !maxId || idOf(d) <= BigInt(maxId)) + .sort().reverse(); +const user = { name: from, nick: "Example" }; +if (process.env.FAKE_ACCOUNT_DATE) user.date = process.env.FAKE_ACCOUNT_DATE; +const line = (d) => '[2, {"tweet_id": ' + idOf(d) + ', "date": "' + d + '", "content": "post at ' + d + + '", "author": ' + JSON.stringify(user) + ', "user": ' + JSON.stringify(user) + '}]\\n'; +if (process.env.FAKE_MODE === "stall" && hits.length) { + process.stdout.write(line(hits[0])); + process.stderr.write("[twitter][info] Waiting for 14 minutes until 15:11:22 (rate limit)\\n"); + setTimeout(() => process.exit(0), 60_000); +} else { + hits.slice(0, cap).forEach((d) => process.stdout.write(line(d))); + process.exit(0); +} +`, +); +await chmod(BIN, 0o755); + +const HANDLE = "example_user"; +const channelsDir = path.join(ROOT, "transcripts", "channels"); + +async function makeChannel(slug: string): Promise<string> { + const channelRoot = path.join(channelsDir, slug); + await mkdir(channelRoot, { recursive: true }); + await writeFile( + path.join(channelRoot, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + postFetcher: "x-gallery-dl", + socialHandle: HANDLE, + platform: "twitter", + name: "Example (X)", + url: `https://x.com/${HANDLE}`, + }), + ); + return channelRoot; +} + +const { fetchPosts, FULL_AND_OLDER_REFUSAL, olderPostsProblem } = await import("./fetchPosts"); +const { + readPostFetchState, + readSeenPostIds, + writePostFetchState, + writePosts, + readAllPosts, +} = await import("../lib/posts-server"); +const { normalizeXTweet } = await import("../social/xNormalize"); +const { getPaths } = await import("../lib/paths"); + +const LOGIN = { cookiesFromBrowser: "firefox" }; + +function idOf(d: string): string { + return String( + BigInt(Date.parse(d.replace(" ", "T") + "Z") - 1288834974657) << 22n, + ); +} + +// The archive as a normal fetch would have left it: the oldest post, plus a +// timeline resume point. +async function seed(channelRoot: string, slug: string): Promise<void> { + const post = normalizeXTweet( + { + tweet_id: idOf(POST_DATES[0]), + date: POST_DATES[0], + content: "archived", + author: { name: HANDLE }, + user: { name: HANDLE }, + }, + slug, + ); + assert.ok(post); + await writePosts(channelRoot, [post]); + await writePostFetchState(channelRoot, { + lastFetchedAt: "2026-01-01T00:00:00.000Z", + cursor: "1/TIMELINE-RESUME", + }); +} + +async function argvLines(): Promise<string[][]> { + try { + return (await readFile(ARGS_LOG, "utf8")) + .trim() + .split("\n") + .filter(Boolean) + .map((l) => JSON.parse(l) as string[]); + } catch { + return []; + } +} + +const queryOf = (argv: string[]) => + new URL(argv[argv.length - 1]).searchParams.get("q") ?? ""; + +test("an older fetch walks back window by window and stops after four empty windows", async () => { + const slug = "older-walk"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + const before = (await argvLines()).length; + const log: string[] = []; + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + olderWindowPauseMs: 0, + onLog: (l) => log.push(l), + }); + assert.equal(result.ok, true, log.join("\n")); + assert.equal(result.complete, true); + assert.equal(result.written, 4); + assert.equal((await readSeenPostIds(channelRoot)).size, 5); + + const runs = (await argvLines()).slice(before); + const queries = runs.map(queryOf); + // Windows tile backwards from the day after the oldest archived post. + assert.deepEqual(queries.slice(0, 3), [ + `from:${HANDLE} since:2020-12-11 until:2021-03-11 include:nativeretweets`, + `from:${HANDLE} since:2020-09-11 until:2020-12-11 include:nativeretweets`, + `from:${HANDLE} since:2020-06-11 until:2020-09-11 include:nativeretweets`, + ]); + // Ten windows: three with posts, the run of four empty ones at the end, and + // the empty ones between that a later post reset. + assert.equal(runs.length, 10, queries.join("\n")); + assert.match(queries[9], /since:2018-09-11 until:2018-12-11/); + + const state = await readPostFetchState(channelRoot); + assert.equal(state?.older?.complete, true); + assert.match(state?.older?.completeReason ?? "", /4 windows of 3 months in a row held no posts/); + // The timeline's own state is exactly as the last normal fetch left it. + assert.equal(state?.cursor, "1/TIMELINE-RESUME"); + assert.equal(state?.lastFetchedAt, "2026-01-01T00:00:00.000Z"); + // Not a sync: the channel's sync stamp is not touched. + const config = JSON.parse(await readFile(path.join(channelRoot, "config.json"), "utf8")); + assert.equal(config.lastSyncedAt, undefined); + + // Re-running says so, and searches nothing. + const again: string[] = []; + const rerun = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + olderWindowPauseMs: 0, + onLog: (l) => again.push(l), + }); + assert.equal(rerun.ok, true); + assert.equal(rerun.written, 0); + assert.equal((await argvLines()).length, before + 10); + assert.ok(again.some((l) => /already complete/.test(l)), again.join("\n")); +}); + +test("a normal fetch ignores the older walk's state, resumes the TIMELINE, and keeps the older state", async () => { + const slug = "older-walk"; + const channelRoot = path.join(channelsDir, slug); + const olderBefore = (await readPostFetchState(channelRoot))?.older; + assert.ok(olderBefore); + const result = await fetchPosts({ paths: getPaths(), slug, settings: LOGIN }); + assert.equal(result.ok, true); + const argv = (await argvLines()).at(-1)!; + assert.equal(argv[argv.length - 1], `https://x.com/${HANDLE}/timeline`); + assert.ok(argv.includes("extractor.twitter.cursor=1/TIMELINE-RESUME")); + assert.equal(argv.some((a) => a.includes("/search")), false); + const state = await readPostFetchState(channelRoot); + assert.deepEqual(state?.older, olderBefore); +}); + +test("a normal fetch with a half-walked older state still reads new posts from the top", async () => { + const slug = "older-half"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + await writePostFetchState(channelRoot, { + older: { since: "2019-01-01", until: "2019-04-01", maxId: "123", emptyWindows: 1 }, + }); + await fetchPosts({ paths: getPaths(), slug, settings: LOGIN }); + const argv = (await argvLines()).at(-1)!; + assert.equal(argv[argv.length - 1], `https://x.com/${HANDLE}/timeline`); + assert.equal(argv.some((a) => a.startsWith("extractor.twitter.cursor=")), false); + assert.deepEqual((await readPostFetchState(channelRoot))?.older, { + since: "2019-01-01", + until: "2019-04-01", + maxId: "123", + emptyWindows: 1, + }); +}); + +test("a floor date ends the walk at that date, and posts below it are not fetched", async () => { + const slug = "older-floor"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + floor: "2020-07-01", + olderWindowPauseMs: 0, + }); + assert.equal(result.ok, true); + assert.equal(result.complete, true); + assert.equal(result.written, 2); + const queries = (await argvLines()).map(queryOf); + assert.match(queries.at(-1)!, /since:2020-07-01 until:2020-09-11/); + const state = await readPostFetchState(channelRoot); + assert.equal(state?.older?.floor, "2020-07-01"); + assert.match(state?.older?.completeReason ?? "", /reached 2020-07-01/); + const dates = (await readAllPosts(channelRoot)).map((p) => p.createdAt.slice(0, 10)); + assert.equal(dates.includes("2020-06-01"), false); +}); + +test("the account's creation date, once a post carries it, is the floor", async () => { + const slug = "older-created"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + process.env.FAKE_ACCOUNT_DATE = "2019-10-05 10:00:00"; + try { + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + olderWindowPauseMs: 0, + }); + assert.equal(result.complete, true); + assert.equal(result.written, 4); + } finally { + delete process.env.FAKE_ACCOUNT_DATE; + } + const state = await readPostFetchState(channelRoot); + assert.equal(state?.older?.accountCreatedAt, "2019-10-05T10:00:00.000Z"); + assert.match(state?.older?.completeReason ?? "", /reached 2019-10-05/); + // It stopped at the window holding the creation date, not four windows on. + assert.match((await argvLines()).map(queryOf).at(-1)!, /since:2019-10-05 until:2019-12-11/); +}); + +test("a cancelled window keeps what it read and resumes INSIDE the window, below the oldest post read", async () => { + const slug = "older-cancel"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + const controller = new AbortController(); + process.env.FAKE_MODE = "stall"; + try { + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + olderWindowPauseMs: 0, + signal: controller.signal, + onLog: (l) => { + if (/rate limit/.test(l)) controller.abort(); + }, + }); + assert.equal(result.ok, true); + assert.equal(result.complete, false); + } finally { + delete process.env.FAKE_MODE; + } + // The stall printed the window's newest post: the archived one. + const state = await readPostFetchState(channelRoot); + assert.deepEqual( + { since: state?.older?.since, until: state?.older?.until, maxId: state?.older?.maxId }, + { since: "2020-12-11", until: "2021-03-11", maxId: idOf(POST_DATES[0]) }, + ); + assert.equal(state?.cursor, "1/TIMELINE-RESUME"); + + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + olderWindowPauseMs: 0, + }); + assert.equal(result.complete, true); + assert.equal(result.written, 4); + const resumed = (await argvLines()).map(queryOf).find((q) => q.includes("max_id:")); + assert.equal( + resumed, + `from:${HANDLE} since:2020-12-11 until:2021-03-11 include:nativeretweets max_id:${idOf(POST_DATES[0])}`, + ); +}); + +test("a limit stops the walk part-way with a resume point", async () => { + const slug = "older-limit"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + const result = await fetchPosts({ + paths: getPaths(), + slug, + settings: LOGIN, + older: true, + limit: 2, + olderWindowPauseMs: 0, + }); + assert.equal(result.complete, false); + assert.equal(result.written, 1); + const argv = (await argvLines()).at(-1)!; + assert.equal(argv[argv.indexOf("--range") + 1], "1-2"); + const older = (await readPostFetchState(channelRoot))?.older; + assert.equal(older?.maxId, idOf(POST_DATES[1])); + assert.equal(older?.complete, undefined); +}); + +test("with no login an older fetch needs cookies and spawns nothing", async () => { + const slug = "older-guest"; + const channelRoot = await makeChannel(slug); + await seed(channelRoot, slug); + const before = (await argvLines()).length; + const result = await fetchPosts({ paths: getPaths(), slug, settings: {}, older: true }); + assert.equal(result.ok, false); + assert.equal(result.needsCookies, true); + assert.match(result.error ?? "", /X search needs a logged-in session/); + assert.equal((await argvLines()).length, before); + const state = await readPostFetchState(channelRoot); + assert.equal(state?.needsCookies, true); + assert.equal(state?.cursor, "1/TIMELINE-RESUME"); + assert.equal(state?.older?.complete, undefined); +}); + +test("full and older together are refused, and so is a fetcher with no older walk", async () => { + const both = await fetchPosts({ + paths: getPaths(), + slug: "older-walk", + settings: LOGIN, + full: true, + older: true, + }); + assert.equal(both.ok, false); + assert.equal(both.error, FULL_AND_OLDER_REFUSAL); + + const badFloor = await fetchPosts({ + paths: getPaths(), + slug: "older-walk", + settings: LOGIN, + older: true, + floor: "2020-02-30", + }); + assert.equal(badFloor.ok, false); + assert.match(badFloor.error ?? "", /not a date/); + + assert.equal(olderPostsProblem({ label: "X", fetchOlder: async () => ({ posts: [], complete: true, position: { since: "", until: "", emptyWindows: 0 } }) }), null); + assert.equal(olderPostsProblem({ label: "Bluesky" }), "Bluesky cannot fetch older posts."); + assert.equal(olderPostsProblem(undefined), "This channel has no post fetcher."); +}); diff --git a/common/controller/fetchPosts.ts b/common/controller/fetchPosts.ts @@ -21,16 +21,25 @@ import { } from "../lib/cookiePolicy"; import { latestPostCreatedAt, + oldestPostCreatedAt, readPostFetchState, readSeenPostIds, writePostFetchState, writePosts, + type OlderBackfillPosition, + type OlderBackfillState, type PostFetchState, } from "../lib/posts-server"; import { handleFromAccountUrl, resolveSocialFetcher, + type SocialFetcher, } from "../social/fetchers"; +import { + firstOlderWindow, + isUtcDay, + olderFloorDay, +} from "../social/olderBackfill"; // Registering the built-in fetchers is a side effect of importing them. Keep // this list here (rather than inside fetchers.ts) so the registry module stays // free of imports from the heavier fetchers. @@ -42,6 +51,7 @@ import "../social/xGalleryDlFetcher"; // only ever used when a channel opts into it via postFetcher: "x-playwright". import "../social/xPlaywrightFetcher"; import "../social/xNitterFetcher"; +import type { XCookieSource } from "../social/xCookieSource"; import { resolveXCookieSourceFor, type XLoginSettings, @@ -57,11 +67,37 @@ export type FetchPostsOptions = { // posts are still deduped against the archive, so this is a safe repair // operation rather than a duplicate-maker. full?: boolean; + // Walk the account's history BACKWARDS from the oldest archived post, with + // the fetcher's `fetchOlder` (X: search windows), instead of fetching new + // posts. Its position is kept in posts-state.json's `older`, apart from the + // timeline cursor. Not with `full`. + older?: boolean; + // The older walk's floor date (YYYY-MM-DD): it stops there. Remembered for + // later runs of the walk. Default: none (it stops after a run of empty + // windows, or at the account's creation date). + floor?: string; + // The older walk's pause between two windows. Default: the fetcher's own. + olderWindowPauseMs?: number; limit?: number; onLog?: (line: string) => void; signal?: AbortSignal; }; +// The refusal for both walks at once, shared with the server action so the +// button, the ops route and this controller say the same thing. +export const FULL_AND_OLDER_REFUSAL = + "A full re-fetch and an older-posts fetch are different walks — run one at a time."; + +// Why this fetcher cannot fetch older posts, or null when it can. +export function olderPostsProblem( + fetcher: Pick<SocialFetcher, "label" | "fetchOlder"> | undefined, +): string | null { + if (!fetcher) return "This channel has no post fetcher."; + return typeof fetcher.fetchOlder === "function" + ? null + : `${fetcher.label} cannot fetch older posts.`; +} + export type FetchPostsResult = { ok: boolean; written: number; @@ -78,6 +114,19 @@ export async function fetchPosts( const log = (line: string) => onLog?.(line); const channelRoot = path.join(paths.channelsDir, slug); + if (opts.full && opts.older) { + return { ok: false, written: 0, skipped: 0, complete: false, error: FULL_AND_OLDER_REFUSAL }; + } + if (opts.floor !== undefined && !isUtcDay(opts.floor)) { + return { + ok: false, + written: 0, + skipped: 0, + complete: false, + error: `"${opts.floor}" is not a date (YYYY-MM-DD)`, + }; + } + const config = await readChannelConfig(paths, slug); if (!config) { return { ok: false, written: 0, skipped: 0, complete: false, error: `No such channel: ${slug}` }; @@ -118,6 +167,30 @@ export async function fetchPosts( const seenIds = await readSeenPostIds(channelRoot); const priorState = await readPostFetchState(channelRoot); + + if (opts.older) { + const problem = olderPostsProblem(fetcher); + if (problem) { + return { ok: false, written: 0, skipped: 0, complete: false, error: problem }; + } + const policy = resolveCookiePolicy(settings, config); + const xLogin = + fetcher.platform === "twitter" + ? await resolveXCookieSourceFor(paths, settings, policy.cookies) + : undefined; + return fetchOlderPosts({ + opts, + fetcher, + channelRoot, + handle, + accountUrl, + seenIds, + priorState, + cookies: alwaysCookies(policy), + cookieSource: xLogin?.source, + browserCookies: xLogin?.browserSpec, + }); + } // A stored cursor means the previous run stopped early (hit its limit, or was // cancelled). Resume the backfill from there rather than applying the // watermark, which would otherwise leave that history permanently missing. @@ -156,6 +229,9 @@ export async function fetchPosts( const state: PostFetchState = { lastFetchedAt: new Date().toISOString(), }; + // The older walk's position rides along untouched: a normal fetch neither + // reads it nor drops it. + if (priorState?.older) state.older = priorState.older; // Progress a long run saves as it goes (a fetcher that streams calls this // between pages): the posts so far, then the resume point they are safe @@ -258,3 +334,169 @@ export async function fetchPosts( error: result.error, }; } + +// The older-posts walk (`FetchPostsOptions.older`). Its own position in +// posts-state.json's `older`; every other field of the state is carried as it +// was, so the timeline cursor and watermark are exactly what the last normal +// fetch left. It does not stamp the channel's lastSyncedAt: it is not a sync, +// and the scheduler's next new-posts fetch stays where it was. +async function fetchOlderPosts(ctx: { + opts: FetchPostsOptions; + fetcher: SocialFetcher; + channelRoot: string; + handle: string; + accountUrl: string; + seenIds: ReadonlySet<string>; + priorState: PostFetchState | null; + cookies?: string; + cookieSource?: XCookieSource; + browserCookies?: string; +}): Promise<FetchPostsResult> { + const { opts, fetcher, channelRoot, handle, priorState } = ctx; + const { slug } = opts; + const log = (line: string) => opts.onLog?.(line); + const fetchOlder = fetcher.fetchOlder!; + const prior = priorState?.older; + + if (prior?.complete) { + log( + `Older posts for ${slug} are already complete` + + (prior.completedAt ? ` (since ${prior.completedAt})` : "") + + (prior.completeReason ? `: ${prior.completeReason}` : "") + + ". Nothing to walk.", + ); + return { ok: true, written: 0, skipped: 0, complete: true }; + } + + const floor = opts.floor ?? prior?.floor; + let accountCreatedAt = prior?.accountCreatedAt; + const startedAt = new Date().toISOString(); + let position: OlderBackfillPosition; + if (prior) { + position = { + since: prior.since, + until: prior.until, + emptyWindows: prior.emptyWindows ?? 0, + ...(prior.maxId ? { maxId: prior.maxId } : {}), + }; + } else { + const oldest = await oldestPostCreatedAt(channelRoot); + position = firstOlderWindow({ + oldestArchivedAt: oldest ?? undefined, + floor: olderFloorDay(floor, accountCreatedAt), + }); + } + + log( + `Fetching older posts for ${slug} via ${fetcher.label} (@${handle}): ` + + (prior ? "resuming at " : "starting at ") + + `${position.since} – ${position.until}` + + (floor ? `, down to ${floor}` : "") + + `; ${ctx.seenIds.size} already archived.`, + ); + + let written = 0; + let skipped = 0; + const shards = new Set<string>(); + const olderState = ( + at: OlderBackfillPosition, + extra: Partial<OlderBackfillState> = {}, + ): OlderBackfillState => ({ + since: at.since, + until: at.until, + emptyWindows: at.emptyWindows, + ...(at.maxId ? { maxId: at.maxId } : {}), + ...(floor ? { floor } : {}), + ...(accountCreatedAt ? { accountCreatedAt } : {}), + lastRunAt: startedAt, + lastWritten: written, + ...extra, + }); + // Everything but `older` is carried from the state as it was. + const save = (older: OlderBackfillState, needsCookies?: boolean) => + writePostFetchState(channelRoot, { + ...(priorState ?? {}), + ...(needsCookies ? { needsCookies: true } : {}), + older, + }); + + const onCheckpoint = async (cp: { + posts: Post[]; + position: OlderBackfillPosition; + accountCreatedAt?: string; + }) => { + const w = await writePosts(channelRoot, cp.posts); + written += w.written; + skipped += w.skipped; + for (const shard of w.shards) shards.add(shard); + accountCreatedAt ??= cp.accountCreatedAt; + await save(olderState(cp.position)); + if (w.written > 0) { + log(`Saved ${w.written} older post(s) (${written} so far this run); resume point recorded.`); + } + }; + + const controller = new AbortController(); + let result; + try { + result = await fetchOlder({ + accountUrl: ctx.accountUrl, + handle, + channelSlug: slug, + seenIds: ctx.seenIds, + cookies: ctx.cookies, + cookieSource: ctx.cookieSource, + browserCookies: ctx.browserCookies, + limit: opts.limit, + signal: opts.signal ?? controller.signal, + onLog: log, + position, + floor, + accountCreatedAt, + windowPauseMs: opts.olderWindowPauseMs, + onCheckpoint, + }); + } catch (err) { + const message = (err as Error).message; + log(`[error] ${message}`); + // Whatever was checkpointed is on disk with its position; keep that one. + const now = (await readPostFetchState(channelRoot))?.older; + await save({ ...(now ?? olderState(position)), lastError: message }); + return { ok: false, written, skipped, complete: false, error: message }; + } + + accountCreatedAt ??= result.accountCreatedAt; + const final = await writePosts(channelRoot, result.posts); + written += final.written; + skipped += final.skipped; + for (const shard of final.shards) shards.add(shard); + log( + `Wrote ${written} older post(s), skipped ${skipped} already archived` + + (shards.size > 0 ? ` (shards: ${[...shards].join(", ")})` : "") + + (result.complete ? " — older posts complete." : " — more history remains."), + ); + if (result.error) log(`[error] ${result.error}`); + await save( + olderState( + result.position, + result.complete + ? { + complete: true, + completedAt: new Date().toISOString(), + ...(result.completeReason ? { completeReason: result.completeReason } : {}), + } + : result.error + ? { lastError: result.error } + : {}, + ), + result.needsCookies, + ); + return { + ok: !result.error, + written, + skipped, + complete: result.complete, + needsCookies: result.needsCookies, + error: result.error, + }; +} diff --git a/common/lib/posts-server.ts b/common/lib/posts-server.ts @@ -211,6 +211,24 @@ export async function latestPostCreatedAt( return null; } +// The oldest archived post's createdAt: where the older-posts backfill starts. +// Only the oldest non-empty shard is read (listPostShards sorts oldest first). +export async function oldestPostCreatedAt( + channelRoot: string, +): Promise<string | null> { + const shards = await listPostShards(channelRoot); + for (const shard of shards) { + const posts = await readPostShard(channelRoot, shard); + if (posts.length === 0) continue; + let oldest = posts[0].createdAt; + for (const post of posts) { + if (post.createdAt < oldest) oldest = post.createdAt; + } + return oldest; + } + return null; +} + // --------------------------------------------------------------------------- // Availability sidecar ("is this post still up?") // --------------------------------------------------------------------------- @@ -310,7 +328,42 @@ export type PostFetchState = { // the post-corpus analogue of the video pipeline's needs_auth outcome. needsCookies?: boolean; lastFetchedCount?: number; + // The TIMELINE walk's resume point (gallery-dl's cursor). Only a normal fetch + // reads it; the older-posts backfill keeps its own position in `older`. cursor?: string; + // The older-posts backfill (a fetcher's `fetchOlder`), which walks the + // account's history backwards through search windows. Kept apart from + // `cursor` so a normal fetch never resumes backwards, and carried through + // every write a normal fetch makes. + older?: OlderBackfillState; +}; + +// Where the older-posts backfill stands. The window is in UTC days, as X's +// `since:`/`until:` search operators take them: posts created on or after +// `since` and before `until`. +export type OlderBackfillPosition = { + since: string; // YYYY-MM-DD + until: string; // YYYY-MM-DD + // Resume INSIDE the window: only posts with an id at or below this one (the + // oldest post the walk has read in this window). + maxId?: string; + // Windows in a row, ending with the last one walked, that held no posts. + emptyWindows: number; +}; + +export type OlderBackfillState = OlderBackfillPosition & { + // The date the walk stops at (YYYY-MM-DD), when one was given. + floor?: string; + // The account's creation time, once a fetched post has carried it: the walk + // never goes below it. + accountCreatedAt?: string; + complete?: boolean; + completedAt?: string; + // Why the walk ended, as a sentence ("reached the floor date 2020-01-01"). + completeReason?: string; + lastRunAt?: string; + lastWritten?: number; + lastError?: string; }; export const POST_FETCH_STATE_FILENAME = "posts-state.json"; diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts @@ -14,6 +14,7 @@ // in the client bundle. import type { Post, PostAvailability, PostPlatform } from "../lib/posts"; +import type { OlderBackfillPosition } from "../lib/posts-server"; import type { XCookieSource } from "./xCookieSource"; export type PostFetchInput = { @@ -80,6 +81,54 @@ export type PostFetchResult = { error?: string; }; +// Input for the older-posts backfill (`SocialFetcher.fetchOlder`): walk the +// account's history BACKWARDS from `position`, a window at a time, below what +// the timeline walk can reach. Login, limit, signal and log are the same as a +// normal fetch's; there is no watermark and no timeline cursor. +export type OlderPostFetchInput = Pick< + PostFetchInput, + | "accountUrl" + | "handle" + | "channelSlug" + | "seenIds" + | "cookies" + | "cookieSource" + | "browserCookies" + | "limit" + | "signal" + | "onLog" +> & { + // Where to start: the stored position of an unfinished walk, or the first + // window below the oldest archived post (olderBackfill.ts, firstOlderWindow). + position: OlderBackfillPosition; + // The date the walk stops at (YYYY-MM-DD), when one is set. + floor?: string; + // The account's creation time, when an earlier run learned it. + accountCreatedAt?: string; + // Pause between two windows' searches, so a run of quiet windows is not a + // burst of requests. Default: the fetcher's own. + windowPauseMs?: number; + // Save progress during the walk: posts not yet handed back, and the position + // they are safe under. Posts passed here are NOT returned again. + onCheckpoint?: (checkpoint: { + posts: Post[]; + position: OlderBackfillPosition; + accountCreatedAt?: string; + }) => Promise<void>; +}; + +export type OlderPostFetchResult = { + posts: Post[]; + // True when the walk has reached its end (`completeReason` says which). + complete: boolean; + completeReason?: string; + // Where the next run resumes, when the walk is not complete. + position: OlderBackfillPosition; + accountCreatedAt?: string; + needsCookies?: boolean; + error?: string; +}; + export type SocialFetcherProbe = { ok: boolean; // Display name for the account, used to autofill the channel form. @@ -132,6 +181,10 @@ export type SocialFetcher = { checkAvailability?( input: PostAvailabilityInput, ): Promise<Map<string, PostAvailability>>; + // Optional: walk the account's history backwards below what `fetch` can + // reach (X: search windows). A fetcher without it has no such backfill, and + // the channel page offers no "Fetch older posts" for it. + fetchOlder?(input: OlderPostFetchInput): Promise<OlderPostFetchResult>; }; // Populated by registerSocialFetcher() from each fetcher module. Indirection diff --git a/common/social/olderBackfill.test.ts b/common/social/olderBackfill.test.ts @@ -0,0 +1,116 @@ +// The older-posts backfill's window arithmetic (olderBackfill.ts). +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/olderBackfill.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { + addUtcDays, + addUtcMonths, + describeOlderBackfill, + firstOlderWindow, + isUtcDay, + olderFloorDay, + stepOlderWindow, +} from "./olderBackfill"; + +test("day arithmetic is UTC and clamps a month to its last day", () => { + assert.equal(addUtcDays("2020-12-31", 1), "2021-01-01"); + assert.equal(addUtcDays("2020-03-01", -1), "2020-02-29"); + assert.equal(addUtcMonths("2024-05-31", -3), "2024-02-29"); + assert.equal(addUtcMonths("2021-01-15", -3), "2020-10-15"); + assert.equal(isUtcDay("2020-02-29"), true); + assert.equal(isUtcDay("2021-02-29"), false); + assert.equal(isUtcDay("2021-2-1"), false); +}); + +test("the first window ends the day after the oldest archived post", () => { + assert.deepEqual(firstOlderWindow({ oldestArchivedAt: "2021-03-10T23:59:00.000Z" }), { + since: "2020-12-11", + until: "2021-03-11", + emptyWindows: 0, + }); + // Nothing archived: it ends tomorrow. + assert.equal( + firstOlderWindow({ now: new Date("2026-10-03T12:00:00Z") }).until, + "2026-10-04", + ); + // A floor inside the first window clamps its start. + assert.equal( + firstOlderWindow({ oldestArchivedAt: "2021-03-10T00:00:00Z", floor: "2021-01-01" }).since, + "2021-01-01", + ); +}); + +test("windows step back with no gap, count empty ones, and end after a run of them", () => { + let pos = firstOlderWindow({ oldestArchivedAt: "2021-03-10T00:00:00Z" }); + const seen: string[] = []; + const held = new Set(["2020-12-11"]); // only the first window holds posts + for (let i = 0; i < 10; i++) { + seen.push(`${pos.since}..${pos.until}`); + const step = stepOlderWindow(pos, { hadPosts: held.has(pos.since) }); + if (step.complete) { + assert.equal(step.emptyWindows, 4); + assert.match(step.reason, /4 windows of 3 months in a row held no posts, back to 2019-12-11/); + break; + } + assert.equal(step.next.until, pos.since); + assert.equal(step.next.maxId, undefined); + pos = step.next; + } + assert.deepEqual(seen, [ + "2020-12-11..2021-03-11", + "2020-09-11..2020-12-11", + "2020-06-11..2020-09-11", + "2020-03-11..2020-06-11", + "2019-12-11..2020-03-11", + ]); +}); + +test("a window with posts resets the empty count", () => { + const step = stepOlderWindow( + { since: "2020-01-01", until: "2020-04-01", emptyWindows: 3 }, + { hadPosts: true }, + ); + assert.equal(step.complete, false); + if (!step.complete) assert.equal(step.next.emptyWindows, 0); +}); + +test("the floor clamps the last window and ends the walk after it", () => { + const step = stepOlderWindow( + { since: "2020-04-01", until: "2020-07-01", emptyWindows: 0 }, + { hadPosts: true, floor: "2020-02-15" }, + ); + assert.equal(step.complete, false); + if (step.complete) return; + assert.deepEqual(step.next, { since: "2020-02-15", until: "2020-04-01", emptyWindows: 0 }); + const last = stepOlderWindow(step.next, { hadPosts: false, floor: "2020-02-15" }); + assert.equal(last.complete, true); + if (last.complete) assert.match(last.reason, /reached 2020-02-15/); +}); + +test("the floor is the later of the given date and the account's creation day", () => { + assert.equal(olderFloorDay(undefined, undefined), undefined); + assert.equal(olderFloorDay("2015-01-01", undefined), "2015-01-01"); + assert.equal(olderFloorDay(undefined, "2012-06-30T22:00:00.000Z"), "2012-06-30"); + assert.equal(olderFloorDay("2015-01-01", "2012-06-30T22:00:00.000Z"), "2015-01-01"); + assert.equal(olderFloorDay("2010-01-01", "2012-06-30T22:00:00.000Z"), "2012-06-30"); +}); + +test("the channel page's one line", () => { + assert.equal(describeOlderBackfill(undefined), "not started"); + assert.equal( + describeOlderBackfill({ since: "2019-01-01", until: "2019-04-01", emptyWindows: 0 }), + "walked back to 2019-04-01; next 2019-01-01 – 2019-04-01", + ); + assert.equal( + describeOlderBackfill({ + since: "2019-01-01", + until: "2019-04-01", + emptyWindows: 4, + complete: true, + completeReason: "reached 2019-01-01, the earliest date to walk to", + }), + "complete — reached 2019-01-01, the earliest date to walk to", + ); +}); diff --git a/common/social/olderBackfill.ts b/common/social/olderBackfill.ts @@ -0,0 +1,152 @@ +// The older-posts backfill's window arithmetic, pure (no I/O) so every step of +// the walk is testable without a fetcher. +// +// The walk goes BACKWARDS from the oldest archived post in fixed windows of +// UTC days — the unit X's `since:`/`until:` search operators take. Each window +// is one bounded search; the next window ends where the last one began, so the +// windows tile the history with no gap and no overlap. The walk ends at a floor +// date (given, or the account's creation date once a post has carried it), or +// after a run of windows in a row that held no posts. + +import type { + OlderBackfillPosition, + OlderBackfillState, +} from "../lib/posts-server"; + +// One window's span. Three months keeps one search bounded (a run that stops +// mid-window resumes inside it) without spending a request per quiet week. +export const OLDER_WINDOW_MONTHS = 3; + +// Windows in a row with no posts before the walk calls the history done: four +// three-month windows, a year with nothing. +export const OLDER_MAX_EMPTY_WINDOWS = 4; + +const DAY_RE = /^\d{4}-\d{2}-\d{2}$/; + +// A real YYYY-MM-DD calendar day (2021-02-30 is not one). +export function isUtcDay(value: string): boolean { + if (!DAY_RE.test(value)) return false; + const ms = Date.parse(`${value}T00:00:00Z`); + return Number.isFinite(ms) && new Date(ms).toISOString().slice(0, 10) === value; +} + +// The UTC day an instant falls on. +export function utcDayOf(instant: string | Date): string { + const d = typeof instant === "string" ? new Date(instant) : instant; + return d.toISOString().slice(0, 10); +} + +export function addUtcDays(day: string, days: number): string { + const [y, m, d] = day.split("-").map(Number); + return new Date(Date.UTC(y, m - 1, d + days)).toISOString().slice(0, 10); +} + +// Calendar months, clamped to the target month's last day (2024-05-31 minus +// three months is 2024-02-29, not March 2nd). +export function addUtcMonths(day: string, months: number): string { + const [y, m, d] = day.split("-").map(Number); + const target = new Date(Date.UTC(y, m - 1 + months, 1)); + const lastDay = new Date( + Date.UTC(target.getUTCFullYear(), target.getUTCMonth() + 1, 0), + ).getUTCDate(); + target.setUTCDate(Math.min(d, lastDay)); + return target.toISOString().slice(0, 10); +} + +const later = (a: string, b: string | undefined): string => + b !== undefined && b > a ? b : a; + +// Where the walk may not go below: the given floor date, or the day the +// account was created, whichever is later. +export function olderFloorDay( + floor: string | undefined, + accountCreatedAt: string | undefined, +): string | undefined { + const created = accountCreatedAt ? utcDayOf(accountCreatedAt) : undefined; + if (floor && created) return later(floor, created); + return floor ?? created; +} + +// The first window: it ends the day AFTER the oldest archived post (`until` is +// exclusive), so the rest of that day is read too — the posts already on disk +// are skipped as duplicates. With nothing archived it ends tomorrow. +export function firstOlderWindow(opts: { + oldestArchivedAt?: string; + now?: Date; + months?: number; + floor?: string; +}): OlderBackfillPosition { + const months = opts.months ?? OLDER_WINDOW_MONTHS; + const until = addUtcDays( + utcDayOf(opts.oldestArchivedAt ?? opts.now ?? new Date()), + 1, + ); + return { + since: later(addUtcMonths(until, -months), opts.floor), + until, + emptyWindows: 0, + }; +} + +// True when the window starts at or below the floor: it is the walk's last. +export function isAtFloor( + position: OlderBackfillPosition, + floor: string | undefined, +): boolean { + return floor !== undefined && position.since <= floor; +} + +export type OlderWindowStep = + | { complete: true; reason: string; emptyWindows: number } + | { complete: false; next: OlderBackfillPosition }; + +// What follows a window walked to its end. `hadPosts` counts any post the +// window held, archived already or not — a window full of known posts is not +// an empty one. +export function stepOlderWindow( + position: OlderBackfillPosition, + opts: { + hadPosts: boolean; + floor?: string; + months?: number; + maxEmptyWindows?: number; + }, +): OlderWindowStep { + const months = opts.months ?? OLDER_WINDOW_MONTHS; + const maxEmpty = opts.maxEmptyWindows ?? OLDER_MAX_EMPTY_WINDOWS; + const emptyWindows = opts.hadPosts ? 0 : position.emptyWindows + 1; + if (isAtFloor(position, opts.floor)) { + return { + complete: true, + reason: `reached ${opts.floor}, the earliest date to walk to`, + emptyWindows, + }; + } + if (emptyWindows >= maxEmpty) { + return { + complete: true, + reason: `${emptyWindows} windows of ${months} months in a row held no posts, back to ${position.since}`, + emptyWindows, + }; + } + const until = position.since; + return { + complete: false, + next: { + since: later(addUtcMonths(until, -months), opts.floor), + until, + emptyWindows, + }, + }; +} + +// One line for the channel page. +export function describeOlderBackfill( + state: OlderBackfillState | undefined, +): string { + if (!state) return "not started"; + if (state.complete) { + return `complete${state.completeReason ? ` — ${state.completeReason}` : ""}`; + } + return `walked back to ${state.until}; next ${state.since} – ${state.until}`; +} diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts @@ -209,3 +209,101 @@ test("stream: a gallery-dl that ignores output.jsonl (one pretty array at exit) test("restart-from-top cursor is the timeline extractor's first state", () => { assert.equal(RESTART_FROM_TOP_CURSOR, "1/"); }); + +// --------------------------------------------------------------------------- +// Search: the older-posts backfill's argv +// --------------------------------------------------------------------------- + +import { + buildGalleryDlSearchArgs, + xSearchHandle, + xSearchQuery, + xSearchUrl, +} from "./xGalleryDlFetcher"; + +// What gallery-dl's search extractor does with the URL's q: "+" to a space, +// THEN unquote (extractor/twitter.py, TwitterSearchExtractor.tweets). +function galleryDlQueryOf(url: string): string { + const m = /\/search\/?\?(?:[^&#]+&)*q=([^&#]+)/.exec(url); + assert.ok(m, url); + return decodeURIComponent(m[1].replace(/\+/g, " ")); +} + +test("search query: from/since/until, reposts included, max_id only when resuming", () => { + const window = { since: "2020-12-11", until: "2021-03-11" }; + assert.equal( + xSearchQuery("example_user", window), + "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets", + ); + assert.equal( + xSearchQuery("@example_user", { ...window, maxId: "1899000000000000001" }), + "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets max_id:1899000000000000001", + ); + assert.throws(() => xSearchQuery("example_user", { ...window, maxId: "12 OR x" }), /not a post id/); +}); + +test("search handle: anything search would read as more than a name is refused", () => { + assert.equal(xSearchHandle(" @Example_User1 "), "Example_User1"); + for (const bad of ["example user", "example:user", 'ex"ample', "example.user", "", "a-b"]) { + assert.throws(() => xSearchHandle(bad), /not an X handle/, bad); + } +}); + +test("search URL: the query survives gallery-dl's decoding exactly, spaces and colons included", () => { + const query = xSearchQuery("example_user", { + since: "2020-12-11", + until: "2021-03-11", + maxId: "1899000000000000001", + }); + const url = xSearchUrl(query); + assert.match(url, /^https:\/\/x\.com\/search\?q=from%3Aexample_user%20since%3A2020-12-11%20/); + assert.ok(url.endsWith("&f=live")); + assert.equal(url.includes(" "), false); + assert.equal(galleryDlQueryOf(url), query); + // An encoded "+" would stay a "+", not become a space. + assert.equal(galleryDlQueryOf(xSearchUrl("a+b c")), "a+b c"); +}); + +test("search argv: the timeline's flags and login, latest-first max_id paging, no cursor", () => { + const argv = buildGalleryDlSearchArgs({ + handle: "example_user", + position: { since: "2020-12-11", until: "2021-03-11" }, + cookieFile: JAR, + limit: 50, + }); + assert.equal(flagValue(argv, "--cookies"), JAR); + assert.equal(flagValue(argv, "--range"), "1-50"); + for (const opt of [ + "output.jsonl=true", + "extractor.twitter.text-tweets=true", + "extractor.twitter.retweets=true", + "extractor.twitter.replies=true", + "extractor.twitter.search-results=latest", + "extractor.twitter.search-pagination=max_id", + ]) { + assert.ok(argv.includes(opt), opt); + } + assert.equal(argv.some((a) => a.startsWith("extractor.twitter.cursor=")), false); + assert.equal(argv.includes("--no-download"), true); + assert.equal( + galleryDlQueryOf(argv[argv.length - 1]), + "from:example_user since:2020-12-11 until:2021-03-11 include:nativeretweets", + ); +}); + +test("stream: tracks the oldest post id read (any post, archived or not) and the account's creation time", () => { + const known = String(1899000000000000000n + 1n); + const s = newStream({ seenIds: new Set([`${known}`]), stopAtKnown: false, accountHandle: "FakeTester" }); + s.pushStdout(jsonl(5)); + s.pushStdout(jsonl(1)); // archived already: still the oldest read + s.pushStdout(jsonl(3)); + assert.equal(s.oldestId, known); + assert.equal(s.accountCreatedAt, undefined); + s.pushStdout( + JSON.stringify([2, { ...tweet(2), user: { name: "faketester", date: "2009-03-10 18:00:00" } }]) + .replace(/"tweet_id":"(\d+)"/, '"tweet_id":$1'), + ); + assert.equal(s.accountCreatedAt, "2009-03-10T18:00:00.000Z"); + assert.equal(s.takePending().length, 3); + assert.equal(s.pendingCount(), 0); +}); diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -29,11 +29,15 @@ import type { Readable } from "node:stream"; import { execa } from "execa"; import { getPaths } from "../lib/paths"; import type { Post } from "../lib/posts"; -import { normalizeXTweet, type XTweetRaw } from "./xNormalize"; +import type { OlderBackfillPosition } from "../lib/posts-server"; +import { normalizeXTweet, xCreatedAt, type XTweetRaw } from "./xNormalize"; +import { olderFloorDay, stepOlderWindow } from "./olderBackfill"; import { readXSessionStatus, xCookieFile } from "./xSessionBroker"; import type { XCookieSource } from "./xCookieSource"; import { registerSocialFetcher, + type OlderPostFetchInput, + type OlderPostFetchResult, type PostFetchInput, type PostFetchResult, type SocialFetcher, @@ -81,22 +85,39 @@ export function timelineUrlFor(accountUrl: string): string { } } -export function buildGalleryDlArgs(opts: { - accountUrl: string; +type GalleryDlLoginAndCap = { cookies?: string; // A cookies.txt exported by the Playwright session broker // (xSessionBroker.ts). PREFERRED over `cookies`: a live browser profile keeps // the session fresh, which is the fix for X cookies expiring in days. cookieFile?: string; limit?: number; - // gallery-dl's own resume point (`extractor.twitter.cursor`), as logged by a - // previous streamed run — see GalleryDlStream. - cursor?: string; - // A long fetch, read as it runs: log at debug level so gallery-dl prints its - // "Cursor: …" line after every page. Off for the probe, whose error message - // is the tail of stderr and must not be buried in debug noise. - stream?: boolean; -}): string[] { +}; + +export function buildGalleryDlArgs( + opts: GalleryDlLoginAndCap & { + accountUrl: string; + // gallery-dl's own resume point (`extractor.twitter.cursor`), as logged by + // a previous streamed run — see GalleryDlStream. + cursor?: string; + // A long fetch, read as it runs: log at debug level so gallery-dl prints + // its "Cursor: …" line after every page. Off for the probe, whose error + // message is the tail of stderr and must not be buried in debug noise. + stream?: boolean; + }, +): string[] { + const args = galleryDlCommonArgs(opts); + if (opts.cursor) { + args.push("-o", `extractor.twitter.cursor=${opts.cursor}`); + } + if (opts.stream) args.push("--verbose"); + args.push(timelineUrlFor(opts.accountUrl)); + return args; +} + +// The flags every gallery-dl read shares: metadata only, streamed, text tweets +// with their retweets and replies, the login, and the --range cap. +function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { const args = [ // One JSON object per item on stdout — we want metadata, never files. "--dump-json", @@ -141,14 +162,115 @@ export function buildGalleryDlArgs(opts: { if (opts.limit && opts.limit > 0) { args.push("--range", `1-${Math.floor(opts.limit)}`); } - if (opts.cursor) { - args.push("-o", `extractor.twitter.cursor=${opts.cursor}`); + return args; +} + +// --------------------------------------------------------------------------- +// Search: the older-posts backfill +// --------------------------------------------------------------------------- +// +// X's timeline only pages back so far; below that, an account's posts are +// reachable through search. gallery-dl's `twitter:search` extractor takes a +// search URL (`https://x.com/search?q=<query>`), reads the query's `from:` user +// for the records' `user`, and pages with `max_id:` (its default +// `search-pagination`): after each page it rewrites the query's `max_id:` to +// just below the oldest tweet it has read. That pagination logs no cursor, so +// this walk's resume point is its own: the oldest post id read in the window, +// passed back as `max_id:` — which gallery-dl's rewrite then replaces, rather +// than adding a second one. +// +// Search needs a logged-in session; a guest run is refused before spawning. + +// An X handle as search's `from:` takes it. Anything else (a space, a quote, a +// colon) would change the query's meaning rather than name an account. +const X_HANDLE_RE = /^[A-Za-z0-9_]{1,50}$/; + +export function xSearchHandle(handle: string): string { + const bare = handle.trim().replace(/^@/, ""); + if (!X_HANDLE_RE.test(bare)) { + throw new Error( + `"${handle}" is not an X handle search can take (letters, digits and "_")`, + ); } - if (opts.stream) args.push("--verbose"); - args.push(timelineUrlFor(opts.accountUrl)); + return bare; +} + +// The query for one window. `include:nativeretweets` keeps the account's +// reposts, as the timeline walk does (search leaves them out by default). +export function xSearchQuery( + handle: string, + position: Pick<OlderBackfillPosition, "since" | "until" | "maxId">, +): string { + const terms = [ + `from:${xSearchHandle(handle)}`, + `since:${position.since}`, + `until:${position.until}`, + "include:nativeretweets", + ]; + if (position.maxId) { + if (!/^\d+$/.test(position.maxId)) { + throw new Error(`"${position.maxId}" is not a post id`); + } + terms.push(`max_id:${position.maxId}`); + } + return terms.join(" "); +} + +// The search URL gallery-dl's search extractor matches. The query is +// percent-encoded whole (a space is %20, a colon %3A): gallery-dl turns "+" +// into a space BEFORE it unquotes, so an encoded "+" survives as itself. +// `f=live` is X's own "Latest" tab, which is also what gallery-dl asks for. +export function xSearchUrl(query: string): string { + return `https://x.com/search?q=${encodeURIComponent(query)}&f=live`; +} + +export function buildGalleryDlSearchArgs( + opts: GalleryDlLoginAndCap & { + handle: string; + position: Pick<OlderBackfillPosition, "since" | "until" | "maxId">; + }, +): string[] { + const args = galleryDlCommonArgs(opts); + args.push( + // Newest first, paged by max_id — the order the resume point relies on. + "-o", + "extractor.twitter.search-results=latest", + "-o", + "extractor.twitter.search-pagination=max_id", + ); + args.push(xSearchUrl(xSearchQuery(opts.handle, opts.position))); return args; } +// Inclusive: the post at the resume point is read again (and skipped as a +// duplicate) rather than risk an off-by-one past it. +function olderId(a: string | undefined, b: string): string { + if (a === undefined) return b; + try { + return BigInt(b) < BigInt(a) ? b : a; + } catch { + return a; + } +} + +// Pause between two windows' searches. Each window is its own gallery-dl run +// (a user lookup, then the search pages), so a stretch of empty windows would +// otherwise be a burst of requests a second apart. +export const OLDER_WINDOW_PAUSE_MS = 15_000; + +function pause(ms: number, signal: AbortSignal): Promise<void> { + if (ms <= 0 || signal.aborted) return Promise.resolve(); + return new Promise((resolve) => { + const t = setTimeout(done, ms); + function done() { + clearTimeout(t); + signal.removeEventListener("abort", done); + resolve(); + } + signal.addEventListener("abort", done, { once: true }); + }); +} + // The timeline extractor's cursor is `<state>[_<tweet id>]/<API cursor>`, and // "1/" is its first state with no API cursor: walk from the top. A run that // stopped before gallery-dl logged any cursor resumes from here rather than @@ -202,6 +324,11 @@ export class GalleryDlStream { rateLimited = false; stopReached = false; cursor?: string; + // The oldest post id read this run (any post, archived or not) — a search + // walk's resume point. + oldestId?: string; + // The account's creation time, from a record whose `user` is the account. + accountCreatedAt?: string; private previousCursor?: string; private savedCursor?: string; private pending: Post[] = []; @@ -216,6 +343,8 @@ export class GalleryDlStream { seenIds: ReadonlySet<string>; since?: string; stopAtKnown: boolean; + // The account's handle, to recognise its `user` record. + accountHandle?: string; onLog?: (line: string) => void; }, ) {} @@ -267,6 +396,14 @@ export class GalleryDlStream { return { posts, cursor }; } + // Every post not yet handed over, whatever the cursor — for a walk whose + // resume point comes from the records themselves (oldestId), not stderr. + takePending(): Post[] { + const posts = this.pending; + this.pending = []; + return posts; + } + // A checkpoint that could not be written goes back into the queue, so the // final write retries it. requeue(posts: Post[]): void { @@ -294,6 +431,18 @@ export class GalleryDlStream { if (!post || this.ids.has(post.id)) return; this.ids.add(post.id); this.distinct++; + if (/^\d+$/.test(post.id)) this.oldestId = olderId(this.oldestId, post.id); + if (!this.accountCreatedAt && this.opts.accountHandle) { + const user = rec.user as Record<string, unknown> | undefined; + if ( + user && + typeof user.name === "string" && + user.name.toLowerCase() === this.opts.accountHandle.toLowerCase() && + typeof user.date === "string" + ) { + this.accountCreatedAt = xCreatedAt({ date: user.date }); + } + } const isKnown = this.opts.seenIds.has(post.id); const isOlder = !isKnown && Boolean(this.opts.since && post.createdAt <= this.opts.since); @@ -416,6 +565,28 @@ function flattenRecords(items: unknown[]): XTweetRaw[] { return out; } +// The login source (galleryDlCookieChoice) for one run, logged: the operator's +// browser, or the session broker's exported jar — kept fresh by a live browser +// profile, so it survives the cookie expiry that otherwise breaks gallery-dl +// within days. One resolution for the timeline walk and the search walk. +async function galleryDlLoginFor( + input: Pick<PostFetchInput, "cookieSource" | "browserCookies" | "cookies" | "onLog">, +): Promise<{ cookies?: string; cookieFile?: string }> { + const paths = getPaths(); + const status = await readXSessionStatus(paths); + const choice = galleryDlCookieChoice({ + source: input.cookieSource, + browserCookies: input.browserCookies, + jarFile: + status.hasCookies && status.looksAuthenticated + ? xCookieFile(paths) + : undefined, + alwaysCookies: input.cookies, + }); + input.onLog?.(choice.note); + return { cookies: choice.cookies, cookieFile: choice.cookieFile }; +} + // Feed a pipe to `onLine` a line at a time; resolves once it has closed. function readLines( input: Readable | null | undefined, @@ -429,6 +600,232 @@ function readLines( }); } +const SEARCH_NEEDS_LOGIN = + "X search needs a logged-in session: connect an X account on /settings, or " + + "set cookiesFromBrowser for this channel (or globally), then run it again."; + +// The older-posts backfill: search windows walked backwards from `position` +// (olderBackfill.ts does the window arithmetic). One gallery-dl run per +// window, a pause between windows, the same per-run budget as a timeline +// backfill across all of them. Progress is saved mid-window (the oldest post +// read so far is the resume point) and at every window boundary. +export async function fetchOlderViaSearch( + input: OlderPostFetchInput, +): Promise<OlderPostFetchResult> { + const { channelSlug, seenIds, signal, onLog } = input; + const bin = getPaths().galleryDlBin; + let handle: string; + try { + handle = xSearchHandle(input.handle); + } catch (err) { + return { + posts: [], + complete: false, + position: input.position, + error: (err as Error).message, + }; + } + const { cookies, cookieFile } = await galleryDlLoginFor(input); + if (!cookies && !cookieFile) { + onLog?.(`[auth] ${SEARCH_NEEDS_LOGIN}`); + return { + posts: [], + complete: false, + position: input.position, + needsCookies: true, + error: SEARCH_NEEDS_LOGIN, + }; + } + + const deadline = Date.now() + BACKFILL_BUDGET_MS; + const pauseMs = input.windowPauseMs ?? OLDER_WINDOW_PAUSE_MS; + let position: OlderBackfillPosition = { ...input.position }; + let accountCreatedAt = input.accountCreatedAt; + // Posts not handed to a checkpoint, returned for the final write. + const carried: Post[] = []; + let checkpointsOn = Boolean(input.onCheckpoint); + let read = 0; + + const save = async (posts: Post[], at: OlderBackfillPosition) => { + if (!checkpointsOn || !input.onCheckpoint) { + carried.push(...posts); + return; + } + try { + await input.onCheckpoint({ posts, position: at, accountCreatedAt }); + } catch (err) { + // Keep the posts for the final write and stop checkpointing, so a saved + // position never runs ahead of what is on disk. + checkpointsOn = false; + carried.push(...posts); + onLog?.(`[warn] could not save progress mid-run: ${(err as Error).message}`); + } + }; + const stopped = ( + at: OlderBackfillPosition, + extra: Partial<OlderPostFetchResult> = {}, + ): OlderPostFetchResult => ({ + posts: carried, + complete: false, + position: at, + accountCreatedAt, + ...extra, + }); + + onLog?.( + `Searching @${handle}'s posts backwards, ${position.since} – ${position.until}` + + (position.maxId ? " from the saved resume point" : "") + + ` (up to ${Math.round(BACKFILL_BUDGET_MS / 60_000)} min this run; progress is saved as it goes)`, + ); + + for (;;) { + const floor = olderFloorDay(input.floor, accountCreatedAt); + if (floor !== undefined && position.until <= floor) { + return { + posts: carried, + complete: true, + completeReason: `the archive already reaches back to ${floor}, the earliest date to walk to`, + position, + accountCreatedAt, + }; + } + const remainingMs = deadline - Date.now(); + if (remainingMs <= 0) { + onLog?.(`Stopped at this run's ${Math.round(BACKFILL_BUDGET_MS / 60_000)}-minute budget; the next run resumes at ${position.since} – ${position.until}.`); + return stopped(position); + } + const remainingLimit = input.limit ? input.limit - read : undefined; + if (remainingLimit !== undefined && remainingLimit <= 0) { + return stopped(position); + } + + const windowStart = position; + const stream = new GalleryDlStream({ + channelSlug, + seenIds, + stopAtKnown: false, + accountHandle: handle, + onLog, + }); + const args = buildGalleryDlSearchArgs({ + handle, + position: windowStart, + cookies, + cookieFile, + limit: remainingLimit, + }); + const subprocess = execa(bin, args, { + reject: false, + buffer: false, + stdin: "ignore", + timeout: remainingMs, + cancelSignal: signal, + env: { PYTHONUNBUFFERED: "1" }, + }); + + // Mid-window checkpoints, one at a time and in order. The resume point is + // the oldest post read so far, which every post taken here is at or above. + let checkpoints: Promise<void> = Promise.resolve(); + let lastCheckpointAt = Date.now(); + const maybeCheckpoint = () => { + if (!checkpointsOn) return; + const due = + stream.pendingCount() >= CHECKPOINT_EVERY_POSTS || + (stream.pendingCount() > 0 && + Date.now() - lastCheckpointAt >= CHECKPOINT_EVERY_MS); + if (!due || !stream.oldestId) return; + lastCheckpointAt = Date.now(); + const posts = stream.takePending(); + const at = { ...windowStart, maxId: stream.oldestId }; + checkpoints = checkpoints.then(() => save(posts, at)); + }; + const stdoutDone = readLines(subprocess.stdout, (line) => { + stream.pushStdout(line); + maybeCheckpoint(); + }); + const stderrDone = readLines(subprocess.stderr, (line) => { + stream.pushStderr(line); + maybeCheckpoint(); + }); + const res = await subprocess; + await Promise.all([stdoutDone, stderrDone]); + await checkpoints; + + const spawnError = `${res.code ?? ""} ${res.message ?? ""}`; + if (res.failed && /ENOENT/.test(spawnError) && stream.distinct === 0) { + throw new Error( + `gallery-dl not found (looked for "${bin}"; set GALLERY_DL_BIN)`, + ); + } + + read += stream.distinct; + accountCreatedAt ??= stream.accountCreatedAt; + carried.push(...stream.finish()); + const here: OlderBackfillPosition = { + ...windowStart, + maxId: stream.oldestId ?? windowStart.maxId, + }; + if (stream.rateLimited) { + onLog?.("[rate-limit] X throttled this search — gallery-dl paused between pages."); + } + onLog?.( + `${windowStart.since} – ${windowStart.until}: read ${stream.distinct} post(s), ${stream.newPosts} new` + + (stream.known ? `, ${stream.known} already archived` : ""), + ); + + if (res.isCanceled || signal.aborted || res.timedOut) { + onLog?.( + res.timedOut + ? `Stopped at this run's ${Math.round(BACKFILL_BUDGET_MS / 60_000)}-minute budget; the next run resumes where this one left off.` + : "Cancelled; the next run resumes where this one left off.", + ); + return stopped(here); + } + if (res.exitCode !== 0) { + const tail = stream.stderrTail(); + if (looksLikeAuthFailure(tail)) { + onLog?.(`[auth] gallery-dl could not authenticate to X:\n${tail}`); + return stopped(here, { needsCookies: true, error: SEARCH_NEEDS_LOGIN }); + } + return stopped(here, { + error: `gallery-dl exited ${res.exitCode ?? res.signal ?? "abnormally"}: ${tail || "(no output)"}`, + }); + } + if (remainingLimit !== undefined && stream.distinct >= remainingLimit) { + return stopped(here); + } + + // The window is walked to its end. + const step = stepOlderWindow(windowStart, { + // A window resumed below a saved point held posts above it. + hadPosts: stream.distinct > 0 || Boolean(windowStart.maxId), + floor: olderFloorDay(input.floor, accountCreatedAt), + }); + if (step.complete) { + onLog?.(`Older-posts backfill complete: ${step.reason}.`); + return { + posts: carried, + complete: true, + completeReason: step.reason, + position: { since: windowStart.since, until: windowStart.until, emptyWindows: step.emptyWindows }, + accountCreatedAt, + }; + } + position = step.next; + // Every window boundary is saved: the posts so far, and the next window. + await save(carried.splice(0), position); + if (signal.aborted) { + onLog?.("Cancelled; the next run resumes where this one left off."); + return stopped(position); + } + await pause(pauseMs, signal); + if (signal.aborted) { + onLog?.("Cancelled; the next run resumes where this one left off."); + return stopped(position); + } + } +} + export const xGalleryDlFetcher: SocialFetcher = { id: "x-gallery-dl", label: "X / Twitter (gallery-dl)", @@ -498,23 +895,8 @@ export const xGalleryDlFetcher: SocialFetcher = { async fetch(input: PostFetchInput): Promise<PostFetchResult> { const { accountUrl, channelSlug, since, seenIds, limit, signal, onLog } = input; - const paths = getPaths(); - const bin = paths.galleryDlBin; - // The login source (galleryDlCookieChoice): the operator's browser, or the - // session broker's exported jar — kept fresh by a live browser profile, so - // it survives the cookie expiry that otherwise breaks gallery-dl within days. - const status = await readXSessionStatus(paths); - const choice = galleryDlCookieChoice({ - source: input.cookieSource, - browserCookies: input.browserCookies, - jarFile: - status.hasCookies && status.looksAuthenticated - ? xCookieFile(paths) - : undefined, - alwaysCookies: input.cookies, - }); - onLog?.(choice.note); - const { cookies, cookieFile } = choice; + const bin = getPaths().galleryDlBin; + const { cookies, cookieFile } = await galleryDlLoginFor(input); const args = buildGalleryDlArgs({ accountUrl, cookies, @@ -673,6 +1055,8 @@ export const xGalleryDlFetcher: SocialFetcher = { ? { posts, complete: false, cursor: resumeAt } : { posts, complete: true }; }, + + fetchOlder: fetchOlderViaSearch, }; registerSocialFetcher(xGalleryDlFetcher); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -2,6 +2,7 @@ ## [Unreleased] - **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched. +- **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history. - **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete. - **`pnpm ops retry-bucket` can run part of a bucket.** Its body takes `"ids"`, a list of video ids, and runs only those, as ticking them on the bucket's card does. Every id must be in the named bucket: one that is not is refused with a 400 naming it, and nothing runs. A job started with `ids` is not replayable, like a checkbox selection in the UI. - **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen. diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts @@ -1,6 +1,6 @@ import test from "node:test"; import assert from "node:assert/strict"; -import { OpsInputError, optSubset } from "./_lib"; +import { OpsInputError, optPositiveInt, optSubset } from "./_lib"; // Run with: // pnpm -C editor exec tsx --test "app/**/*.test.ts" @@ -36,3 +36,19 @@ test("optSubset: an empty or non-string list is refused, never read as 'all'", ( ); } }); + +test("optPositiveInt: absent is undefined; a whole number above zero comes back", () => { + assert.equal(optPositiveInt({}, "limit"), undefined); + assert.equal(optPositiveInt({ limit: 200 }, "limit"), 200); +}); + +test("optPositiveInt: zero, negatives, fractions and strings are refused", () => { + for (const limit of [0, -5, 2.5, "10", null, Number.NaN]) { + assert.throws( + () => optPositiveInt({ limit }, "limit"), + (e: unknown) => + e instanceof OpsInputError && /"limit" must be a whole number above zero/.test(e.message), + JSON.stringify(limit), + ); + } +}); diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts @@ -146,6 +146,17 @@ export function optBool(body: OpsBody, key: string): boolean | undefined { return v; } +// A whole number above zero, or undefined when absent. A type check, not a +// rule: what the number means is the action's business. +export function optPositiveInt(body: OpsBody, key: string): number | undefined { + const v = body[key]; + if (v === undefined) return undefined; + if (typeof v !== "number" || !Number.isInteger(v) || v <= 0) { + throw new OpsInputError(`"${key}" must be a whole number above zero`); + } + return v; +} + export function reqStringArray(body: OpsBody, key: string): string[] { const v = body[key]; if ( diff --git a/editor/app/api/ops/fetch-posts/route.ts b/editor/app/api/ops/fetch-posts/route.ts @@ -0,0 +1,41 @@ +import { fetchPostsAction } from "../../../channels/[slug]/socialActions"; +import { + jobResponse, + ops, + optBool, + optPositiveInt, + optString, + reqSlug, +} from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, full?, older?, floor?, limit?, queueKey? } -> { ok: true, jobId } +// +// A social channel's "Fetch posts" button, over HTTP — and its "Re-fetch full +// history" (`full`) and "Fetch older posts" (`older`, with an optional `floor` +// date, YYYY-MM-DD, that the walk stops at). The job runs on the platform's +// queue, as the buttons' does. +// +// Every refusal is the action's own sentence: a channel that is not a social +// one, `full` and `older` together, a fetcher with no older walk, a floor +// without `older` or that is not a date. +export async function POST(request: Request) { + return ops( + request, + ["slug", "full", "older", "floor", "limit", "queueKey"], + async (body) => { + const slug = reqSlug(body, "slug"); + return jobResponse( + await fetchPostsAction( + slug, + optString(body, "queueKey"), + optBool(body, "full"), + optPositiveInt(body, "limit"), + optBool(body, "older"), + optString(body, "floor"), + ), + ); + }, + ); +} diff --git a/editor/app/channels/[slug]/components/SocialChannelPanel.tsx b/editor/app/channels/[slug]/components/SocialChannelPanel.tsx @@ -30,6 +30,10 @@ export type SocialChannelState = { deletedCount: number; checkedCount: number; canCheckAvailability: boolean; + // The older-posts backfill (a fetcher's `fetchOlder`): whether this fetcher + // has one, and where it stands, as one line. + canFetchOlder: boolean; + olderStatus: string; }; export function SocialChannelPanel({ @@ -63,9 +67,15 @@ export function SocialChannelPanel({ if (res.ok) void res.stream.cancel(); }); - const run = (full: boolean) => + const run = (full: boolean, older = false) => startTransition(async () => { - const res = await fetchPostsAction(slug, undefined, full); + const res = await fetchPostsAction( + slug, + undefined, + full, + undefined, + older || undefined, + ); // The job streams to the jobs page; nothing to consume here, but the // stream must be released or its buffered chunks leak. if (res.ok) void res.stream.cancel(); @@ -111,6 +121,9 @@ export function SocialChannelPanel({ : String(state.lastFetchedCount) } /> + {state.canFetchOlder && ( + <Stat label="Older posts" value={state.olderStatus} /> + )} </dl> {state.availableFetchers.length > 1 && ( @@ -181,6 +194,18 @@ export function SocialChannelPanel({ > Re-fetch full history </button> + {state.canFetchOlder && ( + <button + type="button" + onClick={() => run(false, true)} + disabled={pending} + aria-label="fetch older posts" + title="Walk back from the oldest archived post through search, a few months at a time, for posts older than the timeline reaches. Needs a login. Progress is saved as it goes; a later run continues it, and it says so once the history is done." + className="rounded-md border border-border px-3 py-1 text-sm hover:bg-muted disabled:opacity-50" + > + Fetch older posts + </button> + )} </div> {/* The two failure surfaces a social channel actually has. */} diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -11,6 +11,7 @@ import { } from "yt-dlp-transcript-common/lib/posts-server"; import { isPostGone } from "yt-dlp-transcript-common/lib/posts"; import { resolveSocialFetcher } from "yt-dlp-transcript-common/social/fetchers"; +import { describeOlderBackfill } from "yt-dlp-transcript-common/social/olderBackfill"; import "yt-dlp-transcript-common/social/blueskyFetcher"; import "yt-dlp-transcript-common/social/xGalleryDlFetcher"; import "yt-dlp-transcript-common/social/xPlaywrightFetcher"; @@ -186,6 +187,8 @@ export default async function ChannelDetailPage({ checkedCount: availabilityRecords.length, canCheckAvailability: typeof fetcher?.checkAvailability === "function", + canFetchOlder: typeof fetcher?.fetchOlder === "function", + olderStatus: describeOlderBackfill(fetchState?.older), handle: config.socialHandle ?? slug, accountUrl: config.url, }} diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts @@ -19,7 +19,12 @@ import { readChannelConfig, } from "yt-dlp-transcript-common/controller/channels"; import { isSocialChannel } from "yt-dlp-transcript-common/lib/channelConfig"; -import { fetchPosts } from "yt-dlp-transcript-common/controller/fetchPosts"; +import { + fetchPosts, + FULL_AND_OLDER_REFUSAL, + olderPostsProblem, +} from "yt-dlp-transcript-common/controller/fetchPosts"; +import { isUtcDay } from "yt-dlp-transcript-common/social/olderBackfill"; import { checkPostAvailability, type CheckPostAvailabilityMode, @@ -27,6 +32,7 @@ import { import { getSocialFetcher, listSocialFetchers, + resolveSocialFetcher, } from "yt-dlp-transcript-common/social/fetchers"; import { runManagedFunction, @@ -121,18 +127,39 @@ export async function checkPostAvailabilityAction( }); } +// `older` walks the account's history backwards from the oldest archived post +// (the fetcher's `fetchOlder` — X: search windows) instead of fetching new +// posts; `floor` (YYYY-MM-DD) is the date that walk stops at. Both are refused +// HERE, before a job exists, when they cannot run: with `full`, on a fetcher +// that has no older walk, or with a floor that is not a date. export async function fetchPostsAction( slug: string, queueKey?: string, full?: boolean, limit?: number, + older?: boolean, + floor?: string, ): Promise<StreamActionResult> { + if (full && older) return { ok: false, error: FULL_AND_OLDER_REFUSAL }; + if (floor !== undefined && !older) { + return { ok: false, error: "A floor date applies only to an older-posts fetch." }; + } + if (floor !== undefined && !isUtcDay(floor)) { + return { ok: false, error: `"${floor}" is not a date (YYYY-MM-DD).` }; + } const paths = getPaths(); const config = await readChannelConfig(paths, slug); if (!config) return { ok: false, error: `No such channel: ${slug}` }; if (!isSocialChannel(config)) { return { ok: false, error: `${slug} is not a social channel.` }; } + if (older) { + await registerBuiltinSocialFetchers(); + const problem = olderPostsProblem( + resolveSocialFetcher(config.postFetcher, config.url), + ); + if (problem) return { ok: false, error: problem }; + } // queueKeyForUrl() already routes x.com / bsky.app to platform:x.com / // platform:bsky.app, so per-platform serialization and the existing 429 @@ -144,13 +171,19 @@ export async function fetchPostsAction( queueKey: key, paths, channelSlug: slug, - spec: { kind: "fetch-posts", slug, params: { queueKey, full, limit } }, + spec: { + kind: "fetch-posts", + slug, + params: { queueKey, full, limit, older, floor }, + }, fn: async (onLog, signal) => { const result = await fetchPosts({ paths, slug, settings: getSettings(), full, + older, + floor, limit, onLog, signal, diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -246,7 +246,14 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { }, "fetch-posts": (spec) => { const { p, queueKey } = params(spec); - return fetchPostsAction(spec.slug, queueKey, bool(p.full), num(p.limit)); + return fetchPostsAction( + spec.slug, + queueKey, + bool(p.full), + num(p.limit), + bool(p.older), + str(p.floor), + ); }, "download-missing-subs": (spec) => { const { p, queueKey } = params(spec); diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -192,6 +192,7 @@ test("a traversing slug is refused at the door, on every route that takes one", ["channel-priority", { slugs: ["../../escape"], tier: "paused" }], ["relocate", { slugs: ["../../escape"], root: "/tmp/ops-api-never" }], ["relocate-back", { slugs: ["../../escape"] }], + ["fetch-posts", { slug: "../../escape", older: true }], ]; for (const [action, data] of cases) { const { status, body } = await ops(request, action, data); @@ -267,6 +268,73 @@ test("retry-bucket and transcribe-bucket ids must be in the bucket: a stray is r expect(await listJobIds()).toEqual(before); }); +test("fetch-posts refuses a channel that is not social, full with older, and an older fetch its fetcher cannot do — before any job", async ({ + request, +}) => { + await resetData("title-filter-channel"); + await settings(); + // A social channel whose fetcher has no older-posts walk. Nothing below + // reaches a fetch: every case is refused before a job exists. + await writeChannelConfig("example-bsky", { + handling: "transcribe", + sourceKind: "social", + platform: "bluesky", + postFetcher: "bluesky-atproto", + socialHandle: "example.bsky.social", + name: "Example (Bluesky)", + url: "https://bsky.app/profile/example.bsky.social", + }); + const before = await listJobIds(); + + // The sentence the action gives for a video channel. + const video = await ops(request, "fetch-posts", { slug: "test-filter" }); + expect(video.status).toBe(400); + expect(video.body.error).toBe("test-filter is not a social channel."); + const videoOlder = await ops(request, "fetch-posts", { slug: "test-filter", older: true }); + expect(videoOlder.status).toBe(400); + expect(videoOlder.body.error).toBe("test-filter is not a social channel."); + + // Two walks at once, on any channel. + const both = await ops(request, "fetch-posts", { + slug: "example-bsky", + full: true, + older: true, + }); + expect(both.status).toBe(400); + expect(both.body.error).toMatch(/different walks — run one at a time/); + const bothVideo = await ops(request, "fetch-posts", { + slug: "test-filter", + full: true, + older: true, + }); + expect(bothVideo.status).toBe(400); + expect(bothVideo.body.error).toMatch(/different walks/); + + // A fetcher with no older walk, a floor that is not a date, a floor + // without older, and a limit that is not a count. + const bsky = await ops(request, "fetch-posts", { slug: "example-bsky", older: true }); + expect(bsky.status).toBe(400); + expect(bsky.body.error).toMatch(/cannot fetch older posts/); + const floor = await ops(request, "fetch-posts", { + slug: "example-bsky", + older: true, + floor: "2020-13-01", + }); + expect(floor.status).toBe(400); + expect(floor.body.error).toMatch(/is not a date/); + const floorAlone = await ops(request, "fetch-posts", { + slug: "example-bsky", + floor: "2020-01-01", + }); + expect(floorAlone.status).toBe(400); + expect(floorAlone.body.error).toMatch(/only to an older-posts fetch/); + const limit = await ops(request, "fetch-posts", { slug: "example-bsky", limit: 0 }); + expect(limit.status).toBe(400); + expect(limit.body.error).toMatch(/"limit" must be a whole number above zero/); + + expect(await listJobIds()).toEqual(before); +}); + test("channel-config round-trips a download filter and refuses a bad regex", async ({ page, request, diff --git a/plans/FACTS.md b/plans/FACTS.md @@ -8534,6 +8534,26 @@ phase deletes from the destination. `expiry` is seconds (read as ms when too large for seconds); `lastAccessed` is PRTime (µs). The Chromium family is NOT read here (its values are encrypted with a keyring key): the readers say so and gallery-dl reads it. +- **gallery-dl 1.32.9's `twitter:search`** (`extractor/twitter.py`, `TwitterSearchExtractor`) + matches `…/search?…q=<query>` (q may follow other params; `&f=live` after it is fine) and decodes q + as `unquote(q.replace("+", " "))` — so the query is percent-encoded whole and an encoded `+` + survives. It reads the query's single `from:<user>` to fill each record's `user` (with the + account's `date`, its creation time). Product defaults to "Latest"; `search-pagination` defaults + to `max_id`: after each page it REWRITES the query's `max_id:` (adding one if absent) to just below + the oldest tweet read and sets the API cursor to None — so **a search run logs no + `Cursor:` line and cannot be resumed with `extractor.twitter.cursor`**. It stops after + `search-stop` (default 3) empty pages. Quoted tweets are skipped unless `quoted` is set. +- **The older-posts backfill** (`fetchPosts({ older })` → `SocialFetcher.fetchOlder`; for X + `fetchOlderViaSearch` in `xGalleryDlFetcher.ts`, window arithmetic in `social/olderBackfill.ts`). + Windows of 3 months in UTC days, `from:<h> since:<d> until:<d> include:nativeretweets`, the first + ending the day after the oldest archived post; one gallery-dl run per window, 15 s apart, one + 3-hour budget across them. Its resume point is its own: the oldest post id read in the window, + sent back as `max_id:<id>` (inclusive; gallery-dl's rewrite replaces it). State is + `posts-state.json` `older` (`OlderBackfillState`), apart from the timeline `cursor`; a normal fetch + copies `older` through untouched and an older run copies every other field through. Ends at the + later of the `floor` and the account's creation day, or after 4 empty windows in a row; then + `older.complete` and a re-run fetches nothing. No login = refused before spawning + (`needsCookies`). It does not stamp `lastSyncedAt`. - **`node:sqlite`** (Node 22.23 here; unflagged since 22.13, an ExperimentalWarning once per process) is loaded at run time through an assembled specifier with `webpackIgnore` (`common/social/nodeSqlite.ts`), like `playwrightRuntime.ts`: the editor's Turbopack build has diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -100,6 +100,9 @@ const ACTIONS = [ // The TRANSCRIBE half of a channel's "downloaded, not transcribed" bucket // (retry-bucket is the download retry and skips audio already on disk). "transcribe-bucket", + // A social channel's post fetch: new posts, the full re-walk ("full"), or + // the walk back below the oldest archived post ("older"). + "fetch-posts", "build-index", "build-deploy", "build-site", @@ -317,6 +320,12 @@ export function usage() { ' bucket on the transcription queue, as the channel page\'s Transcribe button', ' does: {"slug"}. "ids": [...] narrows it the same way as on retry-bucket.', "", + 'fetch-posts fetches a social channel\'s new posts: {"slug"}. "full": true', + ' re-walks the whole timeline; "older": true walks back from the oldest', + " archived post through search (X; needs a login), saving its place for", + ' the next run, down to "floor": "YYYY-MM-DD" when given. "limit": N caps', + ' the posts one run reads. "full" and "older" together are refused.', + "", '"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare', " Pages PREVIEW instead of production: the same bundle goes to a branch", " alias, https://<branch>.<project>.pages.dev, and the live site is left", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -378,3 +378,15 @@ test("build-homepage and deploy-homepage are POSTs to their own routes, named in assert.match(usage(), /build-homepage builds the homepage package into homepage\/out/); assert.match(usage(), /project archilyzer \(https:\/\/archilyzer\.pages\.dev\)/); }); + +// A social channel's post fetch: a POST to its route, the body passed through +// untouched — the route and the action judge it ("full" with "older" included). +test("fetch-posts is a POST to its route, named in the usage", () => { + const p = parseArgs(["fetch-posts", "--json", '{"slug":"example-x","older":true}', "--wait"]); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/fetch-posts"); + assert.deepEqual(p.body, { slug: "example-x", older: true }); + assert.equal(p.wait, true); + assert.match(usage(), /Actions:.*transcribe-bucket, fetch-posts/); + assert.match(usage(), /"older": true walks back from the oldest/); +});