// First-class registry of social-post fetchers, modeled on // common/lib/transcriptionApps.ts: each fetcher owns how it detects a URL it // can handle, what config fields it surfaces in the editor form, how it probes // an account cheaply, and how it pages a timeline into normalized `Post`s. // // The registry exists because every X retrieval path rots — cookies expire in // days and X rotates its GraphQL query IDs every few weeks — so the SEAM // matters more than any single implementation. Bluesky, by contrast, is a free // and complete public API and needs no external binary at all. // // Server-side module (fetchers may spawn binaries / read paths). The editor // form consumes `listSocialFetchers()` descriptors instead, exactly as the // settings form consumes listTranscriptionApps() — so this module never lands // in the client bundle. import type { Post, PostAvailability, PostPlatform } from "../lib/posts"; import type { OlderBackfillPosition } from "../lib/posts-server"; import type { XCookieSource } from "./xCookieSource"; import type { PostCaptureState } from "./postCapture"; export type PostFetchInput = { // The account's canonical URL as configured on the channel. accountUrl: string; // The bare handle, without a leading "@" or URL wrapper. handle: string; // The channel slug the produced posts belong to. channelSlug: string; // ISO watermark: stop once posts older than this are reached. Absent on a // first (full-backfill) run AND whenever a previous run stopped early — see // `cursor`, which resumes the backfill instead. since?: string; // Opaque resume point from a previous run that did not finish (hit its // limit, or was cancelled). When set, the fetcher continues paging from here // rather than from the top of the timeline, so a capped first run does not // leave a permanent hole in the archived history. cursor?: string; // Post ids already on disk. Paging stops at the first page containing one, // mirroring how sync() stops at the first already-archived video. seenIds: ReadonlySet; // Resolved cookie spec from resolveCookiePolicy(), when the fetcher needs // credentials. Bluesky ignores it entirely. cookies?: string; // WHERE AN X FETCHER'S LOGIN COMES FROM this run — `social.x.cookieSource`, // resolved by the fetch controller (xCookieSource.ts): "browser" reads the // operator's everyday browser named by `browserCookies`; "profile" uses the // session broker's connected profile. Unset (a caller that does not resolve // it): the profile's jar when it holds a login, else `cookies`, as before the // choice existed. Platforms without a login ignore both. cookieSource?: XCookieSource; // The browser spec the "browser" source reads: the channel's // cookiesFromBrowser over the global one, REGARDLESS of cookieMode — the // source, not yt-dlp's mode, governs the X fetchers. browserCookies?: string; // Soft cap on how many posts to return in one run. Undefined = no cap // beyond the watermark/seen-id stop conditions. limit?: number; // Cap on how many PAGES one run reads, for a fetcher that walks pages (a // forum thread: "the latest N pages"). Ignored by the others. pages?: number; // The pause between two page loads, for a paced page walker (a forum // thread). Default: the fetcher's own. pagePauseMs?: number; // Stop once the walk reaches posts that are already archived. False for a // --full repair, which re-walks everything on purpose. Undefined: the // fetcher's own default. stopAtKnown?: boolean; // Save progress during a long run: called with posts not yet handed back and // the resume point they are safe under. Posts passed here are NOT returned // again in the result. A fetcher that pages in one shot never calls it. onCheckpoint?: (checkpoint: { posts: Post[]; cursor: string }) => Promise; signal: AbortSignal; // A soft stop (the job's Drain): the run ends at its next resume point — // between two pages — keeping that point, rather than walking on. `signal` // cancels outright. A fetcher that pages in one shot may ignore it. drain?: AbortSignal; // Progress/diagnostic sink, wired to the job log. onLog?: (line: string) => void; }; export type PostFetchResult = { posts: Post[]; // False when the run stopped early (limit hit, budget or cancel, or an error // part-way) and more history remains behind `cursor`. complete: boolean; cursor?: string; // Set when the fetcher stopped because credentials are missing or expired. // Maps onto the existing "needs cookies" snapshot bucket rather than // inventing a new failure surface. needsCookies?: boolean; // The run failed part-way. `posts` and `cursor` still hold what it got, so // the controller saves them before reporting the failure. error?: string; // The run stopped on `drain` at a resume point (`cursor`). Not a failure. drained?: boolean; }; // Input for the older-posts backfill (`SocialFetcher.fetchOlder`): walk the // account's history BACKWARDS from `position`, a window at a time, below what // the timeline walk can reach. Login, limit, signal and log are the same as a // normal fetch's; there is no watermark and no timeline cursor. export type OlderPostFetchInput = Pick< PostFetchInput, | "accountUrl" | "handle" | "channelSlug" | "seenIds" | "cookies" | "cookieSource" | "browserCookies" | "limit" | "signal" | "drain" | "onLog" > & { // Where to start: the stored position of an unfinished walk, or the first // window below the oldest archived post (olderBackfill.ts, firstOlderWindow). position: OlderBackfillPosition; // The date the walk stops at (YYYY-MM-DD), when one is set. floor?: string; // The account's creation time, when an earlier run learned it. accountCreatedAt?: string; // Pause between two windows' searches, so a run of quiet windows is not a // burst of requests. Default: the fetcher's own. windowPauseMs?: number; // Save progress during the walk: posts not yet handed back, and the position // they are safe under. Posts passed here are NOT returned again. onCheckpoint?: (checkpoint: { posts: Post[]; position: OlderBackfillPosition; accountCreatedAt?: string; }) => Promise; }; export type OlderPostFetchResult = { posts: Post[]; // True when the walk has reached its end (`completeReason` says which). complete: boolean; completeReason?: string; // Where the next run resumes, when the walk is not complete. position: OlderBackfillPosition; accountCreatedAt?: string; needsCookies?: boolean; error?: string; // The walk stopped on `drain`; `position` is where the next run resumes. drained?: boolean; }; export type SocialFetcherProbe = { ok: boolean; // Display name for the account, used to autofill the channel form. name?: string; handle?: string; // Canonical account URL, normalized. url?: string; avatar?: string; description?: string; postCount?: number; error?: string; }; // Which config fields this fetcher surfaces in the channel form. Mirrors // TranscriptionApp.fields. export type SocialFetcherFields = { // Needs a browser cookie spec (cookiesFromBrowser / cookieMode; for X, the // login source `social.x.cookieSource`). cookies?: boolean; // Needs an external binary whose path is configurable. binPath?: boolean; // Supports capping how many posts one run fetches. limit?: boolean; }; // Input for a deleted-post sweep. Ids are the platform-native post ids, and // `author` is the account handle they belong to (some platforms need it to // build a lookup key). export type PostAvailabilityInput = { ids: ReadonlyArray; handle: string; signal: AbortSignal; onLog?: (line: string) => void; }; // Input for a post capture (`SocialFetcher.captureByIds`): a screenshot of each // post as the platform renders it, and its attached media, for specific // archived posts. The login fields are a normal fetch's; `outDir` is the // channel's `posts-media/`, one directory per post id beneath it // (postCapture.ts owns the layout). export type PostCaptureInput = Pick< PostFetchInput, "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog" | "pagePauseMs" > & { // The channel's URL (a forum capture needs the forum's origin). accountUrl?: string; ids: ReadonlyArray; handle: string; outDir: string; // Take the screenshot / download the media. Both default to true. shots?: boolean; media?: boolean; // Capture again what is already captured. force?: boolean; // Also capture the long-form article a post links to (X Articles), where it // links to one. Defaults to true. articles?: boolean; // What the posts archive holds for each id — its text and expanded links — // for the links a capture follows (an X Article's). Ids absent are read // from the page alone. archived?: ReadonlyMap< string, { text: string; links?: ReadonlyArray; // The post's own URL and media, where the archive has them (forum // posts: the capture opens the URL and downloads the media). url?: string; media?: ReadonlyArray<{ kind: string; url: string; name?: string }>; } >; // A soft stop: no new post is started once it fires, and the one in hand // finishes (`signal` cancels outright). drain?: AbortSignal; }; export type PostCaptureOutcome = { id: string; state: PostCaptureState; // What the capture learned about the post's liveness, for the availability // sidecar. Absent when it learned nothing about the post itself (a login // wall is the session's state, not the post's). availability?: PostAvailability; // Files on disk for this post after the run (shot + media). files: number; error?: string; }; export type PostCaptureResult = { outcomes: PostCaptureOutcome[]; // The run stopped at a login wall, or at a login the media download refused. needsCookies?: boolean; // Why the run stopped before its last id, when it did. The ids after it are // left for a later run — never retried in this one. stoppedEarly?: string; }; export type SocialFetcher = { id: string; label: string; platform: PostPlatform; // True when this fetcher can handle the given account URL. detect(url: string): boolean; fields: SocialFetcherFields; // Cheap metadata lookup for the channel form's "Fetch details" button. probe(url: string, signal?: AbortSignal): Promise; fetch(input: PostFetchInput): Promise; // Optional: report which of these posts are still live at the source. A // fetcher without it simply cannot answer the question, and the sweep skips // that channel rather than guessing — never reporting "deleted" from // ignorance. Ids absent from the returned map are left unchanged. checkAvailability?( input: PostAvailabilityInput, ): Promise>; // Optional: walk the account's history backwards below what `fetch` can // reach (X: search windows). A fetcher without it has no such backfill, and // the channel page offers no "Fetch older posts" for it. fetchOlder?(input: OlderPostFetchInput): Promise; // Optional: capture specific archived posts — a screenshot and the attached // media — into `posts-media//` (X: the logged-in profile shoots the // post, gallery-dl downloads its media). A fetcher without it cannot, and // the capture refuses its channel by name. captureByIds?(input: PostCaptureInput): Promise; }; // Populated by registerSocialFetcher() from each fetcher module. Indirection // (rather than a literal object) keeps this module free of imports from the // individual fetchers, so the heavier ones — which spawn binaries or launch a // browser — are only pulled in by the server entry points that register them. const REGISTRY = new Map(); export function registerSocialFetcher(fetcher: SocialFetcher): void { REGISTRY.set(fetcher.id, fetcher); } export function getSocialFetcher( id: string | undefined | null, ): SocialFetcher | undefined { return id ? REGISTRY.get(id) : undefined; } export function listSocialFetcherIds(): string[] { return [...REGISTRY.keys()]; } // The fetcher that claims this URL. When several match, the first registered // wins — registration order therefore encodes preference, which is how // x-gallery-dl stays the primary X path with x-playwright as its fallback. export function detectSocialFetcher( url: string | undefined | null, ): SocialFetcher | undefined { if (!url) return undefined; for (const fetcher of REGISTRY.values()) { if (fetcher.detect(url)) return fetcher; } return undefined; } // Fetchers for a platform, in registration (preference) order. export function fetchersForPlatform( platform: PostPlatform, ): SocialFetcher[] { return [...REGISTRY.values()].filter((f) => f.platform === platform); } // Resolve the fetcher for a channel: an explicit postFetcher wins, otherwise // fall back to URL detection so an existing channel keeps working if its // configured fetcher is later removed. export function resolveSocialFetcher( postFetcher: string | undefined | null, accountUrl: string | undefined | null, ): SocialFetcher | undefined { return getSocialFetcher(postFetcher) ?? detectSocialFetcher(accountUrl); } // A client-safe view (no functions) for the channel form. export type SocialFetcherDescriptor = { id: string; label: string; platform: PostPlatform; fields: SocialFetcherFields; }; export function listSocialFetchers(): SocialFetcherDescriptor[] { return [...REGISTRY.values()].map((f) => ({ id: f.id, label: f.label, platform: f.platform, fields: f.fields, })); } // --------------------------------------------------------------------------- // Handle parsing // --------------------------------------------------------------------------- // Extract a bare handle from an account URL or a raw "@handle" string. // Returns null when nothing handle-shaped is present. export function handleFromAccountUrl(input: string): string | null { const trimmed = input.trim(); if (!trimmed) return null; if (!/^https?:\/\//i.test(trimmed)) { // Bare handle form: "@name", "name", "name.bsky.social". const bare = trimmed.replace(/^@/, ""); return /^[A-Za-z0-9._-]+$/.test(bare) ? bare : null; } let url: URL; try { url = new URL(trimmed); } catch { return null; } const segments = url.pathname.split("/").filter(Boolean); if (segments.length === 0) return null; // A forum thread: its "." key is the channel's handle. const threadIdx = segments.findIndex((s) => s === "threads"); if (threadIdx >= 0 && /(?:^|\.)\d+$/.test(segments[threadIdx + 1] ?? "")) { return decodeURIComponent(segments[threadIdx + 1]); } // bsky.app/profile/ if (segments[0] === "profile" && segments[1]) { return decodeURIComponent(segments[1]).replace(/^@/, ""); } // x.com/ — skip the reserved paths that are not accounts. const first = decodeURIComponent(segments[0]).replace(/^@/, ""); const RESERVED = new Set([ "home", "explore", "search", "settings", "i", "intent", "notifications", "messages", ]); if (RESERVED.has(first.toLowerCase())) return null; return /^[A-Za-z0-9._-]+$/.test(first) ? first : null; }