commit 5a2b1667e8d3ba644ef5b166f0d4208dd56a6f1c
parent b99630d909dab9bfca1e0cdb36a9898d92a562a4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 16:16:57 -0400
Merge x-post-capture (X post capture: a screenshot and the attached media per post id, capture-posts job + ops route, PostModal shows a capture behind an editor-only prop)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
21 files changed, 2245 insertions(+), 26 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -255,6 +255,7 @@ pnpm ops lane --json '{"lane":"download","held":true}'
pnpm ops refresh-report --json '{"all":true}'
pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}'
pnpm ops fetch-posts --json '{"slug":"example-x","older":true}' --wait
+pnpm ops capture-posts --json '{"slug":"example-x","ids":["1234567890"]}' --wait
pnpm ops get channel the-quartering
pnpm ops list # every action name
```
diff --git a/common/components/PostModal.test.ts b/common/components/PostModal.test.ts
@@ -0,0 +1,63 @@
+// The post viewer's capture panel: shown only from a capture the editor hands
+// it, with the shot and each medium in its own element; nothing without one.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test components/PostModal.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import * as React from "react";
+import { renderToStaticMarkup } from "react-dom/server";
+import {
+ captureMediaKind,
+ PostCapturePanel,
+ type PostCaptureView,
+} from "./PostModal";
+
+// common's tsconfig has `jsx: "preserve"`, so tsx falls back to the classic
+// transform — `React.createElement` on a free `React` (as BrandMark.test.ts).
+(globalThis as { React?: typeof React }).React = React;
+const render = (capture: PostCaptureView) =>
+ renderToStaticMarkup(React.createElement(PostCapturePanel, { capture }));
+
+const BASE: PostCaptureView = {
+ capturedAt: "2026-02-03T04:05:06.000Z",
+ state: "captured",
+ media: [],
+};
+
+test("with media: the shot, each image and video in its own element, the sensitive note", () => {
+ const html = render({
+ ...BASE,
+ sensitive: true,
+ shot: { src: "/capture/demo/1/shot.png" },
+ media: [
+ { src: "/capture/demo/1/1_1.jpg", name: "1_1.jpg" },
+ { src: "/capture/demo/1/1_2.mp4", name: "1_2.mp4" },
+ { src: "/capture/demo/1/1_3.bin", name: "1_3.bin" },
+ ],
+ });
+ assert.match(html, /data-post-capture=""/);
+ assert.match(html, /<img src="\/capture\/demo\/1\/shot\.png"[^>]*data-post-capture-shot=""/);
+ assert.match(html, /data-post-capture-media="image"><img src="\/capture\/demo\/1\/1_1\.jpg"/);
+ assert.match(html, /data-post-capture-media="video"><video src="\/capture\/demo\/1\/1_2\.mp4"/);
+ assert.match(html, /data-post-capture-media="other"><a href="\/capture\/demo\/1\/1_3\.bin"/);
+ assert.match(html, /behind a sensitive-media cover/);
+});
+
+test("a shot with no media shows the shot alone", () => {
+ const html = render({ ...BASE, shot: { src: "/s.png" } });
+ assert.match(html, /data-post-capture-shot/);
+ assert.doesNotMatch(html, /data-post-capture-media/);
+ assert.doesNotMatch(html, /sensitive/);
+});
+
+test("without media or a shot (a deleted post's record): nothing at all", () => {
+ assert.equal(render({ ...BASE, state: "deleted" }), "");
+});
+
+test("media kinds by extension", () => {
+ assert.equal(captureMediaKind("a.JPG"), "image");
+ assert.equal(captureMediaKind("a.webp"), "image");
+ assert.equal(captureMediaKind("a.mp4"), "video");
+ assert.equal(captureMediaKind("a"), "other");
+});
diff --git a/common/components/PostModal.tsx b/common/components/PostModal.tsx
@@ -33,7 +33,28 @@ function platformLabel(platform: Post["platform"]): string {
return platform === "twitter" ? "X" : "Bluesky";
}
-export default function PostModal() {
+// A post's capture (a screenshot and its attached media, taken by the
+// editor's capture-posts job), as the page that renders this modal serves it.
+// EDITOR-ONLY: the files live beside the channel (`posts-media/<id>/`) and are
+// never published, so the export never passes `loadCapture` and never shows
+// any of this.
+export type PostCaptureView = {
+ capturedAt: string;
+ // The capture's state ("captured", "deleted", …), as recorded.
+ state: string;
+ // The post sat behind a sensitive-media cover when it was shot.
+ sensitive?: boolean;
+ shot?: { src: string };
+ media: { src: string; name: string }[];
+};
+
+export type PostModalProps = {
+ // Where the editor reads a post's capture from. Absent (the export): no
+ // capture is looked for or shown.
+ loadCapture?: (post: Post) => Promise<PostCaptureView | null>;
+};
+
+export default function PostModal({ loadCapture }: PostModalProps = {}) {
const { v: slug, vm } = useUrlParams();
const open = vm === "post" && !!slug;
@@ -42,6 +63,11 @@ export default function PostModal() {
queryFn: () => fetchPost(slug!),
enabled: open,
});
+ const capture = useQuery({
+ queryKey: ["post-capture", slug],
+ queryFn: () => loadCapture!(post.data!),
+ enabled: open && !!loadCapture && !!post.data,
+ });
const thread = useQuery({
queryKey: ["post-thread", slug],
queryFn: () => fetchThread(slug!),
@@ -101,7 +127,7 @@ export default function PostModal() {
{post.data && (
<div className="flex flex-col gap-4 px-4 py-4">
- <PostBody post={post.data} primary />
+ <PostBody post={post.data} primary capture={capture.data} />
{/* Thread context: the archived parent + replies around this post.
The post-corpus analogue of a transcript's surrounding cues. */}
@@ -130,12 +156,15 @@ function PostBody({
post,
primary = false,
compact = false,
+ capture,
}: {
post: Post;
primary?: boolean;
compact?: boolean;
+ capture?: PostCaptureView | null;
}) {
const links = useMemo(() => post.links ?? [], [post.links]);
+ const capturedMedia = capture?.media.length ?? 0;
return (
<article
data-post-id={post.id}
@@ -173,6 +202,8 @@ function PostBody({
{post.text}
</p>
+ {capture && <PostCapturePanel capture={capture} />}
+
{links.length > 0 && (
<ul className="mt-2 flex flex-col gap-1">
{links.map((href) => (
@@ -208,7 +239,11 @@ function PostBody({
{post.engagement?.replies != null && (
<span>{post.engagement.replies} replies</span>
)}
- {post.mediaCount ? <span>{post.mediaCount} media (not archived)</span> : null}
+ {capturedMedia > 0 ? (
+ <span>{capturedMedia} media (captured)</span>
+ ) : post.mediaCount ? (
+ <span>{post.mediaCount} media (not archived)</span>
+ ) : null}
</footer>
</article>
);
@@ -221,3 +256,53 @@ function Badge({ children }: { children: React.ReactNode }) {
</span>
);
}
+
+// What a captured file is, by its extension.
+export function captureMediaKind(name: string): "image" | "video" | "other" {
+ const ext = name.toLowerCase().split(".").pop() ?? "";
+ if (["jpg", "jpeg", "png", "gif", "webp", "avif"].includes(ext)) return "image";
+ if (["mp4", "webm", "mov", "m4v"].includes(ext)) return "video";
+ return "other";
+}
+
+// The captured screenshot and media of one post. Nothing when the capture
+// holds neither (a deleted post's record, say).
+export function PostCapturePanel({ capture }: { capture: PostCaptureView }) {
+ if (!capture.shot && capture.media.length === 0) return null;
+ return (
+ <section data-post-capture="" className="mt-3 flex flex-col gap-2">
+ <h3 className="text-xs uppercase tracking-wide text-muted-foreground">
+ Captured {formatWhen(capture.capturedAt)}
+ {capture.sensitive ? " · behind a sensitive-media cover" : ""}
+ </h3>
+ {capture.shot && (
+ <img
+ src={capture.shot.src}
+ alt="Screenshot of the post as captured"
+ data-post-capture-shot=""
+ className="max-w-full rounded-md border border-border"
+ />
+ )}
+ {capture.media.length > 0 && (
+ <ul className="flex flex-col gap-2">
+ {capture.media.map((m) => {
+ const kind = captureMediaKind(m.name);
+ return (
+ <li key={m.src} data-post-capture-media={kind}>
+ {kind === "image" ? (
+ <img src={m.src} alt={m.name} className="max-w-full rounded-md" />
+ ) : kind === "video" ? (
+ <video src={m.src} controls preload="metadata" className="max-w-full rounded-md" />
+ ) : (
+ <a href={m.src} className="text-xs text-primary underline break-all">
+ {m.name}
+ </a>
+ )}
+ </li>
+ );
+ })}
+ </ul>
+ )}
+ </section>
+ );
+}
diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts
@@ -0,0 +1,145 @@
+// capturePosts over a registered fake fetcher: what is refused before any
+// capture, and how the pages' verdicts land in the availability sidecar.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/capturePosts.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "captureposts-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+
+const { getPaths } = await import("../lib/paths");
+const { registerSocialFetcher } = await import("../social/fetchers");
+const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server");
+const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts");
+type Input = import("../social/fetchers").PostCaptureInput;
+type Result = import("../social/fetchers").PostCaptureResult;
+
+// A capturing fetcher whose answer each test sets.
+let calls: Input[] = [];
+let answer: Result = { outcomes: [] };
+registerSocialFetcher({
+ id: "test-capture",
+ label: "Test capture",
+ platform: "bluesky",
+ fields: {},
+ detect: () => false,
+ probe: async () => ({ ok: true }),
+ fetch: async () => ({ posts: [], complete: true }),
+ captureByIds: async (input) => {
+ calls.push(input);
+ return answer;
+ },
+});
+registerSocialFetcher({
+ id: "test-no-capture",
+ label: "No-capture fetcher",
+ platform: "bluesky",
+ fields: {},
+ detect: () => false,
+ probe: async () => ({ ok: true }),
+ fetch: async () => ({ posts: [], complete: true }),
+});
+
+const channelsDir = path.join(ROOT, "transcripts", "channels");
+
+async function makeChannel(slug: string, over: Record<string, unknown> = {}): Promise<string> {
+ const root = path.join(channelsDir, slug);
+ await mkdir(root, { recursive: true });
+ await writeFile(
+ path.join(root, "config.json"),
+ JSON.stringify({
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "bluesky",
+ postFetcher: "test-capture",
+ socialHandle: "example.bsky.social",
+ name: "Example",
+ url: "https://bsky.app/profile/example.bsky.social",
+ ...over,
+ }),
+ );
+ await writeFile(path.join(root, "posts-archive"), "bluesky aaa\nbluesky bbb\nbluesky ccc\n");
+ return root;
+}
+
+const run = (slug: string, ids: string[], over: Record<string, unknown> = {}) =>
+ capturePosts({ paths: getPaths(), slug, settings: {}, ids, ...over });
+
+test("refused before any capture: nothing asked for, no ids, not social, no capture, ids not archived", async () => {
+ await makeChannel("demo-social");
+ await makeChannel("demo-nocapture", { postFetcher: "test-no-capture" });
+ await mkdir(path.join(channelsDir, "demo-video"), { recursive: true });
+ await writeFile(
+ path.join(channelsDir, "demo-video", "config.json"),
+ JSON.stringify({ handling: "transcribe", name: "Video", url: "https://example.com/c" }),
+ );
+ calls = [];
+ assert.equal((await run("demo-social", ["aaa"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE);
+ assert.match((await run("demo-social", [])).error ?? "", /No post ids/);
+ assert.equal((await run("demo-video", ["aaa"])).error, "demo-video is not a social channel.");
+ assert.equal((await run("demo-nocapture", ["aaa"])).error, "No-capture fetcher cannot capture posts.");
+ assert.equal(
+ (await run("demo-social", ["aaa", "zzz", "yyy"])).error,
+ "2 id(s) not in demo-social's posts archive: zzz, yyy",
+ );
+ assert.equal(calls.length, 0);
+});
+
+test("the fetcher gets the ids (deduped), the channel's posts-media dir and the halves asked for", async () => {
+ const root = await makeChannel("demo-args");
+ calls = [];
+ answer = { outcomes: [] };
+ const res = await run("demo-args", ["aaa", "bbb", "aaa"], { media: false, force: true });
+ assert.equal(res.ok, true);
+ assert.equal(calls.length, 1);
+ assert.deepEqual(calls[0].ids, ["aaa", "bbb"]);
+ assert.equal(calls[0].outDir, path.join(root, "posts-media"));
+ assert.equal(calls[0].handle, "example.bsky.social");
+ assert.equal(calls[0].media, false);
+ assert.equal(calls[0].force, true);
+});
+
+test("deleted and walled posts are recorded in posts-availability.json; a login wall is not", async () => {
+ const root = await makeChannel("demo-avail");
+ await writePostAvailability(root, {
+ aaa: { availability: "available", checkedAt: "2026-01-01T00:00:00.000Z" },
+ });
+ answer = {
+ outcomes: [
+ { id: "aaa", state: "deleted", availability: "deleted", files: 0 },
+ { id: "bbb", state: "unavailable", availability: "account_unavailable", files: 0 },
+ { id: "ccc", state: "login-wall", files: 0 },
+ ],
+ needsCookies: true,
+ stoppedEarly: "X asked to log in. Stopped at ccc.",
+ };
+ const res = await run("demo-avail", ["aaa", "bbb", "ccc"]);
+ assert.equal(res.ok, false);
+ assert.equal(res.needsCookies, true);
+ assert.match(res.error ?? "", /Stopped at ccc/);
+ const map = await readPostAvailability(root);
+ assert.equal(map.aaa.availability, "deleted");
+ // The change keeps the moment it was last seen up.
+ assert.deepEqual(map.aaa.history, [{ availability: "available", at: "2026-01-01T00:00:00.000Z" }]);
+ assert.equal(map.bbb.availability, "account_unavailable");
+ assert.equal(map.ccc, undefined);
+});
+
+test("a run stopped by the source fails the job; one cancelled by the operator does not", async () => {
+ await makeChannel("demo-stop");
+ answer = { outcomes: [], stoppedEarly: "X is refusing pages right now." };
+ const stopped = await run("demo-stop", ["aaa"]);
+ assert.equal(stopped.ok, false);
+ assert.match(stopped.error ?? "", /refusing/);
+
+ const ac = new AbortController();
+ ac.abort();
+ answer = { outcomes: [], stoppedEarly: "Cancelled; the rest are left for a later run." };
+ const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal });
+ assert.equal(cancelled.ok, true);
+});
diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts
@@ -0,0 +1,206 @@
+// Capture specific archived posts of a social channel: a screenshot of each
+// post as its platform renders it, and its attached media, into
+// `channels/<slug>/posts-media/<id>/` (social/postCapture.ts owns the layout).
+//
+// The fetcher does the capturing (`SocialFetcher.captureByIds`); this lands
+// what it learned about each post's liveness in the channel's availability
+// sidecar, `posts-availability.json`, through the same merge the deleted-post
+// sweep uses — a deleted or walled post is recorded there, history appended
+// only on a change, and nothing is recorded from ignorance (a login wall says
+// nothing about the post).
+//
+// Only ids already in the channel's posts archive are captured: a capture is
+// of the archive, and an id from somewhere else is refused by name.
+
+import path from "node:path";
+import { readChannelConfig } from "./channels";
+import type { Paths } from "../lib/paths";
+import type { PostAvailability } from "../lib/posts";
+import { isSocialChannel } from "../lib/channelConfig";
+import {
+ alwaysCookies,
+ resolveCookiePolicy,
+ type CookiePolicyInputs,
+} from "../lib/cookiePolicy";
+import {
+ mergePostAvailability,
+ readPostAvailability,
+ readSeenPostIds,
+ writePostAvailability,
+} from "../lib/posts-server";
+import {
+ handleFromAccountUrl,
+ resolveSocialFetcher,
+ type PostCaptureOutcome,
+ type SocialFetcher,
+} from "../social/fetchers";
+import { postsMediaDir } from "../social/postCapture";
+import "../social/blueskyFetcher";
+import "../social/xGalleryDlFetcher";
+import "../social/xPlaywrightFetcher";
+import "../social/xNitterFetcher";
+import {
+ resolveXCookieSourceFor,
+ type XLoginSettings,
+} from "../social/xBrowserLogin";
+
+export type CapturePostsOptions = {
+ paths: Paths;
+ slug: string;
+ settings: CookiePolicyInputs & XLoginSettings;
+ ids: ReadonlyArray<string>;
+ // Take the screenshot / download the media. Both default to true.
+ shots?: boolean;
+ media?: boolean;
+ // Capture again what is already captured.
+ force?: boolean;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // The job's drain: stop between posts.
+ drain?: AbortSignal;
+};
+
+export type CapturePostsResult = {
+ ok: boolean;
+ outcomes: PostCaptureOutcome[];
+ needsCookies?: boolean;
+ error?: string;
+};
+
+// Why this fetcher cannot capture posts, or null when it can. Shared with the
+// server action, so the refusal is one sentence everywhere.
+export function capturePostsProblem(
+ fetcher: Pick<SocialFetcher, "label" | "captureByIds"> | undefined,
+): string | null {
+ if (!fetcher) return "This channel has no post fetcher.";
+ return typeof fetcher.captureByIds === "function"
+ ? null
+ : `${fetcher.label} cannot capture posts.`;
+}
+
+export const NOTHING_TO_CAPTURE =
+ "Nothing to capture: both the screenshot and the media are turned off.";
+
+// The ids not in the channel's archive, or null when every one is.
+export function strayCaptureIds(
+ ids: ReadonlyArray<string>,
+ archived: ReadonlySet<string>,
+): string[] | null {
+ const stray = ids.filter((id) => !archived.has(id));
+ return stray.length ? stray : null;
+}
+
+export function strayIdsRefusal(slug: string, stray: string[]): string {
+ return `${stray.length} id(s) not in ${slug}'s posts archive: ${stray.join(", ")}`;
+}
+
+export async function capturePosts(
+ opts: CapturePostsOptions,
+): Promise<CapturePostsResult> {
+ const { paths, slug, settings, onLog } = opts;
+ const log = (line: string) => onLog?.(line);
+ const fail = (error: string): CapturePostsResult => ({ ok: false, outcomes: [], error });
+ const channelRoot = path.join(paths.channelsDir, slug);
+ const ids = [...new Set(opts.ids)];
+
+ if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE);
+ if (ids.length === 0) return fail("No post ids to capture.");
+
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return fail(`No such channel: ${slug}`);
+ if (!isSocialChannel(config)) return fail(`${slug} is not a social channel.`);
+
+ const accountUrl = config.url ?? "";
+ const fetcher = resolveSocialFetcher(config.postFetcher, accountUrl);
+ const problem = capturePostsProblem(fetcher);
+ if (problem || !fetcher?.captureByIds) return fail(problem ?? "This channel has no post fetcher.");
+
+ const handle = config.socialHandle ?? handleFromAccountUrl(accountUrl) ?? "";
+ if (!handle) return fail(`Could not determine an account handle for ${slug}`);
+
+ const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot));
+ if (stray) return fail(strayIdsRefusal(slug, stray));
+
+ const policy = resolveCookiePolicy(settings, config);
+ const xLogin =
+ fetcher.platform === "twitter"
+ ? await resolveXCookieSourceFor(paths, settings, policy.cookies)
+ : undefined;
+
+ log(
+ `Capturing ${ids.length} post(s) of ${slug} via ${fetcher.label} (@${handle}):` +
+ [
+ opts.shots === false ? "" : " screenshot",
+ opts.media === false ? "" : " media",
+ ].join("") +
+ (opts.force ? ", again where already captured" : "") +
+ ".",
+ );
+
+ const controller = new AbortController();
+ let result;
+ try {
+ result = await fetcher.captureByIds({
+ ids,
+ handle,
+ outDir: postsMediaDir(channelRoot),
+ shots: opts.shots,
+ media: opts.media,
+ force: opts.force,
+ cookies: alwaysCookies(policy),
+ cookieSource: xLogin?.source,
+ browserCookies: xLogin?.browserSpec,
+ signal: opts.signal ?? controller.signal,
+ drain: opts.drain,
+ onLog: log,
+ });
+ } catch (err) {
+ const message = (err as Error).message;
+ log(`[error] ${message}`);
+ return fail(message);
+ }
+
+ // What the pages said about each post, folded into the sidecar.
+ const observations = new Map<string, PostAvailability>();
+ for (const o of result.outcomes) {
+ if (o.availability) observations.set(o.id, o.availability);
+ }
+ if (observations.size > 0) {
+ const { map, newlyDeleted, changed } = mergePostAvailability(
+ await readPostAvailability(channelRoot),
+ observations,
+ new Date().toISOString(),
+ );
+ await writePostAvailability(channelRoot, map);
+ log(
+ `Availability: ${observations.size} recorded, ${changed} changed` +
+ (newlyDeleted.length ? `, newly deleted: ${newlyDeleted.join(", ")}` : "") +
+ ".",
+ );
+ }
+
+ const count = (state: string) => result.outcomes.filter((o) => o.state === state).length;
+ log(
+ `Done: ${count("captured")} captured, ${count("deleted")} deleted, ` +
+ `${count("unavailable")} unavailable, ${count("error")} failed` +
+ (count("login-wall") ? `, ${count("login-wall")} met a login wall` : "") +
+ `; ${ids.length - result.outcomes.length} not attempted or already captured.`,
+ );
+
+ // A run that stopped at a login is a failed run (the job says so); one that
+ // was cancelled or met deleted posts is not.
+ if (result.needsCookies) {
+ return {
+ ok: false,
+ outcomes: result.outcomes,
+ needsCookies: true,
+ error: result.stoppedEarly ?? "The capture needs an X login.",
+ };
+ }
+ const stoppedByOperator =
+ (opts.signal ?? controller.signal).aborted || Boolean(opts.drain?.aborted);
+ if (result.stoppedEarly && !stoppedByOperator) {
+ return { ok: false, outcomes: result.outcomes, error: result.stoppedEarly };
+ }
+ return { ok: true, outcomes: result.outcomes };
+}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -512,6 +512,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
replayable: true,
queueKeyStrategy: "platform",
},
+ // A screenshot and the attached media of specific archived posts, into the
+ // channel's `posts-media/` (controller/capturePosts.ts). On the PLATFORM
+ // queue, like fetch-posts: each post is a page load and a download against
+ // the same source, so it serialises with the fetch and shares its backoff.
+ // Drainable (it stops between posts) and replayable (the ids are in the
+ // spec; posts already captured are skipped on a re-run).
+ "capture-posts": {
+ kind: "capture-posts",
+ label: "Capture posts",
+ drainable: true,
+ replayable: true,
+ queueKeyStrategy: "platform",
+ },
// The posts analogue of the video availability check: which archived posts
// have since been deleted at the source.
"check-post-availability": {
diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts
@@ -16,6 +16,7 @@
import type { Post, PostAvailability, PostPlatform } from "../lib/posts";
import type { OlderBackfillPosition } from "../lib/posts-server";
import type { XCookieSource } from "./xCookieSource";
+import type { PostCaptureState } from "./postCapture";
export type PostFetchInput = {
// The account's canonical URL as configured on the channel.
@@ -164,6 +165,49 @@ export type PostAvailabilityInput = {
onLog?: (line: string) => void;
};
+// Input for a post capture (`SocialFetcher.captureByIds`): a screenshot of each
+// post as the platform renders it, and its attached media, for specific
+// archived posts. The login fields are a normal fetch's; `outDir` is the
+// channel's `posts-media/`, one directory per post id beneath it
+// (postCapture.ts owns the layout).
+export type PostCaptureInput = Pick<
+ PostFetchInput,
+ "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog"
+> & {
+ ids: ReadonlyArray<string>;
+ handle: string;
+ outDir: string;
+ // Take the screenshot / download the media. Both default to true.
+ shots?: boolean;
+ media?: boolean;
+ // Capture again what is already captured.
+ force?: boolean;
+ // A soft stop: no new post is started once it fires, and the one in hand
+ // finishes (`signal` cancels outright).
+ drain?: AbortSignal;
+};
+
+export type PostCaptureOutcome = {
+ id: string;
+ state: PostCaptureState;
+ // What the capture learned about the post's liveness, for the availability
+ // sidecar. Absent when it learned nothing about the post itself (a login
+ // wall is the session's state, not the post's).
+ availability?: PostAvailability;
+ // Files on disk for this post after the run (shot + media).
+ files: number;
+ error?: string;
+};
+
+export type PostCaptureResult = {
+ outcomes: PostCaptureOutcome[];
+ // The run stopped at a login wall, or at a login the media download refused.
+ needsCookies?: boolean;
+ // Why the run stopped before its last id, when it did. The ids after it are
+ // left for a later run — never retried in this one.
+ stoppedEarly?: string;
+};
+
export type SocialFetcher = {
id: string;
label: string;
@@ -185,6 +229,11 @@ export type SocialFetcher = {
// reach (X: search windows). A fetcher without it has no such backfill, and
// the channel page offers no "Fetch older posts" for it.
fetchOlder?(input: OlderPostFetchInput): Promise<OlderPostFetchResult>;
+ // Optional: capture specific archived posts — a screenshot and the attached
+ // media — into `posts-media/<id>/` (X: the logged-in profile shoots the
+ // post, gallery-dl downloads its media). A fetcher without it cannot, and
+ // the capture refuses its channel by name.
+ captureByIds?(input: PostCaptureInput): Promise<PostCaptureResult>;
};
// Populated by registerSocialFetcher() from each fetcher module. Indirection
diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts
@@ -41,6 +41,13 @@ export type PageLike = {
waitForTimeout: (ms: number) => Promise<void>;
evaluate: (fn: string) => Promise<unknown>;
on: (event: string, cb: (arg: never) => void) => void;
+ // The post capture's shot (xPostCapture.ts): a PNG of the page, clipped to
+ // the post. `clip` is in document coordinates when `fullPage` is set.
+ screenshot: (opts?: {
+ type?: "png" | "jpeg";
+ fullPage?: boolean;
+ clip?: { x: number; y: number; width: number; height: number };
+ }) => Promise<Uint8Array>;
};
export type BrowserContextLike = {
diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts
@@ -0,0 +1,159 @@
+// The post capture's on-disk record (capture.json), what a capture still owes
+// a post, and the guard that keeps `posts-media/` out of every export.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/postCapture.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { createHash } from "node:crypto";
+import { mkdir, mkdtemp, readdir, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import { fileURLToPath } from "node:url";
+import {
+ CAPTURE_FILENAME,
+ captureAvailability,
+ captureWork,
+ describeCapturedFile,
+ listCapturedMediaFiles,
+ POSTS_MEDIA_DIRNAME,
+ postCaptureDir,
+ postsMediaDir,
+ readPostCapture,
+ SHOT_FILENAME,
+ writePostCapture,
+ type PostCaptureRecord,
+} from "./postCapture";
+
+const sha = (b: string | Buffer) => createHash("sha256").update(b).digest("hex");
+
+function record(over: Partial<PostCaptureRecord> = {}): PostCaptureRecord {
+ return {
+ version: 1,
+ id: "123",
+ url: "https://x.com/i/status/123",
+ capturedAt: "2026-01-02T03:04:05.000Z",
+ state: "captured",
+ mediaState: "ok",
+ media: [],
+ ...over,
+ };
+}
+
+test("layout: posts-media/<id>/ beside the channel's data, and an id is never a path", () => {
+ assert.equal(postsMediaDir("/c/channels/demo-channel"), "/c/channels/demo-channel/posts-media");
+ assert.equal(POSTS_MEDIA_DIRNAME, "posts-media");
+ assert.equal(postCaptureDir("/o", "1234"), "/o/1234");
+ assert.equal(postCaptureDir("/o", "3kabc_d-e"), "/o/3kabc_d-e");
+ for (const id of ["../x", "a/b", "", ".", "..", "a b"]) {
+ assert.throws(() => postCaptureDir("/o", id), /is not a post id/, JSON.stringify(id));
+ }
+});
+
+test("capture.json: each file's sha256 and byte size, the URLs, round-tripped", async () => {
+ const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "capture-")), "123");
+ await mkdir(dir, { recursive: true });
+ const png = Buffer.from([0x89, 0x50, 0x4e, 0x47, 1, 2, 3]);
+ await writeFile(path.join(dir, SHOT_FILENAME), png);
+ await writeFile(path.join(dir, "123_1.jpg"), "jpeg bytes");
+ const shot = await describeCapturedFile(dir, SHOT_FILENAME, "https://x.com/i/status/123");
+ const media = await describeCapturedFile(dir, "123_1.jpg", "https://pbs.example/m.jpg");
+ assert.deepEqual(shot, {
+ name: SHOT_FILENAME,
+ bytes: png.length,
+ sha256: sha(png),
+ url: "https://x.com/i/status/123",
+ });
+ assert.equal(media.sha256, sha("jpeg bytes"));
+ assert.equal(media.bytes, "jpeg bytes".length);
+ // No URL known: the key is absent, never "undefined".
+ assert.ok(!("url" in (await describeCapturedFile(dir, "123_1.jpg"))));
+
+ const rec = record({ shot, media: [media] });
+ await writePostCapture(dir, rec);
+ assert.deepEqual(JSON.parse(await readFile(path.join(dir, CAPTURE_FILENAME), "utf8")), rec);
+ assert.deepEqual(await readPostCapture(dir), rec);
+ // The record and the shot are not media.
+ assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]);
+});
+
+test("capture.json: absent, unparseable or of another shape reads as no capture", async () => {
+ const dir = await mkdtemp(path.join(os.tmpdir(), "capture-"));
+ assert.equal(await readPostCapture(dir), null);
+ await writeFile(path.join(dir, CAPTURE_FILENAME), "{ truncated");
+ assert.equal(await readPostCapture(dir), null);
+ await writeFile(path.join(dir, CAPTURE_FILENAME), JSON.stringify({ version: 2, id: "1" }));
+ assert.equal(await readPostCapture(dir), null);
+ assert.deepEqual(await listCapturedMediaFiles(path.join(dir, "missing")), []);
+});
+
+test("availability: only what the page said about the post — never a verdict from a login wall or an error", () => {
+ assert.equal(captureAvailability("captured"), "available");
+ assert.equal(captureAvailability("deleted"), "deleted");
+ assert.equal(captureAvailability("unavailable"), "account_unavailable");
+ assert.equal(captureAvailability("login-wall"), undefined);
+ assert.equal(captureAvailability("error"), undefined);
+});
+
+test("what a capture owes: nothing for a settled post, only the missing half otherwise, everything when forced", () => {
+ const all = { shots: true, media: true, force: false };
+ assert.deepEqual(captureWork(null, all), { shot: true, media: true });
+ assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false });
+ const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" });
+ assert.deepEqual(captureWork(done, all), { shot: false, media: false });
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false });
+ // The media failed (or was never asked for): only the media is owed.
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true });
+ assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true });
+ // A shot that failed is owed again.
+ assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false });
+ // Deleted is settled: asking again is a request for the same answer.
+ assert.deepEqual(captureWork(record({ state: "deleted", mediaState: "skipped" }), all), {
+ shot: false,
+ media: false,
+ });
+ // force re-takes whatever was asked for.
+ assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true });
+ assert.deepEqual(captureWork(record({ state: "deleted" }), { shots: false, media: true, force: true }), {
+ shot: false,
+ media: true,
+ });
+});
+
+// THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON
+// pages and copies trees out of export/public — it never reads a channel
+// directory — so nothing it builds can carry posts-media/. Held here the cheap
+// way: no module of the export build, the publish layer or the export app
+// names the directory.
+test("no export or publish source names posts-media/", async () => {
+ const HERE = path.dirname(fileURLToPath(import.meta.url));
+ const repo = path.resolve(HERE, "..", "..");
+ const roots = [
+ path.join(repo, "common", "publish"),
+ path.join(repo, "common", "bin"),
+ path.join(repo, "export", "app"),
+ path.join(repo, "export", "lib"),
+ ];
+ const offenders: string[] = [];
+ const walk = async (dir: string): Promise<void> => {
+ let entries;
+ try {
+ entries = await readdir(dir, { withFileTypes: true });
+ } catch {
+ return;
+ }
+ for (const e of entries) {
+ if (e.name === "node_modules" || e.name.startsWith(".")) continue;
+ const p = path.join(dir, e.name);
+ if (e.isDirectory()) await walk(p);
+ else if (/\.(ts|tsx|mjs|js)$/.test(e.name)) {
+ const text = await readFile(p, "utf8");
+ if (text.includes(POSTS_MEDIA_DIRNAME) || text.includes("POSTS_MEDIA_DIRNAME") || text.includes("social/postCapture")) {
+ offenders.push(path.relative(repo, p));
+ }
+ }
+ }
+ };
+ for (const r of roots) await walk(r);
+ assert.deepEqual(offenders, []);
+});
diff --git a/common/social/postCapture.ts b/common/social/postCapture.ts
@@ -0,0 +1,189 @@
+// Where a post capture lands on disk, and the record that says what it holds.
+//
+// Layout, per channel:
+// channels/<slug>/posts-media/<post id>/shot.png — the post, as rendered
+// channels/<slug>/posts-media/<post id>/<media…> — its attached media
+// channels/<slug>/posts-media/<post id>/capture.json — this module's record
+//
+// A directory per post, unlike the posts themselves (month-sharded JSONL): a
+// capture is a handful of files, and only for the posts someone asked for. It
+// sits BESIDE `data/` and `media/`, never in them — no video reader walks it,
+// and the media tier does not move it.
+//
+// EDITOR-ONLY. The export build serves the index's JSON pages and never reads a
+// channel directory, so nothing here is published (postCapture.test.ts holds
+// the export's sources to that).
+//
+// SERVER-ONLY (node:fs).
+
+import { createHash } from "node:crypto";
+import { createReadStream } from "node:fs";
+import { readdir, stat } from "node:fs/promises";
+import path from "node:path";
+import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server";
+import type { PostAvailability } from "../lib/posts";
+
+export const POSTS_MEDIA_DIRNAME = "posts-media";
+export const CAPTURE_FILENAME = "capture.json";
+export const SHOT_FILENAME = "shot.png";
+
+export function postsMediaDir(channelRoot: string): string {
+ return path.join(channelRoot, POSTS_MEDIA_DIRNAME);
+}
+
+// One post's capture directory. The id is checked here, at the one place a
+// post id becomes a path: a platform-native id is digits (X) or a short
+// alphanumeric key (Bluesky), never a separator.
+export function postCaptureDir(outDir: string, id: string): string {
+ if (!/^[A-Za-z0-9_-]{1,64}$/.test(id)) {
+ throw new Error(`"${id}" is not a post id`);
+ }
+ return path.join(outDir, id);
+}
+
+// What a capture found the post to be.
+// captured — the post rendered, and what was asked for is on disk
+// deleted — the platform says the post is gone
+// unavailable — the post is behind its account's wall (protected,
+// suspended, withheld): the post may exist, it cannot be shown
+// login-wall — the platform asked to log in: the session's state, not the
+// post's, so the run stops
+// error — the page or the download failed; a later run tries again
+export type PostCaptureState =
+ | "captured"
+ | "deleted"
+ | "unavailable"
+ | "login-wall"
+ | "error";
+
+// The liveness a capture state says about the post, for the availability
+// sidecar. A login wall and an error say nothing about the post itself — the
+// sidecar never records a verdict from ignorance.
+export function captureAvailability(
+ state: PostCaptureState,
+): PostAvailability | undefined {
+ switch (state) {
+ case "captured":
+ return "available";
+ case "deleted":
+ return "deleted";
+ case "unavailable":
+ return "account_unavailable";
+ default:
+ return undefined;
+ }
+}
+
+export type CapturedFile = {
+ // The file's name inside the post's directory.
+ name: string;
+ bytes: number;
+ sha256: string;
+ // Where it came from: the media URL for an attachment, the post URL for the
+ // screenshot.
+ url?: string;
+};
+
+// How the media half went: downloaded ("ok"), the post has none ("none"), not
+// asked for or not attempted ("skipped"), or failed ("error").
+export type CaptureMediaState = "ok" | "none" | "skipped" | "error";
+
+export type PostCaptureRecord = {
+ version: 1;
+ id: string;
+ // The post's URL the capture read.
+ url: string;
+ capturedAt: string;
+ state: PostCaptureState;
+ // The post sat behind a sensitive-media interstitial, which the capture
+ // opened before shooting.
+ sensitive?: boolean;
+ // The screenshot, when one was taken.
+ shot?: CapturedFile;
+ mediaState: CaptureMediaState;
+ media: CapturedFile[];
+ error?: string;
+};
+
+export async function fileDigest(
+ file: string,
+): Promise<{ bytes: number; sha256: string }> {
+ const hash = createHash("sha256");
+ await new Promise<void>((resolve, reject) => {
+ const s = createReadStream(file);
+ s.on("data", (chunk) => hash.update(chunk));
+ s.on("error", reject);
+ s.on("end", () => resolve());
+ });
+ const { size } = await stat(file);
+ return { bytes: size, sha256: hash.digest("hex") };
+}
+
+export async function describeCapturedFile(
+ dir: string,
+ name: string,
+ url?: string,
+): Promise<CapturedFile> {
+ const digest = await fileDigest(path.join(dir, name));
+ return { name, ...digest, ...(url ? { url } : {}) };
+}
+
+// The media files in a post's directory: everything but the shot, the record
+// and a download's leftovers.
+export async function listCapturedMediaFiles(dir: string): Promise<string[]> {
+ let names: string[];
+ try {
+ names = await readdir(dir);
+ } catch {
+ return [];
+ }
+ return names
+ .filter(
+ (n) =>
+ n !== SHOT_FILENAME &&
+ n !== CAPTURE_FILENAME &&
+ !n.endsWith(".part") &&
+ !n.startsWith("."),
+ )
+ .sort();
+}
+
+export async function readPostCapture(
+ dir: string,
+): Promise<PostCaptureRecord | null> {
+ const read = await readJsonFile(path.join(dir, CAPTURE_FILENAME));
+ if (!read.ok) return null;
+ const v = read.value as Partial<PostCaptureRecord> | null;
+ if (!v || typeof v !== "object" || v.version !== 1 || typeof v.id !== "string") {
+ return null;
+ }
+ return v as PostCaptureRecord;
+}
+
+export async function writePostCapture(
+ dir: string,
+ record: PostCaptureRecord,
+): Promise<void> {
+ await writeJsonAtomic(path.join(dir, CAPTURE_FILENAME), record, { mkdir: true });
+}
+
+// Which halves of a capture this run still owes a post, given what is on disk.
+// A deleted post is settled: the platform said so, and asking again costs a
+// request for the same answer (`force` asks anyway). A shot on disk is kept; a
+// media download that did not finish ("error", or never attempted) is owed.
+export function captureWork(
+ existing: PostCaptureRecord | null,
+ wanted: { shots: boolean; media: boolean; force: boolean },
+): { shot: boolean; media: boolean } {
+ if (wanted.force || !existing) {
+ return { shot: wanted.shots, media: wanted.media };
+ }
+ if (existing.state === "deleted") return { shot: false, media: false };
+ return {
+ shot: wanted.shots && !existing.shot,
+ media:
+ wanted.media &&
+ existing.mediaState !== "ok" &&
+ existing.mediaState !== "none",
+ };
+}
diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts
@@ -7,7 +7,11 @@ import { test } from "node:test";
import assert from "node:assert/strict";
import {
buildGalleryDlArgs,
+ buildGalleryDlCaptureArgs,
+ CAPTURE_PRINT_TAG,
galleryDlCookieChoice,
+ parseCapturePrints,
+ xRequestPauseMs,
OLDER_WINDOW_PAUSE_MAX_MS,
OLDER_WINDOW_PAUSE_MIN_MS,
olderWindowPauseMs,
@@ -336,3 +340,66 @@ test("the pause between search windows is a fresh random draw in 45–120 s", ()
assert.ok(mid > OLDER_WINDOW_PAUSE_MIN_MS && mid < OLDER_WINDOW_PAUSE_MAX_MS);
assert.ok(OLDER_WINDOW_PAUSE_MIN_MS >= 45_000);
});
+
+// The post capture's media download: the one gallery-dl run that DOWNLOADS.
+test("capture argv: downloads (no --no-download, no --dump-json), videos on, into exactly the post's dir", () => {
+ const argv = buildGalleryDlCaptureArgs({ id: "1234567890", dir: "/corpus/ch/posts-media/1234567890" });
+ assert.ok(!argv.includes("--no-download"));
+ assert.ok(!argv.includes("--dump-json"));
+ assert.ok(argv.includes("extractor.twitter.videos=true"));
+ assert.ok(!argv.includes("extractor.twitter.videos=false"));
+ assert.equal(flagValue(argv, "-D"), "/corpus/ch/posts-media/1234567890");
+ assert.equal(flagValue(argv, "-f"), "{tweet_id}_{num}.{extension}");
+ assert.equal(argv[argv.length - 1], "https://x.com/i/status/1234567890");
+ // A tagged line per file downloaded, and per file already there.
+ const prints = argv.filter((_, i) => argv[i - 1] === "--Print");
+ assert.deepEqual(prints, [
+ `after:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`,
+ `skip:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`,
+ ]);
+});
+
+test("capture argv: paced and logged in exactly as the reads are", () => {
+ const read = buildGalleryDlArgs({ accountUrl: ACCOUNT, cookieFile: JAR });
+ const capture = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookieFile: JAR });
+ for (const flag of [
+ `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`,
+ "extractor.twitter.ratelimit=wait",
+ ]) {
+ assert.ok(read.includes(flag) && capture.includes(flag), flag);
+ }
+ assert.equal(flagValue(capture, "--cookies"), JAR);
+ const browser = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookies: "firefox" });
+ assert.equal(flagValue(browser, "--cookies-from-browser"), "firefox");
+ assert.ok(!browser.includes("--cookies"));
+ const guest = buildGalleryDlCaptureArgs({ id: "42", dir: "/d" });
+ assert.ok(!guest.includes("--cookies") && !guest.includes("--cookies-from-browser"));
+});
+
+test("capture argv: an id that is not X's digits is refused, not turned into a URL", () => {
+ for (const id of ["../etc", "12a", "", "1 2"]) {
+ assert.throws(() => buildGalleryDlCaptureArgs({ id, dir: "/d" }), /is not a post id/);
+ }
+});
+
+test("capture prints: tagged lines name each file once, with its source URL; the rest is ignored", () => {
+ const out = [
+ "/corpus/ch/posts-media/42/42_1.jpg",
+ `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`,
+ `${CAPTURE_PRINT_TAG}\thttps://video.twimg.com/v/clip.mp4\t/corpus/ch/posts-media/42/42_2.mp4`,
+ `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`,
+ `${CAPTURE_PRINT_TAG}\tNone\t/corpus/ch/posts-media/42/42_3.png`,
+ "# /corpus/ch/posts-media/42/42_1.jpg",
+ ].join("\n");
+ assert.deepEqual(parseCapturePrints(out), [
+ { name: "42_1.jpg", url: "https://pbs.twimg.com/media/abc?format=jpg&name=orig" },
+ { name: "42_2.mp4", url: "https://video.twimg.com/v/clip.mp4" },
+ { name: "42_3.png" },
+ ]);
+});
+
+test("the capture's gap before each contact is the reads' 4–10 s, drawn fresh", () => {
+ assert.equal(xRequestPauseMs(() => 0), 4_000);
+ assert.equal(xRequestPauseMs(() => 1), 10_000);
+ assert.equal(xRequestPauseMs(() => 0.5), 7_000);
+});
diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts
@@ -24,6 +24,8 @@
// `--dump-json` contract and are covered by a fixture binary in the editor e2e
// suite; treat the first real run as the spike.
+import { existsSync } from "node:fs";
+import path from "node:path";
import { createInterface } from "node:readline";
import type { Readable } from "node:stream";
import { execa } from "execa";
@@ -32,12 +34,24 @@ import type { Post } from "../lib/posts";
import type { OlderBackfillPosition } from "../lib/posts-server";
import { normalizeXTweet, xCreatedAt, type XTweetRaw } from "./xNormalize";
import { olderFloorDay, stepOlderWindow } from "./olderBackfill";
-import { readXSessionStatus, xCookieFile } from "./xSessionBroker";
+import {
+ launchXProfile,
+ readXSessionStatus,
+ xCookieFile,
+} from "./xSessionBroker";
import type { XCookieSource } from "./xCookieSource";
+import { listCapturedMediaFiles } from "./postCapture";
+import {
+ captureXPosts,
+ xStatusUrl,
+ type MediaDownloadResult,
+} from "./xPostCapture";
import {
registerSocialFetcher,
type OlderPostFetchInput,
type OlderPostFetchResult,
+ type PostCaptureInput,
+ type PostCaptureResult,
type PostFetchInput,
type PostFetchResult,
type SocialFetcher,
@@ -148,35 +162,200 @@ function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] {
"extractor.twitter.videos=false",
"-o",
"extractor.twitter.cards=false",
- // PACING. gallery-dl's default gap between X API requests is 0: it pages as
- // fast as X answers and only slows down when X's rate-limit headers say to.
- // An account-history walk is hundreds of requests, so every read waits a
- // RANDOM 4–10 s before each one (gallery-dl's own "a-b" range, a uniform
- // draw per request): no fixed rhythm, and well under the rate a person
- // scrolling would make. Running into the limit anyway is waited out, never
- // pushed through ("wait" is gallery-dl's default; stated so it stays).
+ ...galleryDlPacingArgs(),
+ ...galleryDlLoginArgs(opts),
+ ];
+ if (opts.limit && opts.limit > 0) {
+ args.push("--range", `1-${Math.floor(opts.limit)}`);
+ }
+ return args;
+}
+
+// PACING. gallery-dl's default gap between X API requests is 0: it pages as
+// fast as X answers and only slows down when X's rate-limit headers say to. An
+// account-history walk is hundreds of requests, so every read waits a RANDOM
+// 4–10 s before each one (gallery-dl's own "a-b" range, a uniform draw per
+// request): no fixed rhythm, and well under the rate a person scrolling would
+// make. Running into the limit anyway is waited out, never pushed through
+// ("wait" is gallery-dl's default; stated so it stays). Every gallery-dl run
+// against X carries these — the reads and the post capture alike.
+function galleryDlPacingArgs(): string[] {
+ return [
"-o",
`extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`,
"-o",
"extractor.twitter.ratelimit=wait",
];
- // Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no
- // account at all (verified: 500 tweets over ~7.5 months for a public
- // account). Guest access is rate-limited far more aggressively than an
- // authenticated session, though — gallery-dl will block for minutes on
- // "Waiting for N minutes (rate limit)" — so credentials remain the path for
- // a deep backfill. When none are configured we simply run as a guest rather
- // than failing.
- if (opts.cookieFile) {
- args.push("--cookies", opts.cookieFile);
- } else if (opts.cookies) {
- // Same browser-cookie spec yt-dlp uses (cookiePolicy.ts), same syntax.
- args.push("--cookies-from-browser", opts.cookies);
+}
+
+// Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no
+// account at all (verified: 500 tweets over ~7.5 months for a public account).
+// Guest access is rate-limited far more aggressively than an authenticated
+// session, though — gallery-dl will block for minutes on "Waiting for N
+// minutes (rate limit)" — so credentials remain the path for a deep backfill.
+// When none are configured we simply run as a guest rather than failing.
+function galleryDlLoginArgs(opts: GalleryDlLoginAndCap): string[] {
+ if (opts.cookieFile) return ["--cookies", opts.cookieFile];
+ // Same browser-cookie spec yt-dlp uses (cookiePolicy.ts), same syntax.
+ if (opts.cookies) return ["--cookies-from-browser", opts.cookies];
+ return [];
+}
+
+// The random gap a post capture waits before each contact with X after its
+// first (a page load, a gallery-dl run): the same 4–10 s draw as
+// X_SLEEP_REQUEST, which paces gallery-dl's requests inside one run.
+export function xRequestPauseMs(rand: () => number = Math.random): number {
+ const [min, max] = X_SLEEP_REQUEST.split("-").map((s) => Number(s) * 1000);
+ return Math.round(min + rand() * (max - min));
+}
+
+// ---------------------------------------------------------------------------
+// Post capture: one post's attached media
+// ---------------------------------------------------------------------------
+//
+// The media half of `captureByIds` (the screenshot half is xPostCapture.ts).
+// One gallery-dl run per post, against the post's own URL, DOWNLOADING — the
+// one gallery-dl run here without --no-download — into the post's capture
+// directory, videos included, paced and logged in exactly as the reads are.
+// gallery-dl prints a tagged line per file (downloaded, or already there), so
+// each file's source URL is known without a second pass.
+
+export const CAPTURE_PRINT_TAG = "ARCHILYZER-CAPTURED";
+
+export function buildGalleryDlCaptureArgs(
+ opts: Omit<GalleryDlLoginAndCap, "limit"> & { id: string; dir: string },
+): string[] {
+ if (!/^\d+$/.test(opts.id)) throw new Error(`"${opts.id}" is not a post id`);
+ return [
+ // Videos too: a capture keeps what the post carried.
+ "-o",
+ "extractor.twitter.videos=true",
+ "-o",
+ "extractor.twitter.cards=false",
+ ...galleryDlPacingArgs(),
+ // Exactly this directory, and names that cannot meet shot.png or
+ // capture.json.
+ "-D",
+ opts.dir,
+ "-f",
+ "{tweet_id}_{num}.{extension}",
+ // A line per file once it is on disk, and per file already there (a
+ // forced re-capture): "<tag>\t<source url>\t<path>".
+ "--Print",
+ `after:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`,
+ "--Print",
+ `skip:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`,
+ ...galleryDlLoginArgs(opts),
+ xStatusUrl(opts.id),
+ ];
+}
+
+// The files a capture run printed, by name (each once).
+export function parseCapturePrints(
+ stdout: string,
+): { name: string; url?: string }[] {
+ const byName = new Map<string, { name: string; url?: string }>();
+ for (const line of stdout.split("\n")) {
+ const parts = line.replace(/\r$/, "").split("\t");
+ if (parts[0] !== CAPTURE_PRINT_TAG || parts.length < 3) continue;
+ const file = parts.slice(2).join("\t").trim();
+ if (!file) continue;
+ const name = path.basename(file);
+ const url = parts[1].trim();
+ byName.set(name, { name, ...(url && url !== "None" ? { url } : {}) });
}
- if (opts.limit && opts.limit > 0) {
- args.push("--range", `1-${Math.floor(opts.limit)}`);
+ return [...byName.values()];
+}
+
+// How long one post's download may take: a long video at a polite pace.
+const CAPTURE_MEDIA_TIMEOUT_MS = 20 * 60_000;
+
+export async function downloadXPostMedia(args: {
+ id: string;
+ dir: string;
+ login: { cookies?: string; cookieFile?: string };
+ signal: AbortSignal;
+ onLog?: (line: string) => void;
+}): Promise<MediaDownloadResult> {
+ const bin = getPaths().galleryDlBin;
+ const res = await execa(
+ bin,
+ buildGalleryDlCaptureArgs({ id: args.id, dir: args.dir, ...args.login }),
+ {
+ reject: false,
+ stdin: "ignore",
+ timeout: CAPTURE_MEDIA_TIMEOUT_MS,
+ cancelSignal: args.signal,
+ env: { PYTHONUNBUFFERED: "1" },
+ },
+ );
+ const spawnError = `${res.code ?? ""} ${res.message ?? ""}`;
+ if (res.failed && /ENOENT/.test(spawnError) && !res.stdout) {
+ throw new Error(`gallery-dl not found (looked for "${bin}"; set GALLERY_DL_BIN)`);
}
- return args;
+ const stderr = `${res.stderr ?? ""}`;
+ for (const line of stderr.split("\n")) {
+ if (line.trim()) args.onLog?.(`gallery-dl: ${line.trim()}`);
+ }
+ if (res.isCanceled || res.timedOut || res.exitCode !== 0) {
+ const tail = stderr.trim().split("\n").slice(-5).join("\n");
+ if (looksLikeAuthFailure(tail)) {
+ return { ok: false, needsCookies: true, error: `gallery-dl could not authenticate to X: ${tail}` };
+ }
+ return {
+ ok: false,
+ error: res.isCanceled
+ ? "Cancelled during the media download."
+ : res.timedOut
+ ? "The media download ran past its time limit."
+ : `gallery-dl exited ${res.exitCode ?? res.signal ?? "abnormally"}: ${tail || "(no output)"}`,
+ };
+ }
+ // The printed files, plus any on disk it did not print (an older
+ // gallery-dl): every one is in the record.
+ const printed = parseCapturePrints(`${res.stdout ?? ""}`);
+ const files = printed.filter((f) => existsSync(path.join(args.dir, f.name)));
+ const names = new Set(files.map((f) => f.name));
+ for (const name of await listCapturedMediaFiles(args.dir)) {
+ if (!names.has(name)) files.push({ name });
+ }
+ return { ok: true, files };
+}
+
+// `captureByIds` for X: the shot through the session profile, the media
+// through gallery-dl, the run paced as one (xPostCapture.ts).
+export async function captureXPostsByIds(
+ input: PostCaptureInput,
+): Promise<PostCaptureResult> {
+ const paths = getPaths();
+ if (input.shots ?? true) {
+ const status = await readXSessionStatus(paths);
+ if (!status.hasProfile) {
+ const why =
+ "No X session profile to shoot posts with: connect an X account on /settings, then run it again.";
+ input.onLog?.(`[auth] ${why}`);
+ return { outcomes: [], needsCookies: true, stoppedEarly: why };
+ }
+ if (!status.looksAuthenticated) {
+ input.onLog?.(
+ "[warn] The X session profile's exported cookies carry no login; X may show its login wall.",
+ );
+ }
+ }
+ let login: Promise<{ cookies?: string; cookieFile?: string }> | undefined;
+ return captureXPosts(input, {
+ openPage: async () => {
+ const context = await launchXProfile(paths, { onLog: input.onLog });
+ const page = context.pages()[0] ?? (await context.newPage());
+ return { page, close: () => context.close() };
+ },
+ downloadMedia: async ({ id, dir, signal, onLog }) => {
+ login ??= galleryDlLoginFor(input);
+ return downloadXPostMedia({ id, dir, login: await login, signal, onLog });
+ },
+ pauseMs: () => xRequestPauseMs(),
+ pause,
+ });
}
// ---------------------------------------------------------------------------
@@ -1081,6 +1260,8 @@ export const xGalleryDlFetcher: SocialFetcher = {
},
fetchOlder: fetchOlderViaSearch,
+
+ captureByIds: captureXPostsByIds,
};
registerSocialFetcher(xGalleryDlFetcher);
diff --git a/common/social/xPostCapture.test.ts b/common/social/xPostCapture.test.ts
@@ -0,0 +1,436 @@
+// The X post capture, against fakes only: a page that answers the capture's
+// evaluate calls from a recorded snapshot, a media downloader that writes
+// files, and a fake gallery-dl binary for the real download path. No browser,
+// no network.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xPostCapture.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { createHash } from "node:crypto";
+import { chmod, mkdir, mkdtemp, readFile, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+import type { PageLike } from "./playwrightRuntime";
+import {
+ captureXPosts,
+ classifyXPostSnapshot,
+ shootXPost,
+ type XCaptureDeps,
+ type XPostSnapshot,
+} from "./xPostCapture";
+import { readPostCapture, SHOT_FILENAME, writePostCapture } from "./postCapture";
+import type { PostCaptureInput } from "./fetchers";
+
+const sha = (b: string | Uint8Array) => createHash("sha256").update(b).digest("hex");
+const PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 9, 9]);
+const RECT = { x: 10, y: 120.4, width: 598.6, height: 300.2 };
+
+const post = (sensitive = false): XPostSnapshot => ({
+ path: "/someone/status/1",
+ text: "the post",
+ article: { rect: RECT, sensitive },
+});
+const page = (text: string, p = "/i/status/1"): XPostSnapshot => ({ path: p, text, article: null });
+
+// --- classification ---------------------------------------------------------
+
+test("classify: a rendered post is captured; its sensitive cover is noted", () => {
+ assert.deepEqual(classifyXPostSnapshot(post()), { state: "captured" });
+ assert.deepEqual(classifyXPostSnapshot(post(true)), { state: "captured", sensitive: true });
+});
+
+test("classify: a login redirect or a logged-out page is a login wall that stops the run", () => {
+ for (const s of [page("", "/i/flow/login"), page("", "/login"), page("Don't miss what's happening. Log in Sign up")]) {
+ const v = classifyXPostSnapshot(s);
+ assert.equal(v.state, "login-wall", JSON.stringify(s));
+ assert.ok(v.stop);
+ }
+});
+
+test("classify: deleted and walled posts are told apart, and neither stops the run", () => {
+ for (const text of [
+ "This Post was deleted by the Post author. Learn more",
+ "Hmm...this page doesn’t exist. Try searching for something else.",
+ ]) {
+ assert.deepEqual(classifyXPostSnapshot(page(text)), { state: "deleted" }, text);
+ }
+ for (const text of [
+ "This Post is from an account that no longer exists. Learn more",
+ "This Post is from a suspended account. Learn more",
+ "You’re unable to view this Post because this account owner limits who can view their Posts.",
+ "This Post is unavailable.",
+ ]) {
+ assert.deepEqual(classifyXPostSnapshot(page(text)), { state: "unavailable" }, text);
+ }
+});
+
+test("classify: X refusing pages stops the run; an age wall and an empty page are errors that do not", () => {
+ const refused = classifyXPostSnapshot(page("Something went wrong. Try reloading. Retry"));
+ assert.equal(refused.state, "error");
+ assert.ok(refused.stop);
+ const age = classifyXPostSnapshot(page("Age-restricted adult content."));
+ assert.equal(age.state, "error");
+ assert.equal(age.stop, undefined);
+ const blank = classifyXPostSnapshot(page(""));
+ assert.equal(blank.state, "error");
+ assert.equal(blank.stop, undefined);
+});
+
+// --- the shot ----------------------------------------------------------------
+
+type FakePage = PageLike & { calls: string[]; shots: unknown[] };
+
+// A page whose snapshot evaluate returns `snaps` in turn (the last repeats);
+// the sensitive-cover click flips a sensitive article to an open one.
+function fakePage(snaps: XPostSnapshot[], opts: { gotoFails?: boolean } = {}): FakePage {
+ let i = 0;
+ const calls: string[] = [];
+ const shots: unknown[] = [];
+ return {
+ calls,
+ shots,
+ goto: async (url: string) => {
+ calls.push(`goto ${url}`);
+ if (opts.gotoFails) throw new Error("net::ERR_TIMED_OUT\nat goto");
+ },
+ waitForSelector: async () => undefined,
+ waitForTimeout: async () => undefined,
+ on: () => {},
+ evaluate: async (script: string) => {
+ if (script.includes("b.click()")) {
+ calls.push("open-sensitive");
+ return 1;
+ }
+ if (script.includes("imgs.length")) return 0;
+ calls.push("snapshot");
+ const s = snaps[Math.min(i, snaps.length - 1)];
+ i++;
+ return s;
+ },
+ screenshot: async (o?: unknown) => {
+ shots.push(o);
+ return PNG;
+ },
+ };
+}
+
+test("shot: the post's own box, clipped from a full-page PNG, hashed beside it", async () => {
+ const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1");
+ const p = fakePage([post()]);
+ const res = await shootXPost(p, "1", dir);
+ assert.equal(res.state, "captured");
+ assert.equal(p.calls[0], "goto https://x.com/i/status/1");
+ assert.deepEqual(p.shots, [
+ { type: "png", fullPage: true, clip: { x: 10, y: 120, width: 599, height: 301 } },
+ ]);
+ assert.deepEqual(new Uint8Array(await readFile(path.join(dir, SHOT_FILENAME))), PNG);
+ assert.deepEqual(res.shot, {
+ name: SHOT_FILENAME,
+ bytes: PNG.length,
+ sha256: sha(PNG),
+ url: "https://x.com/i/status/1",
+ });
+});
+
+test("shot: a sensitive cover is opened before the shot", async () => {
+ const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1");
+ const p = fakePage([post(true), post(false)]);
+ const res = await shootXPost(p, "1", dir);
+ assert.equal(res.state, "captured");
+ assert.equal(res.sensitive, true);
+ assert.ok(p.calls.includes("open-sensitive"));
+ assert.equal(p.shots.length, 1);
+});
+
+test("shot: a deleted post, a login wall or a failed load writes no shot", async () => {
+ for (const [p, state] of [
+ [fakePage([page("This post was deleted by the post author.")]), "deleted"],
+ [fakePage([page("", "/i/flow/login")]), "login-wall"],
+ [fakePage([post()], { gotoFails: true }), "error"],
+ ] as const) {
+ const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1");
+ const res = await shootXPost(p, "1", dir);
+ assert.equal(res.state, state);
+ assert.equal(res.shot, undefined);
+ assert.equal(p.shots.length, 0);
+ }
+});
+
+// --- the run -----------------------------------------------------------------
+
+type Harness = {
+ deps: XCaptureDeps;
+ pauses: number[];
+ opened: number;
+ closed: number;
+ downloads: string[];
+};
+
+// One snapshot per post id; the downloader writes `<id>_1.jpg` unless the id
+// is in `noMedia`.
+function harness(
+ snapsById: Record<string, XPostSnapshot[]>,
+ opts: { noMedia?: string[]; mediaFails?: { id: string; needsCookies?: boolean } } = {},
+): Harness {
+ const h: Harness = { deps: undefined as never, pauses: [], opened: 0, closed: 0, downloads: [] };
+ let current = "";
+ const pages = Object.fromEntries(Object.entries(snapsById).map(([id, s]) => [id, fakePage(s)]));
+ const routed: PageLike = {
+ goto: async (url: string, o?: unknown) => {
+ current = url.split("/").pop()!;
+ return pages[current].goto(url, o);
+ },
+ waitForSelector: async () => undefined,
+ waitForTimeout: async () => undefined,
+ on: () => {},
+ evaluate: (s: string) => pages[current].evaluate(s),
+ screenshot: (o) => pages[current].screenshot(o),
+ };
+ h.deps = {
+ openPage: async () => {
+ h.opened++;
+ return { page: routed, close: async () => void h.closed++ };
+ },
+ downloadMedia: async ({ id, dir }) => {
+ h.downloads.push(id);
+ if (opts.mediaFails?.id === id) {
+ return { ok: false, error: "gallery-dl exited 4", needsCookies: opts.mediaFails.needsCookies };
+ }
+ if (opts.noMedia?.includes(id)) return { ok: true, files: [] };
+ await writeFile(path.join(dir, `${id}_1.jpg`), `media of ${id}`);
+ return { ok: true, files: [{ name: `${id}_1.jpg`, url: `https://pbs.example/${id}.jpg` }] };
+ },
+ pauseMs: () => 5_000,
+ pause: async (ms) => void h.pauses.push(ms),
+ now: () => new Date("2026-02-03T04:05:06.000Z"),
+ };
+ return h;
+}
+
+async function input(ids: string[], over: Partial<PostCaptureInput> = {}): Promise<PostCaptureInput> {
+ return {
+ ids,
+ handle: "example_user",
+ outDir: await mkdtemp(path.join(os.tmpdir(), "posts-media-")),
+ signal: new AbortController().signal,
+ ...over,
+ };
+}
+
+test("run: shot then media per post, a paced gap before every contact after the first, a record per post", async () => {
+ const h = harness({ "11": [post()], "22": [post()] });
+ const inp = await input(["11", "22"]);
+ const res = await captureXPosts(inp, h.deps);
+ assert.equal(res.stoppedEarly, undefined);
+ assert.deepEqual(res.outcomes, [
+ { id: "11", state: "captured", availability: "available", files: 2 },
+ { id: "22", state: "captured", availability: "available", files: 2 },
+ ]);
+ // Four contacts (two pages, two downloads): three gaps.
+ assert.deepEqual(h.pauses, [5_000, 5_000, 5_000]);
+ assert.equal(h.opened, 1);
+ assert.equal(h.closed, 1);
+ const rec = await readPostCapture(path.join(inp.outDir, "11"));
+ assert.deepEqual(rec, {
+ version: 1,
+ id: "11",
+ url: "https://x.com/i/status/11",
+ capturedAt: "2026-02-03T04:05:06.000Z",
+ state: "captured",
+ shot: { name: SHOT_FILENAME, bytes: PNG.length, sha256: sha(PNG), url: "https://x.com/i/status/11" },
+ mediaState: "ok",
+ media: [
+ {
+ name: "11_1.jpg",
+ bytes: "media of 11".length,
+ sha256: sha("media of 11"),
+ url: "https://pbs.example/11.jpg",
+ },
+ ],
+ });
+});
+
+test("run: posts already captured are skipped unless forced; a missing half is all that is redone", async () => {
+ const h = harness({ "11": [post()], "22": [post()] });
+ const inp = await input(["11", "22"]);
+ await captureXPosts(inp, h.deps);
+ h.downloads.length = 0;
+
+ const again = harness({ "11": [post()], "22": [post()] });
+ const res = await captureXPosts(inp, again.deps);
+ assert.deepEqual(res.outcomes, []);
+ assert.equal(again.opened, 0, "no browser for a run with nothing to shoot");
+ assert.deepEqual(again.pauses, []);
+
+ // A post whose media failed earlier: only the media is fetched again.
+ const rec = (await readPostCapture(path.join(inp.outDir, "22")))!;
+ await writePostCapture(path.join(inp.outDir, "22"), { ...rec, mediaState: "error", media: [] });
+ const retry = harness({ "11": [post()], "22": [post()] });
+ const r2 = await captureXPosts(inp, retry.deps);
+ assert.equal(retry.opened, 0);
+ assert.deepEqual(retry.downloads, ["22"]);
+ assert.equal(r2.outcomes.length, 1);
+ assert.equal(r2.outcomes[0].availability, undefined, "a media-only pass says nothing about liveness");
+ const kept = (await readPostCapture(path.join(inp.outDir, "22")))!;
+ assert.equal(kept.shot?.sha256, sha(PNG), "the shot on disk is kept");
+
+ const forced = harness({ "11": [post()], "22": [post()] });
+ const r3 = await captureXPosts({ ...inp, force: true }, forced.deps);
+ assert.equal(r3.outcomes.length, 2);
+ assert.deepEqual(forced.downloads, ["11", "22"]);
+});
+
+test("run: a deleted or walled post is recorded, gets no download, and the run goes on", async () => {
+ const h = harness({
+ "11": [page("This post was deleted by the post author.")],
+ "22": [page("These posts are protected.")],
+ "33": [post()],
+ }, { noMedia: ["33"] });
+ const inp = await input(["11", "22", "33"]);
+ const res = await captureXPosts(inp, h.deps);
+ assert.deepEqual(
+ res.outcomes.map((o) => [o.id, o.state, o.availability, o.files]),
+ [
+ ["11", "deleted", "deleted", 0],
+ ["22", "unavailable", "account_unavailable", 0],
+ ["33", "captured", "available", 1],
+ ],
+ );
+ assert.deepEqual(h.downloads, ["33"]);
+ const deleted = (await readPostCapture(path.join(inp.outDir, "11")))!;
+ assert.equal(deleted.state, "deleted");
+ assert.equal(deleted.shot, undefined);
+ assert.equal(deleted.mediaState, "skipped");
+ assert.equal((await readPostCapture(path.join(inp.outDir, "33")))!.mediaState, "none");
+ // A deleted post is settled: the next run does not ask X again.
+ const next = harness({ "11": [post()], "22": [post()], "33": [post()] });
+ const r2 = await captureXPosts(inp, next.deps);
+ assert.deepEqual(r2.outcomes.map((o) => o.id), ["22"]);
+});
+
+test("run: a login wall stops the run at that post — recorded, no availability verdict, the rest untouched", async () => {
+ const h = harness({ "11": [post()], "22": [page("", "/i/flow/login")], "33": [post()] });
+ const inp = await input(["11", "22", "33"]);
+ const res = await captureXPosts(inp, h.deps);
+ assert.equal(res.needsCookies, true);
+ assert.match(res.stoppedEarly ?? "", /Stopped at 22/);
+ assert.deepEqual(res.outcomes.map((o) => [o.id, o.state, o.availability]), [
+ ["11", "captured", "available"],
+ ["22", "login-wall", undefined],
+ ]);
+ assert.equal((await readPostCapture(path.join(inp.outDir, "22")))!.state, "login-wall");
+ assert.equal(await readPostCapture(path.join(inp.outDir, "33")), null);
+ assert.equal(h.closed, 1, "the browser is closed on the way out");
+});
+
+test("run: a media download refused for want of a login stops the run with needsCookies", async () => {
+ const h = harness({ "11": [post()], "22": [post()] }, { mediaFails: { id: "11", needsCookies: true } });
+ const res = await captureXPosts(await input(["11", "22"]), h.deps);
+ assert.equal(res.needsCookies, true);
+ assert.deepEqual(res.outcomes.map((o) => o.id), ["11"]);
+ assert.equal(res.outcomes[0].state, "captured", "the shot stands");
+});
+
+test("run: three failures in a row stop it rather than paging through the rest", async () => {
+ const h = harness({ "1": [page("")], "2": [page("")], "3": [page("")], "4": [post()] });
+ const res = await captureXPosts(await input(["1", "2", "3", "4"]), h.deps);
+ assert.deepEqual(res.outcomes.map((o) => o.state), ["error", "error", "error"]);
+ assert.match(res.stoppedEarly ?? "", /3 posts in a row failed/);
+});
+
+test("run: a cancel or a drain stops between posts", async () => {
+ const ac = new AbortController();
+ const h = harness({ "11": [post()], "22": [post()] });
+ const inner = h.deps.downloadMedia;
+ h.deps.downloadMedia = async (a) => {
+ const r = await inner(a);
+ ac.abort();
+ return r;
+ };
+ const res = await captureXPosts(await input(["11", "22"], { signal: ac.signal }), h.deps);
+ assert.deepEqual(res.outcomes.map((o) => o.id), ["11"]);
+ assert.match(res.stoppedEarly ?? "", /Cancelled/);
+
+ const drain = new AbortController();
+ drain.abort();
+ const d = harness({ "11": [post()] });
+ const r2 = await captureXPosts(await input(["11"], { drain: drain.signal }), d.deps);
+ assert.deepEqual(r2.outcomes, []);
+ assert.match(r2.stoppedEarly ?? "", /Drained/);
+ assert.equal(d.opened, 0);
+});
+
+test("run: media only, or shots only, does only that half", async () => {
+ const h = harness({ "11": [post()] });
+ const inp = await input(["11"], { shots: false });
+ const res = await captureXPosts(inp, h.deps);
+ assert.equal(h.opened, 0);
+ assert.deepEqual(h.downloads, ["11"]);
+ assert.deepEqual(res.outcomes, [{ id: "11", state: "captured", files: 1 }]);
+
+ const s = harness({ "11": [post()] });
+ const r2 = await captureXPosts(await input(["11"], { media: false }), s.deps);
+ assert.deepEqual(s.downloads, []);
+ assert.deepEqual(r2.outcomes, [{ id: "11", state: "captured", availability: "available", files: 1 }]);
+});
+
+// --- the real download path, over a fake gallery-dl -------------------------
+
+test("media download: gallery-dl's tagged lines and the files it wrote, nothing from the network", async () => {
+ const root = await mkdtemp(path.join(os.tmpdir(), "xmedia-"));
+ const bin = path.join(root, "fake-gallery-dl.mjs");
+ const argsLog = path.join(root, "argv.json");
+ await writeFile(
+ bin,
+ `#!/usr/bin/env node
+import { writeFileSync } from "node:fs";
+import path from "node:path";
+const args = process.argv.slice(2);
+writeFileSync(process.env.FAKE_ARGS_LOG, JSON.stringify(args));
+const dir = args[args.indexOf("-D") + 1];
+const tag = "ARCHILYZER-CAPTURED";
+if (process.env.FAKE_MODE === "auth") {
+ process.stderr.write("[twitter][error] 401 Unauthorized\\n");
+ process.exit(4);
+}
+const file = path.join(dir, "77_1.jpg");
+writeFileSync(file, "pixels");
+process.stdout.write(file + "\\n");
+process.stdout.write(tag + "\\thttps://pbs.example/77.jpg\\t" + file + "\\n");
+`,
+ );
+ await chmod(bin, 0o755);
+ // Paths are read (and cached) on first use: point them at the temp root
+ // before anything asks, so no real corpus is in reach.
+ process.env.TRANSCRIPTS_DIR = path.join(root, "transcripts");
+ process.env.GALLERY_DL_BIN = bin;
+ process.env.FAKE_ARGS_LOG = argsLog;
+ const { downloadXPostMedia } = await import("./xGalleryDlFetcher");
+ const dir = path.join(root, "posts-media", "77");
+ await mkdir(dir, { recursive: true });
+ const ok = await downloadXPostMedia({
+ id: "77",
+ dir,
+ login: { cookieFile: "/jar.txt" },
+ signal: new AbortController().signal,
+ });
+ assert.deepEqual(ok, { ok: true, files: [{ name: "77_1.jpg", url: "https://pbs.example/77.jpg" }] });
+ const argv = JSON.parse(await readFile(argsLog, "utf8")) as string[];
+ assert.ok(!argv.includes("--no-download"));
+ assert.equal(argv[argv.indexOf("--cookies") + 1], "/jar.txt");
+
+ process.env.FAKE_MODE = "auth";
+ try {
+ const refused = await downloadXPostMedia({
+ id: "77",
+ dir,
+ login: {},
+ signal: new AbortController().signal,
+ });
+ assert.equal(refused.ok, false);
+ assert.equal(!refused.ok && refused.needsCookies, true);
+ } finally {
+ delete process.env.FAKE_MODE;
+ }
+});
diff --git a/common/social/xPostCapture.ts b/common/social/xPostCapture.ts
@@ -0,0 +1,402 @@
+// Capturing specific X posts: a screenshot of each post as X renders it,
+// through the logged-in session profile, and its attached media, downloaded by
+// gallery-dl (xGalleryDlFetcher.ts owns that half and wires both into
+// `captureByIds`).
+//
+// PACED LIKE EVERY OTHER X READ. A capture is a page load and a gallery-dl run
+// per post, and each is a contact with X: the run waits a random 4–10 s before
+// every contact after its first (gallery-dl then paces its own requests the
+// same way). A page that says X is refusing — "Something went wrong" — or a
+// login wall stops the run there: the next post would meet the same answer,
+// and asking again at once is exactly the burst the pacing exists to avoid.
+// The ids not reached are left for a later run.
+//
+// What a page can say instead of the post, each recorded, none retried in the
+// run that met it:
+// - a login wall (a redirect to the login flow, or a logged-out page with no
+// post): the session's state, not the post's — the run stops, needsCookies;
+// - a sensitive-media interstitial: opened ("Show"), then shot, and the
+// record says it was there;
+// - "this post was deleted" / "this page doesn't exist": deleted;
+// - a protected, suspended or vanished account, or a withheld post:
+// unavailable.
+//
+// NOT VERIFIED AGAINST LIVE X. The page markers below are X's as of this
+// writing and are tested against recorded snapshots, never x.com; the first
+// real run is the check.
+
+import { mkdir, writeFile } from "node:fs/promises";
+import path from "node:path";
+import type { PageLike } from "./playwrightRuntime";
+import type {
+ PostCaptureInput,
+ PostCaptureOutcome,
+ PostCaptureResult,
+} from "./fetchers";
+import {
+ captureAvailability,
+ captureWork,
+ describeCapturedFile,
+ postCaptureDir,
+ readPostCapture,
+ SHOT_FILENAME,
+ writePostCapture,
+ type CapturedFile,
+ type CaptureMediaState,
+ type PostCaptureRecord,
+ type PostCaptureState,
+} from "./postCapture";
+
+export function xStatusUrl(id: string): string {
+ return `https://x.com/i/status/${id}`;
+}
+
+// What the page showed, read in one evaluate. `article` is the post itself
+// (its box in document coordinates, for the clip), when it rendered.
+export type XPostSnapshot = {
+ path: string;
+ text: string;
+ article: null | {
+ rect: { x: number; y: number; width: number; height: number };
+ // A "Show" / "View" button inside the post: a sensitive-media cover.
+ sensitive: boolean;
+ };
+};
+
+// The post's own article: the one whose timestamp links to this id (a reply's
+// parents render above it as articles too), else X's focal article.
+const SNAPSHOT_SCRIPT = (id: string) => `(() => {
+ const id = ${JSON.stringify(id)};
+ const articles = Array.from(document.querySelectorAll('article[data-testid="tweet"]'));
+ const own = articles.find((a) =>
+ Array.from(a.querySelectorAll('a[href*="/status/"]')).some((l) => {
+ const m = /\\/status\\/(\\d+)/.exec(l.getAttribute("href") || "");
+ return m && m[1] === id && l.querySelector("time");
+ }),
+ ) || articles.find((a) => a.getAttribute("tabindex") === "-1") || null;
+ const main = document.querySelector('[data-testid="primaryColumn"]') || document.body;
+ const text = ((main && main.innerText) || "").slice(0, 4000);
+ let article = null;
+ if (own) {
+ own.setAttribute("data-archilyzer-capture", "");
+ const r = own.getBoundingClientRect();
+ const sensitive = Array.from(own.querySelectorAll('button, [role="button"]')).some(
+ (b) => /^(show|view)$/i.test((b.innerText || "").trim()),
+ );
+ article = {
+ rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height },
+ sensitive,
+ };
+ }
+ return { path: location.pathname, text, article };
+})()`;
+
+// Open a sensitive-media cover inside the post.
+const OPEN_SENSITIVE_SCRIPT = `(() => {
+ const own = document.querySelector('[data-archilyzer-capture]');
+ if (!own) return 0;
+ let n = 0;
+ for (const b of own.querySelectorAll('button, [role="button"]')) {
+ if (/^(show|view)$/i.test((b.innerText || "").trim())) { b.click(); n++; }
+ }
+ return n;
+})()`;
+
+// Let the post's images finish loading (each capped), so the shot is not of
+// grey boxes.
+const IMAGES_LOADED_SCRIPT = `(() => {
+ const own = document.querySelector('[data-archilyzer-capture]') || document;
+ const imgs = Array.from(own.querySelectorAll("img")).filter((i) => !i.complete);
+ return Promise.all(imgs.map((i) => new Promise((r) => {
+ i.addEventListener("load", r, { once: true });
+ i.addEventListener("error", r, { once: true });
+ setTimeout(r, 5000);
+ }))).then(() => imgs.length);
+})()`;
+
+const DELETED_TEXT = [
+ /this (post|tweet) was deleted/i,
+ /this page doesn.t exist/i,
+];
+const UNAVAILABLE_TEXT = [
+ /account (that )?no longer exists/i,
+ /suspended account/i,
+ /account (is )?suspended/i,
+ /these (posts|tweets) are protected/i,
+ /limits who can view their (posts|tweets)/i,
+ /withheld in/i,
+ /this (post|tweet) is unavailable/i,
+];
+const AGE_WALL_TEXT = /age-restricted/i;
+const REFUSED_TEXT = /something went wrong\. try reloading|rate limit exceeded/i;
+const LOGGED_OUT_TEXT = /(log in|sign in|sign up)/i;
+
+export type XPostVerdict = {
+ state: PostCaptureState;
+ sensitive?: boolean;
+ // Set when the run must stop here: the next post would meet the same page.
+ stop?: string;
+ error?: string;
+};
+
+// What a snapshot means. Pure, so every marker is testable without a browser.
+export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict {
+ if (/^\/(i\/flow\/login|login|i\/flow\/signup)\b/.test(s.path)) {
+ return {
+ state: "login-wall",
+ stop: "X asked to log in — the session profile is not logged in.",
+ };
+ }
+ if (s.article) {
+ return { state: "captured", ...(s.article.sensitive ? { sensitive: true } : {}) };
+ }
+ if (REFUSED_TEXT.test(s.text)) {
+ return {
+ state: "error",
+ error: "X answered “Something went wrong” instead of the post.",
+ stop: "X is refusing pages right now; stopping rather than asking again.",
+ };
+ }
+ if (DELETED_TEXT.some((re) => re.test(s.text))) return { state: "deleted" };
+ if (UNAVAILABLE_TEXT.some((re) => re.test(s.text))) return { state: "unavailable" };
+ if (AGE_WALL_TEXT.test(s.text)) {
+ return {
+ state: "error",
+ error: "X shows this post only to an age-verified session.",
+ };
+ }
+ if (LOGGED_OUT_TEXT.test(s.text)) {
+ return {
+ state: "login-wall",
+ stop: "X showed its logged-out page instead of the post.",
+ };
+ }
+ return { state: "error", error: "The post did not render." };
+}
+
+export type XShotResult = XPostVerdict & { shot?: CapturedFile };
+
+// One post's screenshot: load, read the page, open a sensitive cover, shoot
+// the post's own article. Writes `shot.png` into `dir` only for a post that
+// rendered.
+export async function shootXPost(
+ page: PageLike,
+ id: string,
+ dir: string,
+ onLog?: (line: string) => void,
+): Promise<XShotResult> {
+ const url = xStatusUrl(id);
+ try {
+ await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 });
+ } catch (err) {
+ return { state: "error", error: `Could not load ${url}: ${firstLine(err)}` };
+ }
+ // The post, or whatever X shows instead — which has no stable marker, so a
+ // missing article is waited out and the page read anyway.
+ await page
+ .waitForSelector('article[data-testid="tweet"]', { timeout: 20_000 })
+ .catch(() => {});
+ await page.waitForTimeout(1_500);
+ let snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot;
+ const verdict = classifyXPostSnapshot(snap);
+ if (verdict.state !== "captured") return verdict;
+
+ if (verdict.sensitive) {
+ onLog?.(`${id}: a sensitive-media cover — opening it before the shot.`);
+ await page.evaluate(OPEN_SENSITIVE_SCRIPT);
+ await page.waitForTimeout(1_500);
+ snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot;
+ if (!snap.article) {
+ return { state: "error", error: "The post vanished after its sensitive-media cover was opened." };
+ }
+ }
+ await page.evaluate(IMAGES_LOADED_SCRIPT).catch(() => {});
+ // Re-read the box: images that loaded may have grown it.
+ snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot;
+ const rect = snap.article?.rect;
+ if (!rect || rect.width < 1 || rect.height < 1) {
+ return { state: "error", error: "The post rendered with no size to shoot." };
+ }
+ const clip = {
+ x: Math.max(0, Math.floor(rect.x)),
+ y: Math.max(0, Math.floor(rect.y)),
+ width: Math.ceil(rect.width),
+ height: Math.ceil(rect.height),
+ };
+ let png: Uint8Array;
+ try {
+ // fullPage, so a post taller than the window is shot whole; the clip is in
+ // document coordinates, which is what the snapshot measured.
+ png = await page.screenshot({ type: "png", fullPage: true, clip });
+ } catch (err) {
+ return { state: "error", error: `The screenshot failed: ${firstLine(err)}` };
+ }
+ await mkdir(dir, { recursive: true });
+ await writeFile(path.join(dir, SHOT_FILENAME), png);
+ const shot = await describeCapturedFile(dir, SHOT_FILENAME, url);
+ return { ...verdict, shot };
+}
+
+export type MediaDownloadResult =
+ | { ok: true; files: { name: string; url?: string }[] }
+ | { ok: false; error: string; needsCookies?: boolean };
+
+export type XCaptureDeps = {
+ // A page in the logged-in profile. Opened on the first shot, closed at the
+ // end of the run.
+ openPage: () => Promise<{ page: PageLike; close: () => Promise<void> }>;
+ // One post's media into `dir` (gallery-dl in production).
+ downloadMedia: (args: {
+ id: string;
+ dir: string;
+ signal: AbortSignal;
+ onLog?: (line: string) => void;
+ }) => Promise<MediaDownloadResult>;
+ // The gap before each contact with X after the first.
+ pauseMs: () => number;
+ pause: (ms: number, signal: AbortSignal) => Promise<void>;
+ now?: () => Date;
+};
+
+// A run's errors in a row that stop it: a host whose network or browser is
+// failing every post should not page through the rest of the list.
+const STOP_AFTER_ERRORS = 3;
+
+// The capture loop: per id, what is owed (captureWork), the shot, the media,
+// the record. Every post's record is written as soon as that post is done, so
+// a cancel or a crash loses at most the post in hand.
+export async function captureXPosts(
+ input: PostCaptureInput,
+ deps: XCaptureDeps,
+): Promise<PostCaptureResult> {
+ const { signal, onLog } = input;
+ const wanted = {
+ shots: input.shots ?? true,
+ media: input.media ?? true,
+ force: input.force ?? false,
+ };
+ const now = deps.now ?? (() => new Date());
+ const outcomes: PostCaptureOutcome[] = [];
+ let contacts = 0;
+ let errorsInARow = 0;
+ let browser: { page: PageLike; close: () => Promise<void> } | undefined;
+
+ const contact = async () => {
+ if (contacts++ > 0) await deps.pause(deps.pauseMs(), signal);
+ };
+ const stopped = (why: string, extra: Partial<PostCaptureResult> = {}) => {
+ onLog?.(why);
+ return { outcomes, stoppedEarly: why, ...extra };
+ };
+
+ try {
+ for (const [i, id] of input.ids.entries()) {
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ if (input.drain?.aborted) return stopped("Drained; the rest are left for a later run.");
+ const dir = postCaptureDir(input.outDir, id);
+ const existing = await readPostCapture(dir);
+ const work = captureWork(existing, wanted);
+ if (!work.shot && !work.media) {
+ onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`);
+ continue;
+ }
+ onLog?.(`[${i + 1}/${input.ids.length}] ${id}`);
+
+ let state: PostCaptureState | undefined = work.shot ? undefined : existing?.state;
+ let sensitive = work.shot ? undefined : existing?.sensitive;
+ let shot = work.shot ? undefined : existing?.shot;
+ let error: string | undefined;
+ let stop: string | undefined;
+
+ if (work.shot) {
+ await contact();
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ browser ??= await deps.openPage();
+ const res = await shootXPost(browser.page, id, dir, onLog);
+ state = res.state;
+ sensitive = res.sensitive;
+ shot = res.shot;
+ error = res.error;
+ stop = res.stop;
+ }
+
+ let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped");
+ let media: CapturedFile[] = work.media ? [] : (existing?.media ?? []);
+ let needsCookies = false;
+ // Only a post that rendered (or was not shot this run) is worth a
+ // download: X has nothing to give for a deleted or walled one.
+ const postIsThere = state === undefined || state === "captured";
+ if (work.media && postIsThere && !stop) {
+ await contact();
+ if (signal.aborted) return stopped("Cancelled; the rest are left for a later run.");
+ await mkdir(dir, { recursive: true });
+ const got = await deps.downloadMedia({ id, dir, signal, onLog });
+ if (got.ok) {
+ media = [];
+ for (const f of got.files) media.push(await describeCapturedFile(dir, f.name, f.url));
+ mediaState = media.length > 0 ? "ok" : "none";
+ state ??= "captured";
+ } else {
+ mediaState = "error";
+ error = error ? `${error}; ${got.error}` : got.error;
+ state ??= "error";
+ if (got.needsCookies) {
+ needsCookies = true;
+ stop = "gallery-dl could not log in to X for the media.";
+ }
+ }
+ }
+
+ const finalState: PostCaptureState = state ?? "error";
+ const record: PostCaptureRecord = {
+ version: 1,
+ id,
+ url: xStatusUrl(id),
+ capturedAt: now().toISOString(),
+ state: finalState,
+ ...(sensitive ? { sensitive: true } : {}),
+ ...(shot ? { shot } : {}),
+ mediaState,
+ media,
+ ...(error ? { error } : {}),
+ };
+ await writePostCapture(dir, record);
+ outcomes.push({
+ id,
+ state: finalState,
+ // Only the page says whether the post is up: a media-only run has no
+ // verdict to record.
+ ...(work.shot && captureAvailability(finalState)
+ ? { availability: captureAvailability(finalState) }
+ : {}),
+ files: (shot ? 1 : 0) + media.length,
+ ...(error ? { error } : {}),
+ });
+ onLog?.(
+ `${id}: ${finalState}${sensitive ? " (behind a sensitive-media cover)" : ""}` +
+ (shot ? ", shot" : "") +
+ (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") +
+ (error ? ` — ${error}` : ""),
+ );
+
+ if (stop) {
+ return stopped(`${stop} Stopped at ${id}; the rest are left for a later run.`, {
+ needsCookies: needsCookies || finalState === "login-wall",
+ });
+ }
+ errorsInARow = finalState === "error" ? errorsInARow + 1 : 0;
+ if (errorsInARow >= STOP_AFTER_ERRORS) {
+ return stopped(
+ `${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`,
+ );
+ }
+ }
+ return { outcomes };
+ } finally {
+ await browser?.close().catch(() => {});
+ }
+}
+
+function firstLine(err: unknown): string {
+ return ((err as Error)?.message ?? String(err)).split("\n")[0];
+}
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Capture specific X posts: a screenshot of each, and its attached media.** `pnpm ops capture-posts --json '{"slug":"<channel>","ids":["<post id>", …]}'` shoots each post as X shows it, through the connected X profile, and downloads its pictures and videos with gallery-dl, into the channel's `posts-media/<post id>/` beside a `capture.json` that records when, from which URLs, and each file's size and SHA-256. Every id must already be in the channel's posts archive; one that is not is refused by name and nothing runs. `"shots": false` or `"media": false` skips that half, and posts already captured are skipped unless `"force": true`. The job runs on the X queue with a post fetch, so the two never run at once, and waits a random 4–10 seconds before each request to X, as fetches do. A deleted post, or one behind its account's wall (protected, suspended, gone), is recorded as such in the channel's deleted-post record; a post behind a sensitive-media warning is opened and shot. If X asks to log in, or answers "Something went wrong", the job stops at that post and leaves the rest for a later run. Captures are never published: the export does not read them.
- **X posts are fetched more slowly, with random gaps.** Every read of X now waits a random 4 to 10 seconds before each request to X, where it used to page as fast as X answered, and always waits out a rate limit rather than pushing through. When fetching older posts, the pause between one three-month window and the next is a random 45 to 120 seconds instead of a fixed 15. A deep walk of an account's history takes longer; a routine fetch of new posts takes a few seconds more.
- **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched.
- **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history.
diff --git a/editor/app/api/ops/capture-posts/route.ts b/editor/app/api/ops/capture-posts/route.ts
@@ -0,0 +1,41 @@
+import { capturePostsAction } from "../../../channels/[slug]/socialActions";
+import {
+ jobResponse,
+ ops,
+ optBool,
+ optString,
+ reqSlug,
+ reqStringArray,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, ids, shots?, media?, force?, queueKey? } -> { ok: true, jobId }
+//
+// Capture specific archived posts of a social channel: a screenshot of each
+// (`shots`, default true) and its attached media (`media`, default true), into
+// the channel's posts-media/<id>/. Posts already captured are skipped unless
+// `force`. The job runs on the platform's queue, as a post fetch does.
+//
+// Every refusal is the action's own sentence: both halves off, a channel that
+// is not a social one, a fetcher that cannot capture, an id not in the
+// channel's posts archive.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["slug", "ids", "shots", "media", "force", "queueKey"],
+ async (body) => {
+ const slug = reqSlug(body, "slug");
+ return jobResponse(
+ await capturePostsAction(
+ slug,
+ reqStringArray(body, "ids"),
+ optString(body, "queueKey"),
+ optBool(body, "shots"),
+ optBool(body, "media"),
+ optBool(body, "force"),
+ ),
+ );
+ },
+ );
+}
diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts
@@ -6,6 +6,7 @@
// which applies to a post fetch. A social channel has exactly two stages
// (Fetch → Index), and this file owns the first.
+import path from "node:path";
import { revalidatePath } from "next/cache";
import { safeRevalidate } from "../../lib/safeRevalidate";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
@@ -30,6 +31,14 @@ import {
type CheckPostAvailabilityMode,
} from "yt-dlp-transcript-common/controller/checkPostAvailability";
import {
+ capturePosts,
+ capturePostsProblem,
+ NOTHING_TO_CAPTURE,
+ strayCaptureIds,
+ strayIdsRefusal,
+} from "yt-dlp-transcript-common/controller/capturePosts";
+import { readSeenPostIds } from "yt-dlp-transcript-common/lib/posts-server";
+import {
getSocialFetcher,
listSocialFetchers,
resolveSocialFetcher,
@@ -196,3 +205,67 @@ export async function fetchPostsAction(
},
});
}
+
+// A screenshot and the attached media of specific archived posts, into the
+// channel's `posts-media/<id>/`. On the platform queue, as a fetch is, so the
+// two never run against the same source at once. Refused HERE, before a job
+// exists: nothing asked for, no ids, a channel that is not social, a fetcher
+// that cannot capture, or an id that is not in the channel's posts archive
+// (named). Posts already captured are skipped unless `force`.
+export async function capturePostsAction(
+ slug: string,
+ ids: string[],
+ queueKey?: string,
+ shots?: boolean,
+ media?: boolean,
+ force?: boolean,
+): Promise<StreamActionResult> {
+ if (shots === false && media === false) return { ok: false, error: NOTHING_TO_CAPTURE };
+ const wanted = [...new Set(ids)];
+ if (wanted.length === 0) return { ok: false, error: "No post ids to capture." };
+ const paths = getPaths();
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `No such channel: ${slug}` };
+ if (!isSocialChannel(config)) {
+ return { ok: false, error: `${slug} is not a social channel.` };
+ }
+ await registerBuiltinSocialFetchers();
+ const problem = capturePostsProblem(
+ resolveSocialFetcher(config.postFetcher, config.url),
+ );
+ if (problem) return { ok: false, error: problem };
+ const stray = strayCaptureIds(
+ wanted,
+ await readSeenPostIds(path.join(paths.channelsDir, slug)),
+ );
+ if (stray) return { ok: false, error: strayIdsRefusal(slug, stray) };
+
+ const key = resolveQueueKey(downloadQueueKey(config), queueKey);
+ return runManagedFunction({
+ kind: "capture-posts",
+ queueKey: key,
+ paths,
+ channelSlug: slug,
+ spec: {
+ kind: "capture-posts",
+ slug,
+ params: { queueKey, ids: wanted, shots, media, force },
+ },
+ fn: async (onLog, signal, _progress, ctx) => {
+ const result = await capturePosts({
+ paths,
+ slug,
+ settings: getSettings(),
+ ids: wanted,
+ shots,
+ media,
+ force,
+ onLog,
+ signal,
+ drain: ctx.drainSignal,
+ });
+ safeRevalidate([`/channels/${slug}`]);
+ if (!result.ok) throw new Error(result.error ?? "Post capture failed");
+ },
+ });
+}
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -40,6 +40,7 @@ import {
import { backfillChannelAction } from "../channels/[slug]/backfillActions";
import { persistKeptAction } from "../channels/[slug]/persistActions";
import {
+ capturePostsAction,
checkPostAvailabilityAction,
fetchPostsAction,
} from "../channels/[slug]/socialActions";
@@ -255,6 +256,19 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
str(p.floor),
);
},
+ // The ids are the spec's own (a capture is OF specific posts, unlike a
+ // bucket); the re-run skips whatever the first run already captured.
+ "capture-posts": (spec) => {
+ const { p, queueKey } = params(spec);
+ return capturePostsAction(
+ spec.slug,
+ strings(p.ids) ?? [],
+ queueKey,
+ bool(p.shots),
+ bool(p.media),
+ bool(p.force),
+ );
+ },
"download-missing-subs": (spec) => {
const { p, queueKey } = params(spec);
return downloadMissingSubsAction(spec.slug, queueKey, bool(p.abortOnError));
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -193,6 +193,7 @@ test("a traversing slug is refused at the door, on every route that takes one",
["relocate", { slugs: ["../../escape"], root: "/tmp/ops-api-never" }],
["relocate-back", { slugs: ["../../escape"] }],
["fetch-posts", { slug: "../../escape", older: true }],
+ ["capture-posts", { slug: "../../escape", ids: ["1"] }],
];
for (const [action, data] of cases) {
const { status, body } = await ops(request, action, data);
@@ -335,6 +336,68 @@ test("fetch-posts refuses a channel that is not social, full with older, and an
expect(await listJobIds()).toEqual(before);
});
+test("capture-posts refuses what it cannot capture, and an id not in the archive — before any job", async ({
+ request,
+}) => {
+ await resetData("title-filter-channel");
+ await settings();
+ // Nothing below reaches X: every case is refused before a job exists.
+ await writeChannelConfig("example-bsky", {
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "bluesky",
+ postFetcher: "bluesky-atproto",
+ socialHandle: "example.bsky.social",
+ name: "Example (Bluesky)",
+ url: "https://bsky.app/profile/example.bsky.social",
+ });
+ await writeChannelConfig("example-x", {
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "twitter",
+ postFetcher: "x-gallery-dl",
+ socialHandle: "example_user",
+ name: "Example (X)",
+ url: "https://x.com/example_user",
+ });
+ const before = await listJobIds();
+
+ const video = await ops(request, "capture-posts", { slug: "test-filter", ids: ["1"] });
+ expect(video.status).toBe(400);
+ expect(video.body.error).toBe("test-filter is not a social channel.");
+
+ const bsky = await ops(request, "capture-posts", { slug: "example-bsky", ids: ["1"] });
+ expect(bsky.status).toBe(400);
+ expect(bsky.body.error).toMatch(/cannot capture posts/);
+
+ const neither = await ops(request, "capture-posts", {
+ slug: "example-x",
+ ids: ["1"],
+ shots: false,
+ media: false,
+ });
+ expect(neither.status).toBe(400);
+ expect(neither.body.error).toMatch(/both the screenshot and the media are turned off/);
+
+ // The channel's posts archive is empty: every id is a stray, named.
+ const stray = await ops(request, "capture-posts", { slug: "example-x", ids: ["111", "222"] });
+ expect(stray.status).toBe(400);
+ expect(stray.body.error).toBe("2 id(s) not in example-x's posts archive: 111, 222");
+
+ // The body's shape.
+ const noIds = await ops(request, "capture-posts", { slug: "example-x" });
+ expect(noIds.status).toBe(400);
+ expect(noIds.body.error).toMatch(/"ids" is required/);
+ const badFlag = await ops(request, "capture-posts", { slug: "example-x", ids: ["1"], shots: "yes" });
+ expect(badFlag.status).toBe(400);
+ expect(badFlag.body.error).toMatch(/"shots" must be a boolean/);
+ const unknown = await ops(request, "capture-posts", { slug: "example-x", ids: ["1"], limit: 5 });
+ expect(unknown.status).toBe(400);
+ expect(unknown.body.error).toMatch(/unknown key\(s\): limit/);
+
+ expect(await listJobIds()).toEqual(before);
+});
+
test("channel-config round-trips a download filter and refuses a bad regex", async ({
page,
request,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -103,6 +103,8 @@ const ACTIONS = [
// A social channel's post fetch: new posts, the full re-walk ("full"), or
// the walk back below the oldest archived post ("older").
"fetch-posts",
+ // A screenshot and the attached media of specific archived posts.
+ "capture-posts",
"build-index",
"build-deploy",
"build-site",
@@ -326,6 +328,13 @@ export function usage() {
' the next run, down to "floor": "YYYY-MM-DD" when given. "limit": N caps',
' the posts one run reads. "full" and "older" together are refused.',
"",
+ 'capture-posts captures archived posts of a social channel (X): a',
+ ' screenshot of each through the connected X profile, and its attached',
+ ' media through gallery-dl, into the channel\'s posts-media/<id>/:',
+ ' {"slug", "ids": [...]}. Every id must be in the channel\'s posts archive.',
+ ' "shots": false or "media": false skips that half; posts already captured',
+ ' are skipped unless "force": true. Paced like a post fetch, on its queue.',
+ "",
'"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare',
" Pages PREVIEW instead of production: the same bundle goes to a branch",
" alias, https://<branch>.<project>.pages.dev, and the live site is left",
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -390,3 +390,18 @@ test("fetch-posts is a POST to its route, named in the usage", () => {
assert.match(usage(), /Actions:.*transcribe-bucket, fetch-posts/);
assert.match(usage(), /"older": true walks back from the oldest/);
});
+
+// A post capture: a POST to its route, the body passed through untouched — the
+// route and the action judge it (ids in the archive, both halves off).
+test("capture-posts is a POST to its route, named in the usage", () => {
+ const p = parseArgs([
+ "capture-posts",
+ "--json",
+ '{"slug":"example-x","ids":["123"],"media":false}',
+ ]);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/capture-posts");
+ assert.deepEqual(p.body, { slug: "example-x", ids: ["123"], media: false });
+ assert.match(usage(), /Actions:.*fetch-posts, capture-posts/);
+ assert.match(usage(), /Every id must be in the channel's posts archive/);
+});