commit 25f7d342f297ee59448a5e55520f282258677e82
parent c37177bed9fe09b13923b3194f3f4fc1df0f309e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 16:11:49 -0400
common: capturePosts controller and the capture-posts job kind
capturePosts refuses, before any capture, a request for neither half, no
ids, a channel that is not social, a fetcher that cannot capture, and ids
not in the channel's posts archive (named). What the pages said about each
post — deleted, or behind its account's wall — is merged into
posts-availability.json through the deleted-post sweep's merge; a login
wall records nothing. capture-posts runs on the platform queue, like
fetch-posts, and is drainable and replayable.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 364 insertions(+), 0 deletions(-)
diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts
@@ -0,0 +1,145 @@
+// capturePosts over a registered fake fetcher: what is refused before any
+// capture, and how the pages' verdicts land in the availability sidecar.
+//
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/capturePosts.test.ts
+
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, writeFile } from "node:fs/promises";
+import os from "node:os";
+import path from "node:path";
+
+const ROOT = await mkdtemp(path.join(os.tmpdir(), "captureposts-"));
+process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts");
+
+const { getPaths } = await import("../lib/paths");
+const { registerSocialFetcher } = await import("../social/fetchers");
+const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server");
+const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts");
+type Input = import("../social/fetchers").PostCaptureInput;
+type Result = import("../social/fetchers").PostCaptureResult;
+
+// A capturing fetcher whose answer each test sets.
+let calls: Input[] = [];
+let answer: Result = { outcomes: [] };
+registerSocialFetcher({
+ id: "test-capture",
+ label: "Test capture",
+ platform: "bluesky",
+ fields: {},
+ detect: () => false,
+ probe: async () => ({ ok: true }),
+ fetch: async () => ({ posts: [], complete: true }),
+ captureByIds: async (input) => {
+ calls.push(input);
+ return answer;
+ },
+});
+registerSocialFetcher({
+ id: "test-no-capture",
+ label: "No-capture fetcher",
+ platform: "bluesky",
+ fields: {},
+ detect: () => false,
+ probe: async () => ({ ok: true }),
+ fetch: async () => ({ posts: [], complete: true }),
+});
+
+const channelsDir = path.join(ROOT, "transcripts", "channels");
+
+async function makeChannel(slug: string, over: Record<string, unknown> = {}): Promise<string> {
+ const root = path.join(channelsDir, slug);
+ await mkdir(root, { recursive: true });
+ await writeFile(
+ path.join(root, "config.json"),
+ JSON.stringify({
+ handling: "transcribe",
+ sourceKind: "social",
+ platform: "bluesky",
+ postFetcher: "test-capture",
+ socialHandle: "example.bsky.social",
+ name: "Example",
+ url: "https://bsky.app/profile/example.bsky.social",
+ ...over,
+ }),
+ );
+ await writeFile(path.join(root, "posts-archive"), "bluesky aaa\nbluesky bbb\nbluesky ccc\n");
+ return root;
+}
+
+const run = (slug: string, ids: string[], over: Record<string, unknown> = {}) =>
+ capturePosts({ paths: getPaths(), slug, settings: {}, ids, ...over });
+
+test("refused before any capture: nothing asked for, no ids, not social, no capture, ids not archived", async () => {
+ await makeChannel("demo-social");
+ await makeChannel("demo-nocapture", { postFetcher: "test-no-capture" });
+ await mkdir(path.join(channelsDir, "demo-video"), { recursive: true });
+ await writeFile(
+ path.join(channelsDir, "demo-video", "config.json"),
+ JSON.stringify({ handling: "transcribe", name: "Video", url: "https://example.com/c" }),
+ );
+ calls = [];
+ assert.equal((await run("demo-social", ["aaa"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE);
+ assert.match((await run("demo-social", [])).error ?? "", /No post ids/);
+ assert.equal((await run("demo-video", ["aaa"])).error, "demo-video is not a social channel.");
+ assert.equal((await run("demo-nocapture", ["aaa"])).error, "No-capture fetcher cannot capture posts.");
+ assert.equal(
+ (await run("demo-social", ["aaa", "zzz", "yyy"])).error,
+ "2 id(s) not in demo-social's posts archive: zzz, yyy",
+ );
+ assert.equal(calls.length, 0);
+});
+
+test("the fetcher gets the ids (deduped), the channel's posts-media dir and the halves asked for", async () => {
+ const root = await makeChannel("demo-args");
+ calls = [];
+ answer = { outcomes: [] };
+ const res = await run("demo-args", ["aaa", "bbb", "aaa"], { media: false, force: true });
+ assert.equal(res.ok, true);
+ assert.equal(calls.length, 1);
+ assert.deepEqual(calls[0].ids, ["aaa", "bbb"]);
+ assert.equal(calls[0].outDir, path.join(root, "posts-media"));
+ assert.equal(calls[0].handle, "example.bsky.social");
+ assert.equal(calls[0].media, false);
+ assert.equal(calls[0].force, true);
+});
+
+test("deleted and walled posts are recorded in posts-availability.json; a login wall is not", async () => {
+ const root = await makeChannel("demo-avail");
+ await writePostAvailability(root, {
+ aaa: { availability: "available", checkedAt: "2026-01-01T00:00:00.000Z" },
+ });
+ answer = {
+ outcomes: [
+ { id: "aaa", state: "deleted", availability: "deleted", files: 0 },
+ { id: "bbb", state: "unavailable", availability: "account_unavailable", files: 0 },
+ { id: "ccc", state: "login-wall", files: 0 },
+ ],
+ needsCookies: true,
+ stoppedEarly: "X asked to log in. Stopped at ccc.",
+ };
+ const res = await run("demo-avail", ["aaa", "bbb", "ccc"]);
+ assert.equal(res.ok, false);
+ assert.equal(res.needsCookies, true);
+ assert.match(res.error ?? "", /Stopped at ccc/);
+ const map = await readPostAvailability(root);
+ assert.equal(map.aaa.availability, "deleted");
+ // The change keeps the moment it was last seen up.
+ assert.deepEqual(map.aaa.history, [{ availability: "available", at: "2026-01-01T00:00:00.000Z" }]);
+ assert.equal(map.bbb.availability, "account_unavailable");
+ assert.equal(map.ccc, undefined);
+});
+
+test("a run stopped by the source fails the job; one cancelled by the operator does not", async () => {
+ await makeChannel("demo-stop");
+ answer = { outcomes: [], stoppedEarly: "X is refusing pages right now." };
+ const stopped = await run("demo-stop", ["aaa"]);
+ assert.equal(stopped.ok, false);
+ assert.match(stopped.error ?? "", /refusing/);
+
+ const ac = new AbortController();
+ ac.abort();
+ answer = { outcomes: [], stoppedEarly: "Cancelled; the rest are left for a later run." };
+ const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal });
+ assert.equal(cancelled.ok, true);
+});
diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts
@@ -0,0 +1,206 @@
+// Capture specific archived posts of a social channel: a screenshot of each
+// post as its platform renders it, and its attached media, into
+// `channels/<slug>/posts-media/<id>/` (social/postCapture.ts owns the layout).
+//
+// The fetcher does the capturing (`SocialFetcher.captureByIds`); this lands
+// what it learned about each post's liveness in the channel's availability
+// sidecar, `posts-availability.json`, through the same merge the deleted-post
+// sweep uses — a deleted or walled post is recorded there, history appended
+// only on a change, and nothing is recorded from ignorance (a login wall says
+// nothing about the post).
+//
+// Only ids already in the channel's posts archive are captured: a capture is
+// of the archive, and an id from somewhere else is refused by name.
+
+import path from "node:path";
+import { readChannelConfig } from "./channels";
+import type { Paths } from "../lib/paths";
+import type { PostAvailability } from "../lib/posts";
+import { isSocialChannel } from "../lib/channelConfig";
+import {
+ alwaysCookies,
+ resolveCookiePolicy,
+ type CookiePolicyInputs,
+} from "../lib/cookiePolicy";
+import {
+ mergePostAvailability,
+ readPostAvailability,
+ readSeenPostIds,
+ writePostAvailability,
+} from "../lib/posts-server";
+import {
+ handleFromAccountUrl,
+ resolveSocialFetcher,
+ type PostCaptureOutcome,
+ type SocialFetcher,
+} from "../social/fetchers";
+import { postsMediaDir } from "../social/postCapture";
+import "../social/blueskyFetcher";
+import "../social/xGalleryDlFetcher";
+import "../social/xPlaywrightFetcher";
+import "../social/xNitterFetcher";
+import {
+ resolveXCookieSourceFor,
+ type XLoginSettings,
+} from "../social/xBrowserLogin";
+
+export type CapturePostsOptions = {
+ paths: Paths;
+ slug: string;
+ settings: CookiePolicyInputs & XLoginSettings;
+ ids: ReadonlyArray<string>;
+ // Take the screenshot / download the media. Both default to true.
+ shots?: boolean;
+ media?: boolean;
+ // Capture again what is already captured.
+ force?: boolean;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ // The job's drain: stop between posts.
+ drain?: AbortSignal;
+};
+
+export type CapturePostsResult = {
+ ok: boolean;
+ outcomes: PostCaptureOutcome[];
+ needsCookies?: boolean;
+ error?: string;
+};
+
+// Why this fetcher cannot capture posts, or null when it can. Shared with the
+// server action, so the refusal is one sentence everywhere.
+export function capturePostsProblem(
+ fetcher: Pick<SocialFetcher, "label" | "captureByIds"> | undefined,
+): string | null {
+ if (!fetcher) return "This channel has no post fetcher.";
+ return typeof fetcher.captureByIds === "function"
+ ? null
+ : `${fetcher.label} cannot capture posts.`;
+}
+
+export const NOTHING_TO_CAPTURE =
+ "Nothing to capture: both the screenshot and the media are turned off.";
+
+// The ids not in the channel's archive, or null when every one is.
+export function strayCaptureIds(
+ ids: ReadonlyArray<string>,
+ archived: ReadonlySet<string>,
+): string[] | null {
+ const stray = ids.filter((id) => !archived.has(id));
+ return stray.length ? stray : null;
+}
+
+export function strayIdsRefusal(slug: string, stray: string[]): string {
+ return `${stray.length} id(s) not in ${slug}'s posts archive: ${stray.join(", ")}`;
+}
+
+export async function capturePosts(
+ opts: CapturePostsOptions,
+): Promise<CapturePostsResult> {
+ const { paths, slug, settings, onLog } = opts;
+ const log = (line: string) => onLog?.(line);
+ const fail = (error: string): CapturePostsResult => ({ ok: false, outcomes: [], error });
+ const channelRoot = path.join(paths.channelsDir, slug);
+ const ids = [...new Set(opts.ids)];
+
+ if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE);
+ if (ids.length === 0) return fail("No post ids to capture.");
+
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return fail(`No such channel: ${slug}`);
+ if (!isSocialChannel(config)) return fail(`${slug} is not a social channel.`);
+
+ const accountUrl = config.url ?? "";
+ const fetcher = resolveSocialFetcher(config.postFetcher, accountUrl);
+ const problem = capturePostsProblem(fetcher);
+ if (problem || !fetcher?.captureByIds) return fail(problem ?? "This channel has no post fetcher.");
+
+ const handle = config.socialHandle ?? handleFromAccountUrl(accountUrl) ?? "";
+ if (!handle) return fail(`Could not determine an account handle for ${slug}`);
+
+ const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot));
+ if (stray) return fail(strayIdsRefusal(slug, stray));
+
+ const policy = resolveCookiePolicy(settings, config);
+ const xLogin =
+ fetcher.platform === "twitter"
+ ? await resolveXCookieSourceFor(paths, settings, policy.cookies)
+ : undefined;
+
+ log(
+ `Capturing ${ids.length} post(s) of ${slug} via ${fetcher.label} (@${handle}):` +
+ [
+ opts.shots === false ? "" : " screenshot",
+ opts.media === false ? "" : " media",
+ ].join("") +
+ (opts.force ? ", again where already captured" : "") +
+ ".",
+ );
+
+ const controller = new AbortController();
+ let result;
+ try {
+ result = await fetcher.captureByIds({
+ ids,
+ handle,
+ outDir: postsMediaDir(channelRoot),
+ shots: opts.shots,
+ media: opts.media,
+ force: opts.force,
+ cookies: alwaysCookies(policy),
+ cookieSource: xLogin?.source,
+ browserCookies: xLogin?.browserSpec,
+ signal: opts.signal ?? controller.signal,
+ drain: opts.drain,
+ onLog: log,
+ });
+ } catch (err) {
+ const message = (err as Error).message;
+ log(`[error] ${message}`);
+ return fail(message);
+ }
+
+ // What the pages said about each post, folded into the sidecar.
+ const observations = new Map<string, PostAvailability>();
+ for (const o of result.outcomes) {
+ if (o.availability) observations.set(o.id, o.availability);
+ }
+ if (observations.size > 0) {
+ const { map, newlyDeleted, changed } = mergePostAvailability(
+ await readPostAvailability(channelRoot),
+ observations,
+ new Date().toISOString(),
+ );
+ await writePostAvailability(channelRoot, map);
+ log(
+ `Availability: ${observations.size} recorded, ${changed} changed` +
+ (newlyDeleted.length ? `, newly deleted: ${newlyDeleted.join(", ")}` : "") +
+ ".",
+ );
+ }
+
+ const count = (state: string) => result.outcomes.filter((o) => o.state === state).length;
+ log(
+ `Done: ${count("captured")} captured, ${count("deleted")} deleted, ` +
+ `${count("unavailable")} unavailable, ${count("error")} failed` +
+ (count("login-wall") ? `, ${count("login-wall")} met a login wall` : "") +
+ `; ${ids.length - result.outcomes.length} not attempted or already captured.`,
+ );
+
+ // A run that stopped at a login is a failed run (the job says so); one that
+ // was cancelled or met deleted posts is not.
+ if (result.needsCookies) {
+ return {
+ ok: false,
+ outcomes: result.outcomes,
+ needsCookies: true,
+ error: result.stoppedEarly ?? "The capture needs an X login.",
+ };
+ }
+ const stoppedByOperator =
+ (opts.signal ?? controller.signal).aborted || Boolean(opts.drain?.aborted);
+ if (result.stoppedEarly && !stoppedByOperator) {
+ return { ok: false, outcomes: result.outcomes, error: result.stoppedEarly };
+ }
+ return { ok: true, outcomes: result.outcomes };
+}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -512,6 +512,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
replayable: true,
queueKeyStrategy: "platform",
},
+ // A screenshot and the attached media of specific archived posts, into the
+ // channel's `posts-media/` (controller/capturePosts.ts). On the PLATFORM
+ // queue, like fetch-posts: each post is a page load and a download against
+ // the same source, so it serialises with the fetch and shares its backoff.
+ // Drainable (it stops between posts) and replayable (the ids are in the
+ // spec; posts already captured are skipped on a re-run).
+ "capture-posts": {
+ kind: "capture-posts",
+ label: "Capture posts",
+ drainable: true,
+ replayable: true,
+ queueKeyStrategy: "platform",
+ },
// The posts analogue of the video availability check: which archived posts
// have since been deleted at the source.
"check-post-availability": {