Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 5a2b1667e8d3ba644ef5b166f0d4208dd56a6f1c
parent b99630d909dab9bfca1e0cdb36a9898d92a562a4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 16:16:57 -0400

Merge x-post-capture (X post capture: a screenshot and the attached media per post id, capture-posts job + ops route, PostModal shows a capture behind an editor-only prop)

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
MRUNNING_IN_DOCKER.md | 1+
Acommon/components/PostModal.test.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/components/PostModal.tsx | 91++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Acommon/controller/capturePosts.test.ts | 145+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/capturePosts.ts | 206+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.ts | 13+++++++++++++
Mcommon/social/fetchers.ts | 49+++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/playwrightRuntime.ts | 7+++++++
Acommon/social/postCapture.test.ts | 159+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/postCapture.ts | 189+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.test.ts | 67+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/social/xGalleryDlFetcher.ts | 227+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Acommon/social/xPostCapture.test.ts | 436+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/social/xPostCapture.ts | 402+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Aeditor/app/api/ops/capture-posts/route.ts | 41+++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/socialActions.ts | 73+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/jobs/jobReplayRegistry.ts | 14++++++++++++++
Meditor/e2e/ops-api.spec.ts | 63+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mscripts/archilyzer-ops.mjs | 9+++++++++
Mscripts/archilyzer-ops.test.mjs | 15+++++++++++++++
21 files changed, 2245 insertions(+), 26 deletions(-)

diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md @@ -255,6 +255,7 @@ pnpm ops lane --json '{"lane":"download","held":true}' pnpm ops refresh-report --json '{"all":true}' pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}' pnpm ops fetch-posts --json '{"slug":"example-x","older":true}' --wait +pnpm ops capture-posts --json '{"slug":"example-x","ids":["1234567890"]}' --wait pnpm ops get channel the-quartering pnpm ops list # every action name ``` diff --git a/common/components/PostModal.test.ts b/common/components/PostModal.test.ts @@ -0,0 +1,63 @@ +// The post viewer's capture panel: shown only from a capture the editor hands +// it, with the shot and each medium in its own element; nothing without one. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test components/PostModal.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import * as React from "react"; +import { renderToStaticMarkup } from "react-dom/server"; +import { + captureMediaKind, + PostCapturePanel, + type PostCaptureView, +} from "./PostModal"; + +// common's tsconfig has `jsx: "preserve"`, so tsx falls back to the classic +// transform — `React.createElement` on a free `React` (as BrandMark.test.ts). +(globalThis as { React?: typeof React }).React = React; +const render = (capture: PostCaptureView) => + renderToStaticMarkup(React.createElement(PostCapturePanel, { capture })); + +const BASE: PostCaptureView = { + capturedAt: "2026-02-03T04:05:06.000Z", + state: "captured", + media: [], +}; + +test("with media: the shot, each image and video in its own element, the sensitive note", () => { + const html = render({ + ...BASE, + sensitive: true, + shot: { src: "/capture/demo/1/shot.png" }, + media: [ + { src: "/capture/demo/1/1_1.jpg", name: "1_1.jpg" }, + { src: "/capture/demo/1/1_2.mp4", name: "1_2.mp4" }, + { src: "/capture/demo/1/1_3.bin", name: "1_3.bin" }, + ], + }); + assert.match(html, /data-post-capture=""/); + assert.match(html, /<img src="\/capture\/demo\/1\/shot\.png"[^>]*data-post-capture-shot=""/); + assert.match(html, /data-post-capture-media="image"><img src="\/capture\/demo\/1\/1_1\.jpg"/); + assert.match(html, /data-post-capture-media="video"><video src="\/capture\/demo\/1\/1_2\.mp4"/); + assert.match(html, /data-post-capture-media="other"><a href="\/capture\/demo\/1\/1_3\.bin"/); + assert.match(html, /behind a sensitive-media cover/); +}); + +test("a shot with no media shows the shot alone", () => { + const html = render({ ...BASE, shot: { src: "/s.png" } }); + assert.match(html, /data-post-capture-shot/); + assert.doesNotMatch(html, /data-post-capture-media/); + assert.doesNotMatch(html, /sensitive/); +}); + +test("without media or a shot (a deleted post's record): nothing at all", () => { + assert.equal(render({ ...BASE, state: "deleted" }), ""); +}); + +test("media kinds by extension", () => { + assert.equal(captureMediaKind("a.JPG"), "image"); + assert.equal(captureMediaKind("a.webp"), "image"); + assert.equal(captureMediaKind("a.mp4"), "video"); + assert.equal(captureMediaKind("a"), "other"); +}); diff --git a/common/components/PostModal.tsx b/common/components/PostModal.tsx @@ -33,7 +33,28 @@ function platformLabel(platform: Post["platform"]): string { return platform === "twitter" ? "X" : "Bluesky"; } -export default function PostModal() { +// A post's capture (a screenshot and its attached media, taken by the +// editor's capture-posts job), as the page that renders this modal serves it. +// EDITOR-ONLY: the files live beside the channel (`posts-media/<id>/`) and are +// never published, so the export never passes `loadCapture` and never shows +// any of this. +export type PostCaptureView = { + capturedAt: string; + // The capture's state ("captured", "deleted", …), as recorded. + state: string; + // The post sat behind a sensitive-media cover when it was shot. + sensitive?: boolean; + shot?: { src: string }; + media: { src: string; name: string }[]; +}; + +export type PostModalProps = { + // Where the editor reads a post's capture from. Absent (the export): no + // capture is looked for or shown. + loadCapture?: (post: Post) => Promise<PostCaptureView | null>; +}; + +export default function PostModal({ loadCapture }: PostModalProps = {}) { const { v: slug, vm } = useUrlParams(); const open = vm === "post" && !!slug; @@ -42,6 +63,11 @@ export default function PostModal() { queryFn: () => fetchPost(slug!), enabled: open, }); + const capture = useQuery({ + queryKey: ["post-capture", slug], + queryFn: () => loadCapture!(post.data!), + enabled: open && !!loadCapture && !!post.data, + }); const thread = useQuery({ queryKey: ["post-thread", slug], queryFn: () => fetchThread(slug!), @@ -101,7 +127,7 @@ export default function PostModal() { {post.data && ( <div className="flex flex-col gap-4 px-4 py-4"> - <PostBody post={post.data} primary /> + <PostBody post={post.data} primary capture={capture.data} /> {/* Thread context: the archived parent + replies around this post. The post-corpus analogue of a transcript's surrounding cues. */} @@ -130,12 +156,15 @@ function PostBody({ post, primary = false, compact = false, + capture, }: { post: Post; primary?: boolean; compact?: boolean; + capture?: PostCaptureView | null; }) { const links = useMemo(() => post.links ?? [], [post.links]); + const capturedMedia = capture?.media.length ?? 0; return ( <article data-post-id={post.id} @@ -173,6 +202,8 @@ function PostBody({ {post.text} </p> + {capture && <PostCapturePanel capture={capture} />} + {links.length > 0 && ( <ul className="mt-2 flex flex-col gap-1"> {links.map((href) => ( @@ -208,7 +239,11 @@ function PostBody({ {post.engagement?.replies != null && ( <span>{post.engagement.replies} replies</span> )} - {post.mediaCount ? <span>{post.mediaCount} media (not archived)</span> : null} + {capturedMedia > 0 ? ( + <span>{capturedMedia} media (captured)</span> + ) : post.mediaCount ? ( + <span>{post.mediaCount} media (not archived)</span> + ) : null} </footer> </article> ); @@ -221,3 +256,53 @@ function Badge({ children }: { children: React.ReactNode }) { </span> ); } + +// What a captured file is, by its extension. +export function captureMediaKind(name: string): "image" | "video" | "other" { + const ext = name.toLowerCase().split(".").pop() ?? ""; + if (["jpg", "jpeg", "png", "gif", "webp", "avif"].includes(ext)) return "image"; + if (["mp4", "webm", "mov", "m4v"].includes(ext)) return "video"; + return "other"; +} + +// The captured screenshot and media of one post. Nothing when the capture +// holds neither (a deleted post's record, say). +export function PostCapturePanel({ capture }: { capture: PostCaptureView }) { + if (!capture.shot && capture.media.length === 0) return null; + return ( + <section data-post-capture="" className="mt-3 flex flex-col gap-2"> + <h3 className="text-xs uppercase tracking-wide text-muted-foreground"> + Captured {formatWhen(capture.capturedAt)} + {capture.sensitive ? " · behind a sensitive-media cover" : ""} + </h3> + {capture.shot && ( + <img + src={capture.shot.src} + alt="Screenshot of the post as captured" + data-post-capture-shot="" + className="max-w-full rounded-md border border-border" + /> + )} + {capture.media.length > 0 && ( + <ul className="flex flex-col gap-2"> + {capture.media.map((m) => { + const kind = captureMediaKind(m.name); + return ( + <li key={m.src} data-post-capture-media={kind}> + {kind === "image" ? ( + <img src={m.src} alt={m.name} className="max-w-full rounded-md" /> + ) : kind === "video" ? ( + <video src={m.src} controls preload="metadata" className="max-w-full rounded-md" /> + ) : ( + <a href={m.src} className="text-xs text-primary underline break-all"> + {m.name} + </a> + )} + </li> + ); + })} + </ul> + )} + </section> + ); +} diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts @@ -0,0 +1,145 @@ +// capturePosts over a registered fake fetcher: what is refused before any +// capture, and how the pages' verdicts land in the availability sidecar. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/capturePosts.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "captureposts-")); +process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); + +const { getPaths } = await import("../lib/paths"); +const { registerSocialFetcher } = await import("../social/fetchers"); +const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server"); +const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts"); +type Input = import("../social/fetchers").PostCaptureInput; +type Result = import("../social/fetchers").PostCaptureResult; + +// A capturing fetcher whose answer each test sets. +let calls: Input[] = []; +let answer: Result = { outcomes: [] }; +registerSocialFetcher({ + id: "test-capture", + label: "Test capture", + platform: "bluesky", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), + captureByIds: async (input) => { + calls.push(input); + return answer; + }, +}); +registerSocialFetcher({ + id: "test-no-capture", + label: "No-capture fetcher", + platform: "bluesky", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), +}); + +const channelsDir = path.join(ROOT, "transcripts", "channels"); + +async function makeChannel(slug: string, over: Record<string, unknown> = {}): Promise<string> { + const root = path.join(channelsDir, slug); + await mkdir(root, { recursive: true }); + await writeFile( + path.join(root, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + platform: "bluesky", + postFetcher: "test-capture", + socialHandle: "example.bsky.social", + name: "Example", + url: "https://bsky.app/profile/example.bsky.social", + ...over, + }), + ); + await writeFile(path.join(root, "posts-archive"), "bluesky aaa\nbluesky bbb\nbluesky ccc\n"); + return root; +} + +const run = (slug: string, ids: string[], over: Record<string, unknown> = {}) => + capturePosts({ paths: getPaths(), slug, settings: {}, ids, ...over }); + +test("refused before any capture: nothing asked for, no ids, not social, no capture, ids not archived", async () => { + await makeChannel("demo-social"); + await makeChannel("demo-nocapture", { postFetcher: "test-no-capture" }); + await mkdir(path.join(channelsDir, "demo-video"), { recursive: true }); + await writeFile( + path.join(channelsDir, "demo-video", "config.json"), + JSON.stringify({ handling: "transcribe", name: "Video", url: "https://example.com/c" }), + ); + calls = []; + assert.equal((await run("demo-social", ["aaa"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE); + assert.match((await run("demo-social", [])).error ?? "", /No post ids/); + assert.equal((await run("demo-video", ["aaa"])).error, "demo-video is not a social channel."); + assert.equal((await run("demo-nocapture", ["aaa"])).error, "No-capture fetcher cannot capture posts."); + assert.equal( + (await run("demo-social", ["aaa", "zzz", "yyy"])).error, + "2 id(s) not in demo-social's posts archive: zzz, yyy", + ); + assert.equal(calls.length, 0); +}); + +test("the fetcher gets the ids (deduped), the channel's posts-media dir and the halves asked for", async () => { + const root = await makeChannel("demo-args"); + calls = []; + answer = { outcomes: [] }; + const res = await run("demo-args", ["aaa", "bbb", "aaa"], { media: false, force: true }); + assert.equal(res.ok, true); + assert.equal(calls.length, 1); + assert.deepEqual(calls[0].ids, ["aaa", "bbb"]); + assert.equal(calls[0].outDir, path.join(root, "posts-media")); + assert.equal(calls[0].handle, "example.bsky.social"); + assert.equal(calls[0].media, false); + assert.equal(calls[0].force, true); +}); + +test("deleted and walled posts are recorded in posts-availability.json; a login wall is not", async () => { + const root = await makeChannel("demo-avail"); + await writePostAvailability(root, { + aaa: { availability: "available", checkedAt: "2026-01-01T00:00:00.000Z" }, + }); + answer = { + outcomes: [ + { id: "aaa", state: "deleted", availability: "deleted", files: 0 }, + { id: "bbb", state: "unavailable", availability: "account_unavailable", files: 0 }, + { id: "ccc", state: "login-wall", files: 0 }, + ], + needsCookies: true, + stoppedEarly: "X asked to log in. Stopped at ccc.", + }; + const res = await run("demo-avail", ["aaa", "bbb", "ccc"]); + assert.equal(res.ok, false); + assert.equal(res.needsCookies, true); + assert.match(res.error ?? "", /Stopped at ccc/); + const map = await readPostAvailability(root); + assert.equal(map.aaa.availability, "deleted"); + // The change keeps the moment it was last seen up. + assert.deepEqual(map.aaa.history, [{ availability: "available", at: "2026-01-01T00:00:00.000Z" }]); + assert.equal(map.bbb.availability, "account_unavailable"); + assert.equal(map.ccc, undefined); +}); + +test("a run stopped by the source fails the job; one cancelled by the operator does not", async () => { + await makeChannel("demo-stop"); + answer = { outcomes: [], stoppedEarly: "X is refusing pages right now." }; + const stopped = await run("demo-stop", ["aaa"]); + assert.equal(stopped.ok, false); + assert.match(stopped.error ?? "", /refusing/); + + const ac = new AbortController(); + ac.abort(); + answer = { outcomes: [], stoppedEarly: "Cancelled; the rest are left for a later run." }; + const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal }); + assert.equal(cancelled.ok, true); +}); diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts @@ -0,0 +1,206 @@ +// Capture specific archived posts of a social channel: a screenshot of each +// post as its platform renders it, and its attached media, into +// `channels/<slug>/posts-media/<id>/` (social/postCapture.ts owns the layout). +// +// The fetcher does the capturing (`SocialFetcher.captureByIds`); this lands +// what it learned about each post's liveness in the channel's availability +// sidecar, `posts-availability.json`, through the same merge the deleted-post +// sweep uses — a deleted or walled post is recorded there, history appended +// only on a change, and nothing is recorded from ignorance (a login wall says +// nothing about the post). +// +// Only ids already in the channel's posts archive are captured: a capture is +// of the archive, and an id from somewhere else is refused by name. + +import path from "node:path"; +import { readChannelConfig } from "./channels"; +import type { Paths } from "../lib/paths"; +import type { PostAvailability } from "../lib/posts"; +import { isSocialChannel } from "../lib/channelConfig"; +import { + alwaysCookies, + resolveCookiePolicy, + type CookiePolicyInputs, +} from "../lib/cookiePolicy"; +import { + mergePostAvailability, + readPostAvailability, + readSeenPostIds, + writePostAvailability, +} from "../lib/posts-server"; +import { + handleFromAccountUrl, + resolveSocialFetcher, + type PostCaptureOutcome, + type SocialFetcher, +} from "../social/fetchers"; +import { postsMediaDir } from "../social/postCapture"; +import "../social/blueskyFetcher"; +import "../social/xGalleryDlFetcher"; +import "../social/xPlaywrightFetcher"; +import "../social/xNitterFetcher"; +import { + resolveXCookieSourceFor, + type XLoginSettings, +} from "../social/xBrowserLogin"; + +export type CapturePostsOptions = { + paths: Paths; + slug: string; + settings: CookiePolicyInputs & XLoginSettings; + ids: ReadonlyArray<string>; + // Take the screenshot / download the media. Both default to true. + shots?: boolean; + media?: boolean; + // Capture again what is already captured. + force?: boolean; + onLog?: (line: string) => void; + signal?: AbortSignal; + // The job's drain: stop between posts. + drain?: AbortSignal; +}; + +export type CapturePostsResult = { + ok: boolean; + outcomes: PostCaptureOutcome[]; + needsCookies?: boolean; + error?: string; +}; + +// Why this fetcher cannot capture posts, or null when it can. Shared with the +// server action, so the refusal is one sentence everywhere. +export function capturePostsProblem( + fetcher: Pick<SocialFetcher, "label" | "captureByIds"> | undefined, +): string | null { + if (!fetcher) return "This channel has no post fetcher."; + return typeof fetcher.captureByIds === "function" + ? null + : `${fetcher.label} cannot capture posts.`; +} + +export const NOTHING_TO_CAPTURE = + "Nothing to capture: both the screenshot and the media are turned off."; + +// The ids not in the channel's archive, or null when every one is. +export function strayCaptureIds( + ids: ReadonlyArray<string>, + archived: ReadonlySet<string>, +): string[] | null { + const stray = ids.filter((id) => !archived.has(id)); + return stray.length ? stray : null; +} + +export function strayIdsRefusal(slug: string, stray: string[]): string { + return `${stray.length} id(s) not in ${slug}'s posts archive: ${stray.join(", ")}`; +} + +export async function capturePosts( + opts: CapturePostsOptions, +): Promise<CapturePostsResult> { + const { paths, slug, settings, onLog } = opts; + const log = (line: string) => onLog?.(line); + const fail = (error: string): CapturePostsResult => ({ ok: false, outcomes: [], error }); + const channelRoot = path.join(paths.channelsDir, slug); + const ids = [...new Set(opts.ids)]; + + if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE); + if (ids.length === 0) return fail("No post ids to capture."); + + const config = await readChannelConfig(paths, slug); + if (!config) return fail(`No such channel: ${slug}`); + if (!isSocialChannel(config)) return fail(`${slug} is not a social channel.`); + + const accountUrl = config.url ?? ""; + const fetcher = resolveSocialFetcher(config.postFetcher, accountUrl); + const problem = capturePostsProblem(fetcher); + if (problem || !fetcher?.captureByIds) return fail(problem ?? "This channel has no post fetcher."); + + const handle = config.socialHandle ?? handleFromAccountUrl(accountUrl) ?? ""; + if (!handle) return fail(`Could not determine an account handle for ${slug}`); + + const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot)); + if (stray) return fail(strayIdsRefusal(slug, stray)); + + const policy = resolveCookiePolicy(settings, config); + const xLogin = + fetcher.platform === "twitter" + ? await resolveXCookieSourceFor(paths, settings, policy.cookies) + : undefined; + + log( + `Capturing ${ids.length} post(s) of ${slug} via ${fetcher.label} (@${handle}):` + + [ + opts.shots === false ? "" : " screenshot", + opts.media === false ? "" : " media", + ].join("") + + (opts.force ? ", again where already captured" : "") + + ".", + ); + + const controller = new AbortController(); + let result; + try { + result = await fetcher.captureByIds({ + ids, + handle, + outDir: postsMediaDir(channelRoot), + shots: opts.shots, + media: opts.media, + force: opts.force, + cookies: alwaysCookies(policy), + cookieSource: xLogin?.source, + browserCookies: xLogin?.browserSpec, + signal: opts.signal ?? controller.signal, + drain: opts.drain, + onLog: log, + }); + } catch (err) { + const message = (err as Error).message; + log(`[error] ${message}`); + return fail(message); + } + + // What the pages said about each post, folded into the sidecar. + const observations = new Map<string, PostAvailability>(); + for (const o of result.outcomes) { + if (o.availability) observations.set(o.id, o.availability); + } + if (observations.size > 0) { + const { map, newlyDeleted, changed } = mergePostAvailability( + await readPostAvailability(channelRoot), + observations, + new Date().toISOString(), + ); + await writePostAvailability(channelRoot, map); + log( + `Availability: ${observations.size} recorded, ${changed} changed` + + (newlyDeleted.length ? `, newly deleted: ${newlyDeleted.join(", ")}` : "") + + ".", + ); + } + + const count = (state: string) => result.outcomes.filter((o) => o.state === state).length; + log( + `Done: ${count("captured")} captured, ${count("deleted")} deleted, ` + + `${count("unavailable")} unavailable, ${count("error")} failed` + + (count("login-wall") ? `, ${count("login-wall")} met a login wall` : "") + + `; ${ids.length - result.outcomes.length} not attempted or already captured.`, + ); + + // A run that stopped at a login is a failed run (the job says so); one that + // was cancelled or met deleted posts is not. + if (result.needsCookies) { + return { + ok: false, + outcomes: result.outcomes, + needsCookies: true, + error: result.stoppedEarly ?? "The capture needs an X login.", + }; + } + const stoppedByOperator = + (opts.signal ?? controller.signal).aborted || Boolean(opts.drain?.aborted); + if (result.stoppedEarly && !stoppedByOperator) { + return { ok: false, outcomes: result.outcomes, error: result.stoppedEarly }; + } + return { ok: true, outcomes: result.outcomes }; +} diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -512,6 +512,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: true, queueKeyStrategy: "platform", }, + // A screenshot and the attached media of specific archived posts, into the + // channel's `posts-media/` (controller/capturePosts.ts). On the PLATFORM + // queue, like fetch-posts: each post is a page load and a download against + // the same source, so it serialises with the fetch and shares its backoff. + // Drainable (it stops between posts) and replayable (the ids are in the + // spec; posts already captured are skipped on a re-run). + "capture-posts": { + kind: "capture-posts", + label: "Capture posts", + drainable: true, + replayable: true, + queueKeyStrategy: "platform", + }, // The posts analogue of the video availability check: which archived posts // have since been deleted at the source. "check-post-availability": { diff --git a/common/social/fetchers.ts b/common/social/fetchers.ts @@ -16,6 +16,7 @@ import type { Post, PostAvailability, PostPlatform } from "../lib/posts"; import type { OlderBackfillPosition } from "../lib/posts-server"; import type { XCookieSource } from "./xCookieSource"; +import type { PostCaptureState } from "./postCapture"; export type PostFetchInput = { // The account's canonical URL as configured on the channel. @@ -164,6 +165,49 @@ export type PostAvailabilityInput = { onLog?: (line: string) => void; }; +// Input for a post capture (`SocialFetcher.captureByIds`): a screenshot of each +// post as the platform renders it, and its attached media, for specific +// archived posts. The login fields are a normal fetch's; `outDir` is the +// channel's `posts-media/`, one directory per post id beneath it +// (postCapture.ts owns the layout). +export type PostCaptureInput = Pick< + PostFetchInput, + "cookies" | "cookieSource" | "browserCookies" | "signal" | "onLog" +> & { + ids: ReadonlyArray<string>; + handle: string; + outDir: string; + // Take the screenshot / download the media. Both default to true. + shots?: boolean; + media?: boolean; + // Capture again what is already captured. + force?: boolean; + // A soft stop: no new post is started once it fires, and the one in hand + // finishes (`signal` cancels outright). + drain?: AbortSignal; +}; + +export type PostCaptureOutcome = { + id: string; + state: PostCaptureState; + // What the capture learned about the post's liveness, for the availability + // sidecar. Absent when it learned nothing about the post itself (a login + // wall is the session's state, not the post's). + availability?: PostAvailability; + // Files on disk for this post after the run (shot + media). + files: number; + error?: string; +}; + +export type PostCaptureResult = { + outcomes: PostCaptureOutcome[]; + // The run stopped at a login wall, or at a login the media download refused. + needsCookies?: boolean; + // Why the run stopped before its last id, when it did. The ids after it are + // left for a later run — never retried in this one. + stoppedEarly?: string; +}; + export type SocialFetcher = { id: string; label: string; @@ -185,6 +229,11 @@ export type SocialFetcher = { // reach (X: search windows). A fetcher without it has no such backfill, and // the channel page offers no "Fetch older posts" for it. fetchOlder?(input: OlderPostFetchInput): Promise<OlderPostFetchResult>; + // Optional: capture specific archived posts — a screenshot and the attached + // media — into `posts-media/<id>/` (X: the logged-in profile shoots the + // post, gallery-dl downloads its media). A fetcher without it cannot, and + // the capture refuses its channel by name. + captureByIds?(input: PostCaptureInput): Promise<PostCaptureResult>; }; // Populated by registerSocialFetcher() from each fetcher module. Indirection diff --git a/common/social/playwrightRuntime.ts b/common/social/playwrightRuntime.ts @@ -41,6 +41,13 @@ export type PageLike = { waitForTimeout: (ms: number) => Promise<void>; evaluate: (fn: string) => Promise<unknown>; on: (event: string, cb: (arg: never) => void) => void; + // The post capture's shot (xPostCapture.ts): a PNG of the page, clipped to + // the post. `clip` is in document coordinates when `fullPage` is set. + screenshot: (opts?: { + type?: "png" | "jpeg"; + fullPage?: boolean; + clip?: { x: number; y: number; width: number; height: number }; + }) => Promise<Uint8Array>; }; export type BrowserContextLike = { diff --git a/common/social/postCapture.test.ts b/common/social/postCapture.test.ts @@ -0,0 +1,159 @@ +// The post capture's on-disk record (capture.json), what a capture still owes +// a post, and the guard that keeps `posts-media/` out of every export. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/postCapture.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { mkdir, mkdtemp, readdir, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { fileURLToPath } from "node:url"; +import { + CAPTURE_FILENAME, + captureAvailability, + captureWork, + describeCapturedFile, + listCapturedMediaFiles, + POSTS_MEDIA_DIRNAME, + postCaptureDir, + postsMediaDir, + readPostCapture, + SHOT_FILENAME, + writePostCapture, + type PostCaptureRecord, +} from "./postCapture"; + +const sha = (b: string | Buffer) => createHash("sha256").update(b).digest("hex"); + +function record(over: Partial<PostCaptureRecord> = {}): PostCaptureRecord { + return { + version: 1, + id: "123", + url: "https://x.com/i/status/123", + capturedAt: "2026-01-02T03:04:05.000Z", + state: "captured", + mediaState: "ok", + media: [], + ...over, + }; +} + +test("layout: posts-media/<id>/ beside the channel's data, and an id is never a path", () => { + assert.equal(postsMediaDir("/c/channels/demo-channel"), "/c/channels/demo-channel/posts-media"); + assert.equal(POSTS_MEDIA_DIRNAME, "posts-media"); + assert.equal(postCaptureDir("/o", "1234"), "/o/1234"); + assert.equal(postCaptureDir("/o", "3kabc_d-e"), "/o/3kabc_d-e"); + for (const id of ["../x", "a/b", "", ".", "..", "a b"]) { + assert.throws(() => postCaptureDir("/o", id), /is not a post id/, JSON.stringify(id)); + } +}); + +test("capture.json: each file's sha256 and byte size, the URLs, round-tripped", async () => { + const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "capture-")), "123"); + await mkdir(dir, { recursive: true }); + const png = Buffer.from([0x89, 0x50, 0x4e, 0x47, 1, 2, 3]); + await writeFile(path.join(dir, SHOT_FILENAME), png); + await writeFile(path.join(dir, "123_1.jpg"), "jpeg bytes"); + const shot = await describeCapturedFile(dir, SHOT_FILENAME, "https://x.com/i/status/123"); + const media = await describeCapturedFile(dir, "123_1.jpg", "https://pbs.example/m.jpg"); + assert.deepEqual(shot, { + name: SHOT_FILENAME, + bytes: png.length, + sha256: sha(png), + url: "https://x.com/i/status/123", + }); + assert.equal(media.sha256, sha("jpeg bytes")); + assert.equal(media.bytes, "jpeg bytes".length); + // No URL known: the key is absent, never "undefined". + assert.ok(!("url" in (await describeCapturedFile(dir, "123_1.jpg")))); + + const rec = record({ shot, media: [media] }); + await writePostCapture(dir, rec); + assert.deepEqual(JSON.parse(await readFile(path.join(dir, CAPTURE_FILENAME), "utf8")), rec); + assert.deepEqual(await readPostCapture(dir), rec); + // The record and the shot are not media. + assert.deepEqual(await listCapturedMediaFiles(dir), ["123_1.jpg"]); +}); + +test("capture.json: absent, unparseable or of another shape reads as no capture", async () => { + const dir = await mkdtemp(path.join(os.tmpdir(), "capture-")); + assert.equal(await readPostCapture(dir), null); + await writeFile(path.join(dir, CAPTURE_FILENAME), "{ truncated"); + assert.equal(await readPostCapture(dir), null); + await writeFile(path.join(dir, CAPTURE_FILENAME), JSON.stringify({ version: 2, id: "1" })); + assert.equal(await readPostCapture(dir), null); + assert.deepEqual(await listCapturedMediaFiles(path.join(dir, "missing")), []); +}); + +test("availability: only what the page said about the post — never a verdict from a login wall or an error", () => { + assert.equal(captureAvailability("captured"), "available"); + assert.equal(captureAvailability("deleted"), "deleted"); + assert.equal(captureAvailability("unavailable"), "account_unavailable"); + assert.equal(captureAvailability("login-wall"), undefined); + assert.equal(captureAvailability("error"), undefined); +}); + +test("what a capture owes: nothing for a settled post, only the missing half otherwise, everything when forced", () => { + const all = { shots: true, media: true, force: false }; + assert.deepEqual(captureWork(null, all), { shot: true, media: true }); + assert.deepEqual(captureWork(null, { ...all, media: false }), { shot: true, media: false }); + const done = record({ shot: { name: SHOT_FILENAME, bytes: 1, sha256: "x" }, mediaState: "ok" }); + assert.deepEqual(captureWork(done, all), { shot: false, media: false }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "none" }), all), { shot: false, media: false }); + // The media failed (or was never asked for): only the media is owed. + assert.deepEqual(captureWork(record({ ...done, mediaState: "error" }), all), { shot: false, media: true }); + assert.deepEqual(captureWork(record({ ...done, mediaState: "skipped" }), all), { shot: false, media: true }); + // A shot that failed is owed again. + assert.deepEqual(captureWork(record({ state: "error", mediaState: "ok" }), all), { shot: true, media: false }); + // Deleted is settled: asking again is a request for the same answer. + assert.deepEqual(captureWork(record({ state: "deleted", mediaState: "skipped" }), all), { + shot: false, + media: false, + }); + // force re-takes whatever was asked for. + assert.deepEqual(captureWork(done, { ...all, force: true }), { shot: true, media: true }); + assert.deepEqual(captureWork(record({ state: "deleted" }), { shots: false, media: true, force: true }), { + shot: false, + media: true, + }); +}); + +// THE EXPORT NEVER PUBLISHES A CAPTURE. The export serves the index's JSON +// pages and copies trees out of export/public — it never reads a channel +// directory — so nothing it builds can carry posts-media/. Held here the cheap +// way: no module of the export build, the publish layer or the export app +// names the directory. +test("no export or publish source names posts-media/", async () => { + const HERE = path.dirname(fileURLToPath(import.meta.url)); + const repo = path.resolve(HERE, "..", ".."); + const roots = [ + path.join(repo, "common", "publish"), + path.join(repo, "common", "bin"), + path.join(repo, "export", "app"), + path.join(repo, "export", "lib"), + ]; + const offenders: string[] = []; + const walk = async (dir: string): Promise<void> => { + let entries; + try { + entries = await readdir(dir, { withFileTypes: true }); + } catch { + return; + } + for (const e of entries) { + if (e.name === "node_modules" || e.name.startsWith(".")) continue; + const p = path.join(dir, e.name); + if (e.isDirectory()) await walk(p); + else if (/\.(ts|tsx|mjs|js)$/.test(e.name)) { + const text = await readFile(p, "utf8"); + if (text.includes(POSTS_MEDIA_DIRNAME) || text.includes("POSTS_MEDIA_DIRNAME") || text.includes("social/postCapture")) { + offenders.push(path.relative(repo, p)); + } + } + } + }; + for (const r of roots) await walk(r); + assert.deepEqual(offenders, []); +}); diff --git a/common/social/postCapture.ts b/common/social/postCapture.ts @@ -0,0 +1,189 @@ +// Where a post capture lands on disk, and the record that says what it holds. +// +// Layout, per channel: +// channels/<slug>/posts-media/<post id>/shot.png — the post, as rendered +// channels/<slug>/posts-media/<post id>/<media…> — its attached media +// channels/<slug>/posts-media/<post id>/capture.json — this module's record +// +// A directory per post, unlike the posts themselves (month-sharded JSONL): a +// capture is a handful of files, and only for the posts someone asked for. It +// sits BESIDE `data/` and `media/`, never in them — no video reader walks it, +// and the media tier does not move it. +// +// EDITOR-ONLY. The export build serves the index's JSON pages and never reads a +// channel directory, so nothing here is published (postCapture.test.ts holds +// the export's sources to that). +// +// SERVER-ONLY (node:fs). + +import { createHash } from "node:crypto"; +import { createReadStream } from "node:fs"; +import { readdir, stat } from "node:fs/promises"; +import path from "node:path"; +import { readJsonFile, writeJsonAtomic } from "../lib/jsonFile-server"; +import type { PostAvailability } from "../lib/posts"; + +export const POSTS_MEDIA_DIRNAME = "posts-media"; +export const CAPTURE_FILENAME = "capture.json"; +export const SHOT_FILENAME = "shot.png"; + +export function postsMediaDir(channelRoot: string): string { + return path.join(channelRoot, POSTS_MEDIA_DIRNAME); +} + +// One post's capture directory. The id is checked here, at the one place a +// post id becomes a path: a platform-native id is digits (X) or a short +// alphanumeric key (Bluesky), never a separator. +export function postCaptureDir(outDir: string, id: string): string { + if (!/^[A-Za-z0-9_-]{1,64}$/.test(id)) { + throw new Error(`"${id}" is not a post id`); + } + return path.join(outDir, id); +} + +// What a capture found the post to be. +// captured — the post rendered, and what was asked for is on disk +// deleted — the platform says the post is gone +// unavailable — the post is behind its account's wall (protected, +// suspended, withheld): the post may exist, it cannot be shown +// login-wall — the platform asked to log in: the session's state, not the +// post's, so the run stops +// error — the page or the download failed; a later run tries again +export type PostCaptureState = + | "captured" + | "deleted" + | "unavailable" + | "login-wall" + | "error"; + +// The liveness a capture state says about the post, for the availability +// sidecar. A login wall and an error say nothing about the post itself — the +// sidecar never records a verdict from ignorance. +export function captureAvailability( + state: PostCaptureState, +): PostAvailability | undefined { + switch (state) { + case "captured": + return "available"; + case "deleted": + return "deleted"; + case "unavailable": + return "account_unavailable"; + default: + return undefined; + } +} + +export type CapturedFile = { + // The file's name inside the post's directory. + name: string; + bytes: number; + sha256: string; + // Where it came from: the media URL for an attachment, the post URL for the + // screenshot. + url?: string; +}; + +// How the media half went: downloaded ("ok"), the post has none ("none"), not +// asked for or not attempted ("skipped"), or failed ("error"). +export type CaptureMediaState = "ok" | "none" | "skipped" | "error"; + +export type PostCaptureRecord = { + version: 1; + id: string; + // The post's URL the capture read. + url: string; + capturedAt: string; + state: PostCaptureState; + // The post sat behind a sensitive-media interstitial, which the capture + // opened before shooting. + sensitive?: boolean; + // The screenshot, when one was taken. + shot?: CapturedFile; + mediaState: CaptureMediaState; + media: CapturedFile[]; + error?: string; +}; + +export async function fileDigest( + file: string, +): Promise<{ bytes: number; sha256: string }> { + const hash = createHash("sha256"); + await new Promise<void>((resolve, reject) => { + const s = createReadStream(file); + s.on("data", (chunk) => hash.update(chunk)); + s.on("error", reject); + s.on("end", () => resolve()); + }); + const { size } = await stat(file); + return { bytes: size, sha256: hash.digest("hex") }; +} + +export async function describeCapturedFile( + dir: string, + name: string, + url?: string, +): Promise<CapturedFile> { + const digest = await fileDigest(path.join(dir, name)); + return { name, ...digest, ...(url ? { url } : {}) }; +} + +// The media files in a post's directory: everything but the shot, the record +// and a download's leftovers. +export async function listCapturedMediaFiles(dir: string): Promise<string[]> { + let names: string[]; + try { + names = await readdir(dir); + } catch { + return []; + } + return names + .filter( + (n) => + n !== SHOT_FILENAME && + n !== CAPTURE_FILENAME && + !n.endsWith(".part") && + !n.startsWith("."), + ) + .sort(); +} + +export async function readPostCapture( + dir: string, +): Promise<PostCaptureRecord | null> { + const read = await readJsonFile(path.join(dir, CAPTURE_FILENAME)); + if (!read.ok) return null; + const v = read.value as Partial<PostCaptureRecord> | null; + if (!v || typeof v !== "object" || v.version !== 1 || typeof v.id !== "string") { + return null; + } + return v as PostCaptureRecord; +} + +export async function writePostCapture( + dir: string, + record: PostCaptureRecord, +): Promise<void> { + await writeJsonAtomic(path.join(dir, CAPTURE_FILENAME), record, { mkdir: true }); +} + +// Which halves of a capture this run still owes a post, given what is on disk. +// A deleted post is settled: the platform said so, and asking again costs a +// request for the same answer (`force` asks anyway). A shot on disk is kept; a +// media download that did not finish ("error", or never attempted) is owed. +export function captureWork( + existing: PostCaptureRecord | null, + wanted: { shots: boolean; media: boolean; force: boolean }, +): { shot: boolean; media: boolean } { + if (wanted.force || !existing) { + return { shot: wanted.shots, media: wanted.media }; + } + if (existing.state === "deleted") return { shot: false, media: false }; + return { + shot: wanted.shots && !existing.shot, + media: + wanted.media && + existing.mediaState !== "ok" && + existing.mediaState !== "none", + }; +} diff --git a/common/social/xGalleryDlFetcher.test.ts b/common/social/xGalleryDlFetcher.test.ts @@ -7,7 +7,11 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { buildGalleryDlArgs, + buildGalleryDlCaptureArgs, + CAPTURE_PRINT_TAG, galleryDlCookieChoice, + parseCapturePrints, + xRequestPauseMs, OLDER_WINDOW_PAUSE_MAX_MS, OLDER_WINDOW_PAUSE_MIN_MS, olderWindowPauseMs, @@ -336,3 +340,66 @@ test("the pause between search windows is a fresh random draw in 45–120 s", () assert.ok(mid > OLDER_WINDOW_PAUSE_MIN_MS && mid < OLDER_WINDOW_PAUSE_MAX_MS); assert.ok(OLDER_WINDOW_PAUSE_MIN_MS >= 45_000); }); + +// The post capture's media download: the one gallery-dl run that DOWNLOADS. +test("capture argv: downloads (no --no-download, no --dump-json), videos on, into exactly the post's dir", () => { + const argv = buildGalleryDlCaptureArgs({ id: "1234567890", dir: "/corpus/ch/posts-media/1234567890" }); + assert.ok(!argv.includes("--no-download")); + assert.ok(!argv.includes("--dump-json")); + assert.ok(argv.includes("extractor.twitter.videos=true")); + assert.ok(!argv.includes("extractor.twitter.videos=false")); + assert.equal(flagValue(argv, "-D"), "/corpus/ch/posts-media/1234567890"); + assert.equal(flagValue(argv, "-f"), "{tweet_id}_{num}.{extension}"); + assert.equal(argv[argv.length - 1], "https://x.com/i/status/1234567890"); + // A tagged line per file downloaded, and per file already there. + const prints = argv.filter((_, i) => argv[i - 1] === "--Print"); + assert.deepEqual(prints, [ + `after:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, + `skip:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, + ]); +}); + +test("capture argv: paced and logged in exactly as the reads are", () => { + const read = buildGalleryDlArgs({ accountUrl: ACCOUNT, cookieFile: JAR }); + const capture = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookieFile: JAR }); + for (const flag of [ + `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`, + "extractor.twitter.ratelimit=wait", + ]) { + assert.ok(read.includes(flag) && capture.includes(flag), flag); + } + assert.equal(flagValue(capture, "--cookies"), JAR); + const browser = buildGalleryDlCaptureArgs({ id: "42", dir: "/d", cookies: "firefox" }); + assert.equal(flagValue(browser, "--cookies-from-browser"), "firefox"); + assert.ok(!browser.includes("--cookies")); + const guest = buildGalleryDlCaptureArgs({ id: "42", dir: "/d" }); + assert.ok(!guest.includes("--cookies") && !guest.includes("--cookies-from-browser")); +}); + +test("capture argv: an id that is not X's digits is refused, not turned into a URL", () => { + for (const id of ["../etc", "12a", "", "1 2"]) { + assert.throws(() => buildGalleryDlCaptureArgs({ id, dir: "/d" }), /is not a post id/); + } +}); + +test("capture prints: tagged lines name each file once, with its source URL; the rest is ignored", () => { + const out = [ + "/corpus/ch/posts-media/42/42_1.jpg", + `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`, + `${CAPTURE_PRINT_TAG}\thttps://video.twimg.com/v/clip.mp4\t/corpus/ch/posts-media/42/42_2.mp4`, + `${CAPTURE_PRINT_TAG}\thttps://pbs.twimg.com/media/abc?format=jpg&name=orig\t/corpus/ch/posts-media/42/42_1.jpg`, + `${CAPTURE_PRINT_TAG}\tNone\t/corpus/ch/posts-media/42/42_3.png`, + "# /corpus/ch/posts-media/42/42_1.jpg", + ].join("\n"); + assert.deepEqual(parseCapturePrints(out), [ + { name: "42_1.jpg", url: "https://pbs.twimg.com/media/abc?format=jpg&name=orig" }, + { name: "42_2.mp4", url: "https://video.twimg.com/v/clip.mp4" }, + { name: "42_3.png" }, + ]); +}); + +test("the capture's gap before each contact is the reads' 4–10 s, drawn fresh", () => { + assert.equal(xRequestPauseMs(() => 0), 4_000); + assert.equal(xRequestPauseMs(() => 1), 10_000); + assert.equal(xRequestPauseMs(() => 0.5), 7_000); +}); diff --git a/common/social/xGalleryDlFetcher.ts b/common/social/xGalleryDlFetcher.ts @@ -24,6 +24,8 @@ // `--dump-json` contract and are covered by a fixture binary in the editor e2e // suite; treat the first real run as the spike. +import { existsSync } from "node:fs"; +import path from "node:path"; import { createInterface } from "node:readline"; import type { Readable } from "node:stream"; import { execa } from "execa"; @@ -32,12 +34,24 @@ import type { Post } from "../lib/posts"; import type { OlderBackfillPosition } from "../lib/posts-server"; import { normalizeXTweet, xCreatedAt, type XTweetRaw } from "./xNormalize"; import { olderFloorDay, stepOlderWindow } from "./olderBackfill"; -import { readXSessionStatus, xCookieFile } from "./xSessionBroker"; +import { + launchXProfile, + readXSessionStatus, + xCookieFile, +} from "./xSessionBroker"; import type { XCookieSource } from "./xCookieSource"; +import { listCapturedMediaFiles } from "./postCapture"; +import { + captureXPosts, + xStatusUrl, + type MediaDownloadResult, +} from "./xPostCapture"; import { registerSocialFetcher, type OlderPostFetchInput, type OlderPostFetchResult, + type PostCaptureInput, + type PostCaptureResult, type PostFetchInput, type PostFetchResult, type SocialFetcher, @@ -148,35 +162,200 @@ function galleryDlCommonArgs(opts: GalleryDlLoginAndCap): string[] { "extractor.twitter.videos=false", "-o", "extractor.twitter.cards=false", - // PACING. gallery-dl's default gap between X API requests is 0: it pages as - // fast as X answers and only slows down when X's rate-limit headers say to. - // An account-history walk is hundreds of requests, so every read waits a - // RANDOM 4–10 s before each one (gallery-dl's own "a-b" range, a uniform - // draw per request): no fixed rhythm, and well under the rate a person - // scrolling would make. Running into the limit anyway is waited out, never - // pushed through ("wait" is gallery-dl's default; stated so it stays). + ...galleryDlPacingArgs(), + ...galleryDlLoginArgs(opts), + ]; + if (opts.limit && opts.limit > 0) { + args.push("--range", `1-${Math.floor(opts.limit)}`); + } + return args; +} + +// PACING. gallery-dl's default gap between X API requests is 0: it pages as +// fast as X answers and only slows down when X's rate-limit headers say to. An +// account-history walk is hundreds of requests, so every read waits a RANDOM +// 4–10 s before each one (gallery-dl's own "a-b" range, a uniform draw per +// request): no fixed rhythm, and well under the rate a person scrolling would +// make. Running into the limit anyway is waited out, never pushed through +// ("wait" is gallery-dl's default; stated so it stays). Every gallery-dl run +// against X carries these — the reads and the post capture alike. +function galleryDlPacingArgs(): string[] { + return [ "-o", `extractor.twitter.sleep-request=${X_SLEEP_REQUEST}`, "-o", "extractor.twitter.ratelimit=wait", ]; - // Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no - // account at all (verified: 500 tweets over ~7.5 months for a public - // account). Guest access is rate-limited far more aggressively than an - // authenticated session, though — gallery-dl will block for minutes on - // "Waiting for N minutes (rate limit)" — so credentials remain the path for - // a deep backfill. When none are configured we simply run as a guest rather - // than failing. - if (opts.cookieFile) { - args.push("--cookies", opts.cookieFile); - } else if (opts.cookies) { - // Same browser-cookie spec yt-dlp uses (cookiePolicy.ts), same syntax. - args.push("--cookies-from-browser", opts.cookies); +} + +// Cookies are OPTIONAL. gallery-dl reads X timelines on a guest token with no +// account at all (verified: 500 tweets over ~7.5 months for a public account). +// Guest access is rate-limited far more aggressively than an authenticated +// session, though — gallery-dl will block for minutes on "Waiting for N +// minutes (rate limit)" — so credentials remain the path for a deep backfill. +// When none are configured we simply run as a guest rather than failing. +function galleryDlLoginArgs(opts: GalleryDlLoginAndCap): string[] { + if (opts.cookieFile) return ["--cookies", opts.cookieFile]; + // Same browser-cookie spec yt-dlp uses (cookiePolicy.ts), same syntax. + if (opts.cookies) return ["--cookies-from-browser", opts.cookies]; + return []; +} + +// The random gap a post capture waits before each contact with X after its +// first (a page load, a gallery-dl run): the same 4–10 s draw as +// X_SLEEP_REQUEST, which paces gallery-dl's requests inside one run. +export function xRequestPauseMs(rand: () => number = Math.random): number { + const [min, max] = X_SLEEP_REQUEST.split("-").map((s) => Number(s) * 1000); + return Math.round(min + rand() * (max - min)); +} + +// --------------------------------------------------------------------------- +// Post capture: one post's attached media +// --------------------------------------------------------------------------- +// +// The media half of `captureByIds` (the screenshot half is xPostCapture.ts). +// One gallery-dl run per post, against the post's own URL, DOWNLOADING — the +// one gallery-dl run here without --no-download — into the post's capture +// directory, videos included, paced and logged in exactly as the reads are. +// gallery-dl prints a tagged line per file (downloaded, or already there), so +// each file's source URL is known without a second pass. + +export const CAPTURE_PRINT_TAG = "ARCHILYZER-CAPTURED"; + +export function buildGalleryDlCaptureArgs( + opts: Omit<GalleryDlLoginAndCap, "limit"> & { id: string; dir: string }, +): string[] { + if (!/^\d+$/.test(opts.id)) throw new Error(`"${opts.id}" is not a post id`); + return [ + // Videos too: a capture keeps what the post carried. + "-o", + "extractor.twitter.videos=true", + "-o", + "extractor.twitter.cards=false", + ...galleryDlPacingArgs(), + // Exactly this directory, and names that cannot meet shot.png or + // capture.json. + "-D", + opts.dir, + "-f", + "{tweet_id}_{num}.{extension}", + // A line per file once it is on disk, and per file already there (a + // forced re-capture): "<tag>\t<source url>\t<path>". + "--Print", + `after:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, + "--Print", + `skip:${CAPTURE_PRINT_TAG}\t{_url}\t{_path}`, + ...galleryDlLoginArgs(opts), + xStatusUrl(opts.id), + ]; +} + +// The files a capture run printed, by name (each once). +export function parseCapturePrints( + stdout: string, +): { name: string; url?: string }[] { + const byName = new Map<string, { name: string; url?: string }>(); + for (const line of stdout.split("\n")) { + const parts = line.replace(/\r$/, "").split("\t"); + if (parts[0] !== CAPTURE_PRINT_TAG || parts.length < 3) continue; + const file = parts.slice(2).join("\t").trim(); + if (!file) continue; + const name = path.basename(file); + const url = parts[1].trim(); + byName.set(name, { name, ...(url && url !== "None" ? { url } : {}) }); } - if (opts.limit && opts.limit > 0) { - args.push("--range", `1-${Math.floor(opts.limit)}`); + return [...byName.values()]; +} + +// How long one post's download may take: a long video at a polite pace. +const CAPTURE_MEDIA_TIMEOUT_MS = 20 * 60_000; + +export async function downloadXPostMedia(args: { + id: string; + dir: string; + login: { cookies?: string; cookieFile?: string }; + signal: AbortSignal; + onLog?: (line: string) => void; +}): Promise<MediaDownloadResult> { + const bin = getPaths().galleryDlBin; + const res = await execa( + bin, + buildGalleryDlCaptureArgs({ id: args.id, dir: args.dir, ...args.login }), + { + reject: false, + stdin: "ignore", + timeout: CAPTURE_MEDIA_TIMEOUT_MS, + cancelSignal: args.signal, + env: { PYTHONUNBUFFERED: "1" }, + }, + ); + const spawnError = `${res.code ?? ""} ${res.message ?? ""}`; + if (res.failed && /ENOENT/.test(spawnError) && !res.stdout) { + throw new Error(`gallery-dl not found (looked for "${bin}"; set GALLERY_DL_BIN)`); } - return args; + const stderr = `${res.stderr ?? ""}`; + for (const line of stderr.split("\n")) { + if (line.trim()) args.onLog?.(`gallery-dl: ${line.trim()}`); + } + if (res.isCanceled || res.timedOut || res.exitCode !== 0) { + const tail = stderr.trim().split("\n").slice(-5).join("\n"); + if (looksLikeAuthFailure(tail)) { + return { ok: false, needsCookies: true, error: `gallery-dl could not authenticate to X: ${tail}` }; + } + return { + ok: false, + error: res.isCanceled + ? "Cancelled during the media download." + : res.timedOut + ? "The media download ran past its time limit." + : `gallery-dl exited ${res.exitCode ?? res.signal ?? "abnormally"}: ${tail || "(no output)"}`, + }; + } + // The printed files, plus any on disk it did not print (an older + // gallery-dl): every one is in the record. + const printed = parseCapturePrints(`${res.stdout ?? ""}`); + const files = printed.filter((f) => existsSync(path.join(args.dir, f.name))); + const names = new Set(files.map((f) => f.name)); + for (const name of await listCapturedMediaFiles(args.dir)) { + if (!names.has(name)) files.push({ name }); + } + return { ok: true, files }; +} + +// `captureByIds` for X: the shot through the session profile, the media +// through gallery-dl, the run paced as one (xPostCapture.ts). +export async function captureXPostsByIds( + input: PostCaptureInput, +): Promise<PostCaptureResult> { + const paths = getPaths(); + if (input.shots ?? true) { + const status = await readXSessionStatus(paths); + if (!status.hasProfile) { + const why = + "No X session profile to shoot posts with: connect an X account on /settings, then run it again."; + input.onLog?.(`[auth] ${why}`); + return { outcomes: [], needsCookies: true, stoppedEarly: why }; + } + if (!status.looksAuthenticated) { + input.onLog?.( + "[warn] The X session profile's exported cookies carry no login; X may show its login wall.", + ); + } + } + let login: Promise<{ cookies?: string; cookieFile?: string }> | undefined; + return captureXPosts(input, { + openPage: async () => { + const context = await launchXProfile(paths, { onLog: input.onLog }); + const page = context.pages()[0] ?? (await context.newPage()); + return { page, close: () => context.close() }; + }, + downloadMedia: async ({ id, dir, signal, onLog }) => { + login ??= galleryDlLoginFor(input); + return downloadXPostMedia({ id, dir, login: await login, signal, onLog }); + }, + pauseMs: () => xRequestPauseMs(), + pause, + }); } // --------------------------------------------------------------------------- @@ -1081,6 +1260,8 @@ export const xGalleryDlFetcher: SocialFetcher = { }, fetchOlder: fetchOlderViaSearch, + + captureByIds: captureXPostsByIds, }; registerSocialFetcher(xGalleryDlFetcher); diff --git a/common/social/xPostCapture.test.ts b/common/social/xPostCapture.test.ts @@ -0,0 +1,436 @@ +// The X post capture, against fakes only: a page that answers the capture's +// evaluate calls from a recorded snapshot, a media downloader that writes +// files, and a fake gallery-dl binary for the real download path. No browser, +// no network. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test social/xPostCapture.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { createHash } from "node:crypto"; +import { chmod, mkdir, mkdtemp, readFile, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import type { PageLike } from "./playwrightRuntime"; +import { + captureXPosts, + classifyXPostSnapshot, + shootXPost, + type XCaptureDeps, + type XPostSnapshot, +} from "./xPostCapture"; +import { readPostCapture, SHOT_FILENAME, writePostCapture } from "./postCapture"; +import type { PostCaptureInput } from "./fetchers"; + +const sha = (b: string | Uint8Array) => createHash("sha256").update(b).digest("hex"); +const PNG = new Uint8Array([0x89, 0x50, 0x4e, 0x47, 9, 9]); +const RECT = { x: 10, y: 120.4, width: 598.6, height: 300.2 }; + +const post = (sensitive = false): XPostSnapshot => ({ + path: "/someone/status/1", + text: "the post", + article: { rect: RECT, sensitive }, +}); +const page = (text: string, p = "/i/status/1"): XPostSnapshot => ({ path: p, text, article: null }); + +// --- classification --------------------------------------------------------- + +test("classify: a rendered post is captured; its sensitive cover is noted", () => { + assert.deepEqual(classifyXPostSnapshot(post()), { state: "captured" }); + assert.deepEqual(classifyXPostSnapshot(post(true)), { state: "captured", sensitive: true }); +}); + +test("classify: a login redirect or a logged-out page is a login wall that stops the run", () => { + for (const s of [page("", "/i/flow/login"), page("", "/login"), page("Don't miss what's happening. Log in Sign up")]) { + const v = classifyXPostSnapshot(s); + assert.equal(v.state, "login-wall", JSON.stringify(s)); + assert.ok(v.stop); + } +}); + +test("classify: deleted and walled posts are told apart, and neither stops the run", () => { + for (const text of [ + "This Post was deleted by the Post author. Learn more", + "Hmm...this page doesn’t exist. Try searching for something else.", + ]) { + assert.deepEqual(classifyXPostSnapshot(page(text)), { state: "deleted" }, text); + } + for (const text of [ + "This Post is from an account that no longer exists. Learn more", + "This Post is from a suspended account. Learn more", + "You’re unable to view this Post because this account owner limits who can view their Posts.", + "This Post is unavailable.", + ]) { + assert.deepEqual(classifyXPostSnapshot(page(text)), { state: "unavailable" }, text); + } +}); + +test("classify: X refusing pages stops the run; an age wall and an empty page are errors that do not", () => { + const refused = classifyXPostSnapshot(page("Something went wrong. Try reloading. Retry")); + assert.equal(refused.state, "error"); + assert.ok(refused.stop); + const age = classifyXPostSnapshot(page("Age-restricted adult content.")); + assert.equal(age.state, "error"); + assert.equal(age.stop, undefined); + const blank = classifyXPostSnapshot(page("")); + assert.equal(blank.state, "error"); + assert.equal(blank.stop, undefined); +}); + +// --- the shot ---------------------------------------------------------------- + +type FakePage = PageLike & { calls: string[]; shots: unknown[] }; + +// A page whose snapshot evaluate returns `snaps` in turn (the last repeats); +// the sensitive-cover click flips a sensitive article to an open one. +function fakePage(snaps: XPostSnapshot[], opts: { gotoFails?: boolean } = {}): FakePage { + let i = 0; + const calls: string[] = []; + const shots: unknown[] = []; + return { + calls, + shots, + goto: async (url: string) => { + calls.push(`goto ${url}`); + if (opts.gotoFails) throw new Error("net::ERR_TIMED_OUT\nat goto"); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: async (script: string) => { + if (script.includes("b.click()")) { + calls.push("open-sensitive"); + return 1; + } + if (script.includes("imgs.length")) return 0; + calls.push("snapshot"); + const s = snaps[Math.min(i, snaps.length - 1)]; + i++; + return s; + }, + screenshot: async (o?: unknown) => { + shots.push(o); + return PNG; + }, + }; +} + +test("shot: the post's own box, clipped from a full-page PNG, hashed beside it", async () => { + const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1"); + const p = fakePage([post()]); + const res = await shootXPost(p, "1", dir); + assert.equal(res.state, "captured"); + assert.equal(p.calls[0], "goto https://x.com/i/status/1"); + assert.deepEqual(p.shots, [ + { type: "png", fullPage: true, clip: { x: 10, y: 120, width: 599, height: 301 } }, + ]); + assert.deepEqual(new Uint8Array(await readFile(path.join(dir, SHOT_FILENAME))), PNG); + assert.deepEqual(res.shot, { + name: SHOT_FILENAME, + bytes: PNG.length, + sha256: sha(PNG), + url: "https://x.com/i/status/1", + }); +}); + +test("shot: a sensitive cover is opened before the shot", async () => { + const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1"); + const p = fakePage([post(true), post(false)]); + const res = await shootXPost(p, "1", dir); + assert.equal(res.state, "captured"); + assert.equal(res.sensitive, true); + assert.ok(p.calls.includes("open-sensitive")); + assert.equal(p.shots.length, 1); +}); + +test("shot: a deleted post, a login wall or a failed load writes no shot", async () => { + for (const [p, state] of [ + [fakePage([page("This post was deleted by the post author.")]), "deleted"], + [fakePage([page("", "/i/flow/login")]), "login-wall"], + [fakePage([post()], { gotoFails: true }), "error"], + ] as const) { + const dir = path.join(await mkdtemp(path.join(os.tmpdir(), "xshot-")), "1"); + const res = await shootXPost(p, "1", dir); + assert.equal(res.state, state); + assert.equal(res.shot, undefined); + assert.equal(p.shots.length, 0); + } +}); + +// --- the run ----------------------------------------------------------------- + +type Harness = { + deps: XCaptureDeps; + pauses: number[]; + opened: number; + closed: number; + downloads: string[]; +}; + +// One snapshot per post id; the downloader writes `<id>_1.jpg` unless the id +// is in `noMedia`. +function harness( + snapsById: Record<string, XPostSnapshot[]>, + opts: { noMedia?: string[]; mediaFails?: { id: string; needsCookies?: boolean } } = {}, +): Harness { + const h: Harness = { deps: undefined as never, pauses: [], opened: 0, closed: 0, downloads: [] }; + let current = ""; + const pages = Object.fromEntries(Object.entries(snapsById).map(([id, s]) => [id, fakePage(s)])); + const routed: PageLike = { + goto: async (url: string, o?: unknown) => { + current = url.split("/").pop()!; + return pages[current].goto(url, o); + }, + waitForSelector: async () => undefined, + waitForTimeout: async () => undefined, + on: () => {}, + evaluate: (s: string) => pages[current].evaluate(s), + screenshot: (o) => pages[current].screenshot(o), + }; + h.deps = { + openPage: async () => { + h.opened++; + return { page: routed, close: async () => void h.closed++ }; + }, + downloadMedia: async ({ id, dir }) => { + h.downloads.push(id); + if (opts.mediaFails?.id === id) { + return { ok: false, error: "gallery-dl exited 4", needsCookies: opts.mediaFails.needsCookies }; + } + if (opts.noMedia?.includes(id)) return { ok: true, files: [] }; + await writeFile(path.join(dir, `${id}_1.jpg`), `media of ${id}`); + return { ok: true, files: [{ name: `${id}_1.jpg`, url: `https://pbs.example/${id}.jpg` }] }; + }, + pauseMs: () => 5_000, + pause: async (ms) => void h.pauses.push(ms), + now: () => new Date("2026-02-03T04:05:06.000Z"), + }; + return h; +} + +async function input(ids: string[], over: Partial<PostCaptureInput> = {}): Promise<PostCaptureInput> { + return { + ids, + handle: "example_user", + outDir: await mkdtemp(path.join(os.tmpdir(), "posts-media-")), + signal: new AbortController().signal, + ...over, + }; +} + +test("run: shot then media per post, a paced gap before every contact after the first, a record per post", async () => { + const h = harness({ "11": [post()], "22": [post()] }); + const inp = await input(["11", "22"]); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.stoppedEarly, undefined); + assert.deepEqual(res.outcomes, [ + { id: "11", state: "captured", availability: "available", files: 2 }, + { id: "22", state: "captured", availability: "available", files: 2 }, + ]); + // Four contacts (two pages, two downloads): three gaps. + assert.deepEqual(h.pauses, [5_000, 5_000, 5_000]); + assert.equal(h.opened, 1); + assert.equal(h.closed, 1); + const rec = await readPostCapture(path.join(inp.outDir, "11")); + assert.deepEqual(rec, { + version: 1, + id: "11", + url: "https://x.com/i/status/11", + capturedAt: "2026-02-03T04:05:06.000Z", + state: "captured", + shot: { name: SHOT_FILENAME, bytes: PNG.length, sha256: sha(PNG), url: "https://x.com/i/status/11" }, + mediaState: "ok", + media: [ + { + name: "11_1.jpg", + bytes: "media of 11".length, + sha256: sha("media of 11"), + url: "https://pbs.example/11.jpg", + }, + ], + }); +}); + +test("run: posts already captured are skipped unless forced; a missing half is all that is redone", async () => { + const h = harness({ "11": [post()], "22": [post()] }); + const inp = await input(["11", "22"]); + await captureXPosts(inp, h.deps); + h.downloads.length = 0; + + const again = harness({ "11": [post()], "22": [post()] }); + const res = await captureXPosts(inp, again.deps); + assert.deepEqual(res.outcomes, []); + assert.equal(again.opened, 0, "no browser for a run with nothing to shoot"); + assert.deepEqual(again.pauses, []); + + // A post whose media failed earlier: only the media is fetched again. + const rec = (await readPostCapture(path.join(inp.outDir, "22")))!; + await writePostCapture(path.join(inp.outDir, "22"), { ...rec, mediaState: "error", media: [] }); + const retry = harness({ "11": [post()], "22": [post()] }); + const r2 = await captureXPosts(inp, retry.deps); + assert.equal(retry.opened, 0); + assert.deepEqual(retry.downloads, ["22"]); + assert.equal(r2.outcomes.length, 1); + assert.equal(r2.outcomes[0].availability, undefined, "a media-only pass says nothing about liveness"); + const kept = (await readPostCapture(path.join(inp.outDir, "22")))!; + assert.equal(kept.shot?.sha256, sha(PNG), "the shot on disk is kept"); + + const forced = harness({ "11": [post()], "22": [post()] }); + const r3 = await captureXPosts({ ...inp, force: true }, forced.deps); + assert.equal(r3.outcomes.length, 2); + assert.deepEqual(forced.downloads, ["11", "22"]); +}); + +test("run: a deleted or walled post is recorded, gets no download, and the run goes on", async () => { + const h = harness({ + "11": [page("This post was deleted by the post author.")], + "22": [page("These posts are protected.")], + "33": [post()], + }, { noMedia: ["33"] }); + const inp = await input(["11", "22", "33"]); + const res = await captureXPosts(inp, h.deps); + assert.deepEqual( + res.outcomes.map((o) => [o.id, o.state, o.availability, o.files]), + [ + ["11", "deleted", "deleted", 0], + ["22", "unavailable", "account_unavailable", 0], + ["33", "captured", "available", 1], + ], + ); + assert.deepEqual(h.downloads, ["33"]); + const deleted = (await readPostCapture(path.join(inp.outDir, "11")))!; + assert.equal(deleted.state, "deleted"); + assert.equal(deleted.shot, undefined); + assert.equal(deleted.mediaState, "skipped"); + assert.equal((await readPostCapture(path.join(inp.outDir, "33")))!.mediaState, "none"); + // A deleted post is settled: the next run does not ask X again. + const next = harness({ "11": [post()], "22": [post()], "33": [post()] }); + const r2 = await captureXPosts(inp, next.deps); + assert.deepEqual(r2.outcomes.map((o) => o.id), ["22"]); +}); + +test("run: a login wall stops the run at that post — recorded, no availability verdict, the rest untouched", async () => { + const h = harness({ "11": [post()], "22": [page("", "/i/flow/login")], "33": [post()] }); + const inp = await input(["11", "22", "33"]); + const res = await captureXPosts(inp, h.deps); + assert.equal(res.needsCookies, true); + assert.match(res.stoppedEarly ?? "", /Stopped at 22/); + assert.deepEqual(res.outcomes.map((o) => [o.id, o.state, o.availability]), [ + ["11", "captured", "available"], + ["22", "login-wall", undefined], + ]); + assert.equal((await readPostCapture(path.join(inp.outDir, "22")))!.state, "login-wall"); + assert.equal(await readPostCapture(path.join(inp.outDir, "33")), null); + assert.equal(h.closed, 1, "the browser is closed on the way out"); +}); + +test("run: a media download refused for want of a login stops the run with needsCookies", async () => { + const h = harness({ "11": [post()], "22": [post()] }, { mediaFails: { id: "11", needsCookies: true } }); + const res = await captureXPosts(await input(["11", "22"]), h.deps); + assert.equal(res.needsCookies, true); + assert.deepEqual(res.outcomes.map((o) => o.id), ["11"]); + assert.equal(res.outcomes[0].state, "captured", "the shot stands"); +}); + +test("run: three failures in a row stop it rather than paging through the rest", async () => { + const h = harness({ "1": [page("")], "2": [page("")], "3": [page("")], "4": [post()] }); + const res = await captureXPosts(await input(["1", "2", "3", "4"]), h.deps); + assert.deepEqual(res.outcomes.map((o) => o.state), ["error", "error", "error"]); + assert.match(res.stoppedEarly ?? "", /3 posts in a row failed/); +}); + +test("run: a cancel or a drain stops between posts", async () => { + const ac = new AbortController(); + const h = harness({ "11": [post()], "22": [post()] }); + const inner = h.deps.downloadMedia; + h.deps.downloadMedia = async (a) => { + const r = await inner(a); + ac.abort(); + return r; + }; + const res = await captureXPosts(await input(["11", "22"], { signal: ac.signal }), h.deps); + assert.deepEqual(res.outcomes.map((o) => o.id), ["11"]); + assert.match(res.stoppedEarly ?? "", /Cancelled/); + + const drain = new AbortController(); + drain.abort(); + const d = harness({ "11": [post()] }); + const r2 = await captureXPosts(await input(["11"], { drain: drain.signal }), d.deps); + assert.deepEqual(r2.outcomes, []); + assert.match(r2.stoppedEarly ?? "", /Drained/); + assert.equal(d.opened, 0); +}); + +test("run: media only, or shots only, does only that half", async () => { + const h = harness({ "11": [post()] }); + const inp = await input(["11"], { shots: false }); + const res = await captureXPosts(inp, h.deps); + assert.equal(h.opened, 0); + assert.deepEqual(h.downloads, ["11"]); + assert.deepEqual(res.outcomes, [{ id: "11", state: "captured", files: 1 }]); + + const s = harness({ "11": [post()] }); + const r2 = await captureXPosts(await input(["11"], { media: false }), s.deps); + assert.deepEqual(s.downloads, []); + assert.deepEqual(r2.outcomes, [{ id: "11", state: "captured", availability: "available", files: 1 }]); +}); + +// --- the real download path, over a fake gallery-dl ------------------------- + +test("media download: gallery-dl's tagged lines and the files it wrote, nothing from the network", async () => { + const root = await mkdtemp(path.join(os.tmpdir(), "xmedia-")); + const bin = path.join(root, "fake-gallery-dl.mjs"); + const argsLog = path.join(root, "argv.json"); + await writeFile( + bin, + `#!/usr/bin/env node +import { writeFileSync } from "node:fs"; +import path from "node:path"; +const args = process.argv.slice(2); +writeFileSync(process.env.FAKE_ARGS_LOG, JSON.stringify(args)); +const dir = args[args.indexOf("-D") + 1]; +const tag = "ARCHILYZER-CAPTURED"; +if (process.env.FAKE_MODE === "auth") { + process.stderr.write("[twitter][error] 401 Unauthorized\\n"); + process.exit(4); +} +const file = path.join(dir, "77_1.jpg"); +writeFileSync(file, "pixels"); +process.stdout.write(file + "\\n"); +process.stdout.write(tag + "\\thttps://pbs.example/77.jpg\\t" + file + "\\n"); +`, + ); + await chmod(bin, 0o755); + // Paths are read (and cached) on first use: point them at the temp root + // before anything asks, so no real corpus is in reach. + process.env.TRANSCRIPTS_DIR = path.join(root, "transcripts"); + process.env.GALLERY_DL_BIN = bin; + process.env.FAKE_ARGS_LOG = argsLog; + const { downloadXPostMedia } = await import("./xGalleryDlFetcher"); + const dir = path.join(root, "posts-media", "77"); + await mkdir(dir, { recursive: true }); + const ok = await downloadXPostMedia({ + id: "77", + dir, + login: { cookieFile: "/jar.txt" }, + signal: new AbortController().signal, + }); + assert.deepEqual(ok, { ok: true, files: [{ name: "77_1.jpg", url: "https://pbs.example/77.jpg" }] }); + const argv = JSON.parse(await readFile(argsLog, "utf8")) as string[]; + assert.ok(!argv.includes("--no-download")); + assert.equal(argv[argv.indexOf("--cookies") + 1], "/jar.txt"); + + process.env.FAKE_MODE = "auth"; + try { + const refused = await downloadXPostMedia({ + id: "77", + dir, + login: {}, + signal: new AbortController().signal, + }); + assert.equal(refused.ok, false); + assert.equal(!refused.ok && refused.needsCookies, true); + } finally { + delete process.env.FAKE_MODE; + } +}); diff --git a/common/social/xPostCapture.ts b/common/social/xPostCapture.ts @@ -0,0 +1,402 @@ +// Capturing specific X posts: a screenshot of each post as X renders it, +// through the logged-in session profile, and its attached media, downloaded by +// gallery-dl (xGalleryDlFetcher.ts owns that half and wires both into +// `captureByIds`). +// +// PACED LIKE EVERY OTHER X READ. A capture is a page load and a gallery-dl run +// per post, and each is a contact with X: the run waits a random 4–10 s before +// every contact after its first (gallery-dl then paces its own requests the +// same way). A page that says X is refusing — "Something went wrong" — or a +// login wall stops the run there: the next post would meet the same answer, +// and asking again at once is exactly the burst the pacing exists to avoid. +// The ids not reached are left for a later run. +// +// What a page can say instead of the post, each recorded, none retried in the +// run that met it: +// - a login wall (a redirect to the login flow, or a logged-out page with no +// post): the session's state, not the post's — the run stops, needsCookies; +// - a sensitive-media interstitial: opened ("Show"), then shot, and the +// record says it was there; +// - "this post was deleted" / "this page doesn't exist": deleted; +// - a protected, suspended or vanished account, or a withheld post: +// unavailable. +// +// NOT VERIFIED AGAINST LIVE X. The page markers below are X's as of this +// writing and are tested against recorded snapshots, never x.com; the first +// real run is the check. + +import { mkdir, writeFile } from "node:fs/promises"; +import path from "node:path"; +import type { PageLike } from "./playwrightRuntime"; +import type { + PostCaptureInput, + PostCaptureOutcome, + PostCaptureResult, +} from "./fetchers"; +import { + captureAvailability, + captureWork, + describeCapturedFile, + postCaptureDir, + readPostCapture, + SHOT_FILENAME, + writePostCapture, + type CapturedFile, + type CaptureMediaState, + type PostCaptureRecord, + type PostCaptureState, +} from "./postCapture"; + +export function xStatusUrl(id: string): string { + return `https://x.com/i/status/${id}`; +} + +// What the page showed, read in one evaluate. `article` is the post itself +// (its box in document coordinates, for the clip), when it rendered. +export type XPostSnapshot = { + path: string; + text: string; + article: null | { + rect: { x: number; y: number; width: number; height: number }; + // A "Show" / "View" button inside the post: a sensitive-media cover. + sensitive: boolean; + }; +}; + +// The post's own article: the one whose timestamp links to this id (a reply's +// parents render above it as articles too), else X's focal article. +const SNAPSHOT_SCRIPT = (id: string) => `(() => { + const id = ${JSON.stringify(id)}; + const articles = Array.from(document.querySelectorAll('article[data-testid="tweet"]')); + const own = articles.find((a) => + Array.from(a.querySelectorAll('a[href*="/status/"]')).some((l) => { + const m = /\\/status\\/(\\d+)/.exec(l.getAttribute("href") || ""); + return m && m[1] === id && l.querySelector("time"); + }), + ) || articles.find((a) => a.getAttribute("tabindex") === "-1") || null; + const main = document.querySelector('[data-testid="primaryColumn"]') || document.body; + const text = ((main && main.innerText) || "").slice(0, 4000); + let article = null; + if (own) { + own.setAttribute("data-archilyzer-capture", ""); + const r = own.getBoundingClientRect(); + const sensitive = Array.from(own.querySelectorAll('button, [role="button"]')).some( + (b) => /^(show|view)$/i.test((b.innerText || "").trim()), + ); + article = { + rect: { x: r.left + window.scrollX, y: r.top + window.scrollY, width: r.width, height: r.height }, + sensitive, + }; + } + return { path: location.pathname, text, article }; +})()`; + +// Open a sensitive-media cover inside the post. +const OPEN_SENSITIVE_SCRIPT = `(() => { + const own = document.querySelector('[data-archilyzer-capture]'); + if (!own) return 0; + let n = 0; + for (const b of own.querySelectorAll('button, [role="button"]')) { + if (/^(show|view)$/i.test((b.innerText || "").trim())) { b.click(); n++; } + } + return n; +})()`; + +// Let the post's images finish loading (each capped), so the shot is not of +// grey boxes. +const IMAGES_LOADED_SCRIPT = `(() => { + const own = document.querySelector('[data-archilyzer-capture]') || document; + const imgs = Array.from(own.querySelectorAll("img")).filter((i) => !i.complete); + return Promise.all(imgs.map((i) => new Promise((r) => { + i.addEventListener("load", r, { once: true }); + i.addEventListener("error", r, { once: true }); + setTimeout(r, 5000); + }))).then(() => imgs.length); +})()`; + +const DELETED_TEXT = [ + /this (post|tweet) was deleted/i, + /this page doesn.t exist/i, +]; +const UNAVAILABLE_TEXT = [ + /account (that )?no longer exists/i, + /suspended account/i, + /account (is )?suspended/i, + /these (posts|tweets) are protected/i, + /limits who can view their (posts|tweets)/i, + /withheld in/i, + /this (post|tweet) is unavailable/i, +]; +const AGE_WALL_TEXT = /age-restricted/i; +const REFUSED_TEXT = /something went wrong\. try reloading|rate limit exceeded/i; +const LOGGED_OUT_TEXT = /(log in|sign in|sign up)/i; + +export type XPostVerdict = { + state: PostCaptureState; + sensitive?: boolean; + // Set when the run must stop here: the next post would meet the same page. + stop?: string; + error?: string; +}; + +// What a snapshot means. Pure, so every marker is testable without a browser. +export function classifyXPostSnapshot(s: XPostSnapshot): XPostVerdict { + if (/^\/(i\/flow\/login|login|i\/flow\/signup)\b/.test(s.path)) { + return { + state: "login-wall", + stop: "X asked to log in — the session profile is not logged in.", + }; + } + if (s.article) { + return { state: "captured", ...(s.article.sensitive ? { sensitive: true } : {}) }; + } + if (REFUSED_TEXT.test(s.text)) { + return { + state: "error", + error: "X answered “Something went wrong” instead of the post.", + stop: "X is refusing pages right now; stopping rather than asking again.", + }; + } + if (DELETED_TEXT.some((re) => re.test(s.text))) return { state: "deleted" }; + if (UNAVAILABLE_TEXT.some((re) => re.test(s.text))) return { state: "unavailable" }; + if (AGE_WALL_TEXT.test(s.text)) { + return { + state: "error", + error: "X shows this post only to an age-verified session.", + }; + } + if (LOGGED_OUT_TEXT.test(s.text)) { + return { + state: "login-wall", + stop: "X showed its logged-out page instead of the post.", + }; + } + return { state: "error", error: "The post did not render." }; +} + +export type XShotResult = XPostVerdict & { shot?: CapturedFile }; + +// One post's screenshot: load, read the page, open a sensitive cover, shoot +// the post's own article. Writes `shot.png` into `dir` only for a post that +// rendered. +export async function shootXPost( + page: PageLike, + id: string, + dir: string, + onLog?: (line: string) => void, +): Promise<XShotResult> { + const url = xStatusUrl(id); + try { + await page.goto(url, { waitUntil: "domcontentloaded", timeout: 60_000 }); + } catch (err) { + return { state: "error", error: `Could not load ${url}: ${firstLine(err)}` }; + } + // The post, or whatever X shows instead — which has no stable marker, so a + // missing article is waited out and the page read anyway. + await page + .waitForSelector('article[data-testid="tweet"]', { timeout: 20_000 }) + .catch(() => {}); + await page.waitForTimeout(1_500); + let snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot; + const verdict = classifyXPostSnapshot(snap); + if (verdict.state !== "captured") return verdict; + + if (verdict.sensitive) { + onLog?.(`${id}: a sensitive-media cover — opening it before the shot.`); + await page.evaluate(OPEN_SENSITIVE_SCRIPT); + await page.waitForTimeout(1_500); + snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot; + if (!snap.article) { + return { state: "error", error: "The post vanished after its sensitive-media cover was opened." }; + } + } + await page.evaluate(IMAGES_LOADED_SCRIPT).catch(() => {}); + // Re-read the box: images that loaded may have grown it. + snap = (await page.evaluate(SNAPSHOT_SCRIPT(id))) as XPostSnapshot; + const rect = snap.article?.rect; + if (!rect || rect.width < 1 || rect.height < 1) { + return { state: "error", error: "The post rendered with no size to shoot." }; + } + const clip = { + x: Math.max(0, Math.floor(rect.x)), + y: Math.max(0, Math.floor(rect.y)), + width: Math.ceil(rect.width), + height: Math.ceil(rect.height), + }; + let png: Uint8Array; + try { + // fullPage, so a post taller than the window is shot whole; the clip is in + // document coordinates, which is what the snapshot measured. + png = await page.screenshot({ type: "png", fullPage: true, clip }); + } catch (err) { + return { state: "error", error: `The screenshot failed: ${firstLine(err)}` }; + } + await mkdir(dir, { recursive: true }); + await writeFile(path.join(dir, SHOT_FILENAME), png); + const shot = await describeCapturedFile(dir, SHOT_FILENAME, url); + return { ...verdict, shot }; +} + +export type MediaDownloadResult = + | { ok: true; files: { name: string; url?: string }[] } + | { ok: false; error: string; needsCookies?: boolean }; + +export type XCaptureDeps = { + // A page in the logged-in profile. Opened on the first shot, closed at the + // end of the run. + openPage: () => Promise<{ page: PageLike; close: () => Promise<void> }>; + // One post's media into `dir` (gallery-dl in production). + downloadMedia: (args: { + id: string; + dir: string; + signal: AbortSignal; + onLog?: (line: string) => void; + }) => Promise<MediaDownloadResult>; + // The gap before each contact with X after the first. + pauseMs: () => number; + pause: (ms: number, signal: AbortSignal) => Promise<void>; + now?: () => Date; +}; + +// A run's errors in a row that stop it: a host whose network or browser is +// failing every post should not page through the rest of the list. +const STOP_AFTER_ERRORS = 3; + +// The capture loop: per id, what is owed (captureWork), the shot, the media, +// the record. Every post's record is written as soon as that post is done, so +// a cancel or a crash loses at most the post in hand. +export async function captureXPosts( + input: PostCaptureInput, + deps: XCaptureDeps, +): Promise<PostCaptureResult> { + const { signal, onLog } = input; + const wanted = { + shots: input.shots ?? true, + media: input.media ?? true, + force: input.force ?? false, + }; + const now = deps.now ?? (() => new Date()); + const outcomes: PostCaptureOutcome[] = []; + let contacts = 0; + let errorsInARow = 0; + let browser: { page: PageLike; close: () => Promise<void> } | undefined; + + const contact = async () => { + if (contacts++ > 0) await deps.pause(deps.pauseMs(), signal); + }; + const stopped = (why: string, extra: Partial<PostCaptureResult> = {}) => { + onLog?.(why); + return { outcomes, stoppedEarly: why, ...extra }; + }; + + try { + for (const [i, id] of input.ids.entries()) { + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + if (input.drain?.aborted) return stopped("Drained; the rest are left for a later run."); + const dir = postCaptureDir(input.outDir, id); + const existing = await readPostCapture(dir); + const work = captureWork(existing, wanted); + if (!work.shot && !work.media) { + onLog?.(`${id}: already captured (${existing?.state ?? "nothing asked for"}) — skipped.`); + continue; + } + onLog?.(`[${i + 1}/${input.ids.length}] ${id}`); + + let state: PostCaptureState | undefined = work.shot ? undefined : existing?.state; + let sensitive = work.shot ? undefined : existing?.sensitive; + let shot = work.shot ? undefined : existing?.shot; + let error: string | undefined; + let stop: string | undefined; + + if (work.shot) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + browser ??= await deps.openPage(); + const res = await shootXPost(browser.page, id, dir, onLog); + state = res.state; + sensitive = res.sensitive; + shot = res.shot; + error = res.error; + stop = res.stop; + } + + let mediaState: CaptureMediaState = work.media ? "skipped" : (existing?.mediaState ?? "skipped"); + let media: CapturedFile[] = work.media ? [] : (existing?.media ?? []); + let needsCookies = false; + // Only a post that rendered (or was not shot this run) is worth a + // download: X has nothing to give for a deleted or walled one. + const postIsThere = state === undefined || state === "captured"; + if (work.media && postIsThere && !stop) { + await contact(); + if (signal.aborted) return stopped("Cancelled; the rest are left for a later run."); + await mkdir(dir, { recursive: true }); + const got = await deps.downloadMedia({ id, dir, signal, onLog }); + if (got.ok) { + media = []; + for (const f of got.files) media.push(await describeCapturedFile(dir, f.name, f.url)); + mediaState = media.length > 0 ? "ok" : "none"; + state ??= "captured"; + } else { + mediaState = "error"; + error = error ? `${error}; ${got.error}` : got.error; + state ??= "error"; + if (got.needsCookies) { + needsCookies = true; + stop = "gallery-dl could not log in to X for the media."; + } + } + } + + const finalState: PostCaptureState = state ?? "error"; + const record: PostCaptureRecord = { + version: 1, + id, + url: xStatusUrl(id), + capturedAt: now().toISOString(), + state: finalState, + ...(sensitive ? { sensitive: true } : {}), + ...(shot ? { shot } : {}), + mediaState, + media, + ...(error ? { error } : {}), + }; + await writePostCapture(dir, record); + outcomes.push({ + id, + state: finalState, + // Only the page says whether the post is up: a media-only run has no + // verdict to record. + ...(work.shot && captureAvailability(finalState) + ? { availability: captureAvailability(finalState) } + : {}), + files: (shot ? 1 : 0) + media.length, + ...(error ? { error } : {}), + }); + onLog?.( + `${id}: ${finalState}${sensitive ? " (behind a sensitive-media cover)" : ""}` + + (shot ? ", shot" : "") + + (mediaState === "ok" ? `, ${media.length} media file(s)` : mediaState === "none" ? ", no media" : "") + + (error ? ` — ${error}` : ""), + ); + + if (stop) { + return stopped(`${stop} Stopped at ${id}; the rest are left for a later run.`, { + needsCookies: needsCookies || finalState === "login-wall", + }); + } + errorsInARow = finalState === "error" ? errorsInARow + 1 : 0; + if (errorsInARow >= STOP_AFTER_ERRORS) { + return stopped( + `${STOP_AFTER_ERRORS} posts in a row failed; stopping rather than paging through the rest.`, + ); + } + } + return { outcomes }; + } finally { + await browser?.close().catch(() => {}); + } +} + +function firstLine(err: unknown): string { + return ((err as Error)?.message ?? String(err)).split("\n")[0]; +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Capture specific X posts: a screenshot of each, and its attached media.** `pnpm ops capture-posts --json '{"slug":"<channel>","ids":["<post id>", …]}'` shoots each post as X shows it, through the connected X profile, and downloads its pictures and videos with gallery-dl, into the channel's `posts-media/<post id>/` beside a `capture.json` that records when, from which URLs, and each file's size and SHA-256. Every id must already be in the channel's posts archive; one that is not is refused by name and nothing runs. `"shots": false` or `"media": false` skips that half, and posts already captured are skipped unless `"force": true`. The job runs on the X queue with a post fetch, so the two never run at once, and waits a random 4–10 seconds before each request to X, as fetches do. A deleted post, or one behind its account's wall (protected, suspended, gone), is recorded as such in the channel's deleted-post record; a post behind a sensitive-media warning is opened and shot. If X asks to log in, or answers "Something went wrong", the job stops at that post and leaves the rest for a later run. Captures are never published: the export does not read them. - **X posts are fetched more slowly, with random gaps.** Every read of X now waits a random 4 to 10 seconds before each request to X, where it used to page as fast as X answered, and always waits out a rate limit rather than pushing through. When fetching older posts, the pause between one three-month window and the next is a random 45 to 120 seconds instead of a fixed 15. A deep walk of an account's history takes longer; a routine fetch of new posts takes a few seconds more. - **The MCP's search tools take `date_from` and `date_to` as `2024-10-26` as well as `20241026`, and refuse a date they cannot read.** `search_transcripts` and `enumerate_matches` used to accept only `YYYYMMDD`: any other spelling was dropped with a footer warning and the search ran with no date bound, so a whole-corpus count could be read as the bounded one. Dashed, slashed and dotted dates and ISO timestamps are now normalised, and anything else is an error and nothing is searched. - **An X channel can fetch posts older than its timeline reaches.** X's timeline only pages back so far, so a fetch could end, and call the history done, well short of an account's first post. The new **Fetch older posts** button on an X channel's page (or `pnpm ops fetch-posts --json '{"slug":"<channel>","older":true}'`) walks back from the oldest archived post through X search, three months at a time, and saves posts the same way a normal fetch does; posts already archived are skipped. It needs a login, as search does: without one it stops at once and the channel shows **Needs credentials**. A run saves its place as it goes and stops after three hours; the next run continues from there. The walk ends at the account's creation date, after a year of windows with no posts, or at a date you give as `"floor": "YYYY-MM-DD"`, and the page's **Older posts** line then says it is complete; running it again says so and fetches nothing. A normal **Fetch posts** is unaffected and still fetches new posts from the top. Bluesky channels have no such button: their fetch already reads the whole history. diff --git a/editor/app/api/ops/capture-posts/route.ts b/editor/app/api/ops/capture-posts/route.ts @@ -0,0 +1,41 @@ +import { capturePostsAction } from "../../../channels/[slug]/socialActions"; +import { + jobResponse, + ops, + optBool, + optString, + reqSlug, + reqStringArray, +} from "../_lib"; + +export const dynamic = "force-dynamic"; + +// POST { slug, ids, shots?, media?, force?, queueKey? } -> { ok: true, jobId } +// +// Capture specific archived posts of a social channel: a screenshot of each +// (`shots`, default true) and its attached media (`media`, default true), into +// the channel's posts-media/<id>/. Posts already captured are skipped unless +// `force`. The job runs on the platform's queue, as a post fetch does. +// +// Every refusal is the action's own sentence: both halves off, a channel that +// is not a social one, a fetcher that cannot capture, an id not in the +// channel's posts archive. +export async function POST(request: Request) { + return ops( + request, + ["slug", "ids", "shots", "media", "force", "queueKey"], + async (body) => { + const slug = reqSlug(body, "slug"); + return jobResponse( + await capturePostsAction( + slug, + reqStringArray(body, "ids"), + optString(body, "queueKey"), + optBool(body, "shots"), + optBool(body, "media"), + optBool(body, "force"), + ), + ); + }, + ); +} diff --git a/editor/app/channels/[slug]/socialActions.ts b/editor/app/channels/[slug]/socialActions.ts @@ -6,6 +6,7 @@ // which applies to a post fetch. A social channel has exactly two stages // (Fetch → Index), and this file owns the first. +import path from "node:path"; import { revalidatePath } from "next/cache"; import { safeRevalidate } from "../../lib/safeRevalidate"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; @@ -30,6 +31,14 @@ import { type CheckPostAvailabilityMode, } from "yt-dlp-transcript-common/controller/checkPostAvailability"; import { + capturePosts, + capturePostsProblem, + NOTHING_TO_CAPTURE, + strayCaptureIds, + strayIdsRefusal, +} from "yt-dlp-transcript-common/controller/capturePosts"; +import { readSeenPostIds } from "yt-dlp-transcript-common/lib/posts-server"; +import { getSocialFetcher, listSocialFetchers, resolveSocialFetcher, @@ -196,3 +205,67 @@ export async function fetchPostsAction( }, }); } + +// A screenshot and the attached media of specific archived posts, into the +// channel's `posts-media/<id>/`. On the platform queue, as a fetch is, so the +// two never run against the same source at once. Refused HERE, before a job +// exists: nothing asked for, no ids, a channel that is not social, a fetcher +// that cannot capture, or an id that is not in the channel's posts archive +// (named). Posts already captured are skipped unless `force`. +export async function capturePostsAction( + slug: string, + ids: string[], + queueKey?: string, + shots?: boolean, + media?: boolean, + force?: boolean, +): Promise<StreamActionResult> { + if (shots === false && media === false) return { ok: false, error: NOTHING_TO_CAPTURE }; + const wanted = [...new Set(ids)]; + if (wanted.length === 0) return { ok: false, error: "No post ids to capture." }; + const paths = getPaths(); + const config = await readChannelConfig(paths, slug); + if (!config) return { ok: false, error: `No such channel: ${slug}` }; + if (!isSocialChannel(config)) { + return { ok: false, error: `${slug} is not a social channel.` }; + } + await registerBuiltinSocialFetchers(); + const problem = capturePostsProblem( + resolveSocialFetcher(config.postFetcher, config.url), + ); + if (problem) return { ok: false, error: problem }; + const stray = strayCaptureIds( + wanted, + await readSeenPostIds(path.join(paths.channelsDir, slug)), + ); + if (stray) return { ok: false, error: strayIdsRefusal(slug, stray) }; + + const key = resolveQueueKey(downloadQueueKey(config), queueKey); + return runManagedFunction({ + kind: "capture-posts", + queueKey: key, + paths, + channelSlug: slug, + spec: { + kind: "capture-posts", + slug, + params: { queueKey, ids: wanted, shots, media, force }, + }, + fn: async (onLog, signal, _progress, ctx) => { + const result = await capturePosts({ + paths, + slug, + settings: getSettings(), + ids: wanted, + shots, + media, + force, + onLog, + signal, + drain: ctx.drainSignal, + }); + safeRevalidate([`/channels/${slug}`]); + if (!result.ok) throw new Error(result.error ?? "Post capture failed"); + }, + }); +} diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -40,6 +40,7 @@ import { import { backfillChannelAction } from "../channels/[slug]/backfillActions"; import { persistKeptAction } from "../channels/[slug]/persistActions"; import { + capturePostsAction, checkPostAvailabilityAction, fetchPostsAction, } from "../channels/[slug]/socialActions"; @@ -255,6 +256,19 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { str(p.floor), ); }, + // The ids are the spec's own (a capture is OF specific posts, unlike a + // bucket); the re-run skips whatever the first run already captured. + "capture-posts": (spec) => { + const { p, queueKey } = params(spec); + return capturePostsAction( + spec.slug, + strings(p.ids) ?? [], + queueKey, + bool(p.shots), + bool(p.media), + bool(p.force), + ); + }, "download-missing-subs": (spec) => { const { p, queueKey } = params(spec); return downloadMissingSubsAction(spec.slug, queueKey, bool(p.abortOnError)); diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts @@ -193,6 +193,7 @@ test("a traversing slug is refused at the door, on every route that takes one", ["relocate", { slugs: ["../../escape"], root: "/tmp/ops-api-never" }], ["relocate-back", { slugs: ["../../escape"] }], ["fetch-posts", { slug: "../../escape", older: true }], + ["capture-posts", { slug: "../../escape", ids: ["1"] }], ]; for (const [action, data] of cases) { const { status, body } = await ops(request, action, data); @@ -335,6 +336,68 @@ test("fetch-posts refuses a channel that is not social, full with older, and an expect(await listJobIds()).toEqual(before); }); +test("capture-posts refuses what it cannot capture, and an id not in the archive — before any job", async ({ + request, +}) => { + await resetData("title-filter-channel"); + await settings(); + // Nothing below reaches X: every case is refused before a job exists. + await writeChannelConfig("example-bsky", { + handling: "transcribe", + sourceKind: "social", + platform: "bluesky", + postFetcher: "bluesky-atproto", + socialHandle: "example.bsky.social", + name: "Example (Bluesky)", + url: "https://bsky.app/profile/example.bsky.social", + }); + await writeChannelConfig("example-x", { + handling: "transcribe", + sourceKind: "social", + platform: "twitter", + postFetcher: "x-gallery-dl", + socialHandle: "example_user", + name: "Example (X)", + url: "https://x.com/example_user", + }); + const before = await listJobIds(); + + const video = await ops(request, "capture-posts", { slug: "test-filter", ids: ["1"] }); + expect(video.status).toBe(400); + expect(video.body.error).toBe("test-filter is not a social channel."); + + const bsky = await ops(request, "capture-posts", { slug: "example-bsky", ids: ["1"] }); + expect(bsky.status).toBe(400); + expect(bsky.body.error).toMatch(/cannot capture posts/); + + const neither = await ops(request, "capture-posts", { + slug: "example-x", + ids: ["1"], + shots: false, + media: false, + }); + expect(neither.status).toBe(400); + expect(neither.body.error).toMatch(/both the screenshot and the media are turned off/); + + // The channel's posts archive is empty: every id is a stray, named. + const stray = await ops(request, "capture-posts", { slug: "example-x", ids: ["111", "222"] }); + expect(stray.status).toBe(400); + expect(stray.body.error).toBe("2 id(s) not in example-x's posts archive: 111, 222"); + + // The body's shape. + const noIds = await ops(request, "capture-posts", { slug: "example-x" }); + expect(noIds.status).toBe(400); + expect(noIds.body.error).toMatch(/"ids" is required/); + const badFlag = await ops(request, "capture-posts", { slug: "example-x", ids: ["1"], shots: "yes" }); + expect(badFlag.status).toBe(400); + expect(badFlag.body.error).toMatch(/"shots" must be a boolean/); + const unknown = await ops(request, "capture-posts", { slug: "example-x", ids: ["1"], limit: 5 }); + expect(unknown.status).toBe(400); + expect(unknown.body.error).toMatch(/unknown key\(s\): limit/); + + expect(await listJobIds()).toEqual(before); +}); + test("channel-config round-trips a download filter and refuses a bad regex", async ({ page, request, diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs @@ -103,6 +103,8 @@ const ACTIONS = [ // A social channel's post fetch: new posts, the full re-walk ("full"), or // the walk back below the oldest archived post ("older"). "fetch-posts", + // A screenshot and the attached media of specific archived posts. + "capture-posts", "build-index", "build-deploy", "build-site", @@ -326,6 +328,13 @@ export function usage() { ' the next run, down to "floor": "YYYY-MM-DD" when given. "limit": N caps', ' the posts one run reads. "full" and "older" together are refused.', "", + 'capture-posts captures archived posts of a social channel (X): a', + ' screenshot of each through the connected X profile, and its attached', + ' media through gallery-dl, into the channel\'s posts-media/<id>/:', + ' {"slug", "ids": [...]}. Every id must be in the channel\'s posts archive.', + ' "shots": false or "media": false skips that half; posts already captured', + ' are skipped unless "force": true. Paced like a post fetch, on its queue.', + "", '"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare', " Pages PREVIEW instead of production: the same bundle goes to a branch", " alias, https://<branch>.<project>.pages.dev, and the live site is left", diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs @@ -390,3 +390,18 @@ test("fetch-posts is a POST to its route, named in the usage", () => { assert.match(usage(), /Actions:.*transcribe-bucket, fetch-posts/); assert.match(usage(), /"older": true walks back from the oldest/); }); + +// A post capture: a POST to its route, the body passed through untouched — the +// route and the action judge it (ids in the archive, both halves off). +test("capture-posts is a POST to its route, named in the usage", () => { + const p = parseArgs([ + "capture-posts", + "--json", + '{"slug":"example-x","ids":["123"],"media":false}', + ]); + assert.equal(p.method, "POST"); + assert.equal(p.path, "/api/ops/capture-posts"); + assert.deepEqual(p.body, { slug: "example-x", ids: ["123"], media: false }); + assert.match(usage(), /Actions:.*fetch-posts, capture-posts/); + assert.match(usage(), /Every id must be in the channel's posts archive/); +});