Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 25f7d342f297ee59448a5e55520f282258677e82
parent c37177bed9fe09b13923b3194f3f4fc1df0f309e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun,  4 Oct 2026 16:11:49 -0400

common: capturePosts controller and the capture-posts job kind

capturePosts refuses, before any capture, a request for neither half, no
ids, a channel that is not social, a fetcher that cannot capture, and ids
not in the channel's posts archive (named). What the pages said about each
post — deleted, or behind its account's wall — is merged into
posts-availability.json through the deleted-post sweep's merge; a login
wall records nothing. capture-posts runs on the platform queue, like
fetch-posts, and is drainable and replayable.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Acommon/controller/capturePosts.test.ts | 145+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/capturePosts.ts | 206+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.ts | 13+++++++++++++
3 files changed, 364 insertions(+), 0 deletions(-)

diff --git a/common/controller/capturePosts.test.ts b/common/controller/capturePosts.test.ts @@ -0,0 +1,145 @@ +// capturePosts over a registered fake fetcher: what is refused before any +// capture, and how the pages' verdicts land in the availability sidecar. +// +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test controller/capturePosts.test.ts + +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; + +const ROOT = await mkdtemp(path.join(os.tmpdir(), "captureposts-")); +process.env.TRANSCRIPTS_DIR = path.join(ROOT, "transcripts"); + +const { getPaths } = await import("../lib/paths"); +const { registerSocialFetcher } = await import("../social/fetchers"); +const { readPostAvailability, writePostAvailability } = await import("../lib/posts-server"); +const { capturePosts, NOTHING_TO_CAPTURE } = await import("./capturePosts"); +type Input = import("../social/fetchers").PostCaptureInput; +type Result = import("../social/fetchers").PostCaptureResult; + +// A capturing fetcher whose answer each test sets. +let calls: Input[] = []; +let answer: Result = { outcomes: [] }; +registerSocialFetcher({ + id: "test-capture", + label: "Test capture", + platform: "bluesky", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), + captureByIds: async (input) => { + calls.push(input); + return answer; + }, +}); +registerSocialFetcher({ + id: "test-no-capture", + label: "No-capture fetcher", + platform: "bluesky", + fields: {}, + detect: () => false, + probe: async () => ({ ok: true }), + fetch: async () => ({ posts: [], complete: true }), +}); + +const channelsDir = path.join(ROOT, "transcripts", "channels"); + +async function makeChannel(slug: string, over: Record<string, unknown> = {}): Promise<string> { + const root = path.join(channelsDir, slug); + await mkdir(root, { recursive: true }); + await writeFile( + path.join(root, "config.json"), + JSON.stringify({ + handling: "transcribe", + sourceKind: "social", + platform: "bluesky", + postFetcher: "test-capture", + socialHandle: "example.bsky.social", + name: "Example", + url: "https://bsky.app/profile/example.bsky.social", + ...over, + }), + ); + await writeFile(path.join(root, "posts-archive"), "bluesky aaa\nbluesky bbb\nbluesky ccc\n"); + return root; +} + +const run = (slug: string, ids: string[], over: Record<string, unknown> = {}) => + capturePosts({ paths: getPaths(), slug, settings: {}, ids, ...over }); + +test("refused before any capture: nothing asked for, no ids, not social, no capture, ids not archived", async () => { + await makeChannel("demo-social"); + await makeChannel("demo-nocapture", { postFetcher: "test-no-capture" }); + await mkdir(path.join(channelsDir, "demo-video"), { recursive: true }); + await writeFile( + path.join(channelsDir, "demo-video", "config.json"), + JSON.stringify({ handling: "transcribe", name: "Video", url: "https://example.com/c" }), + ); + calls = []; + assert.equal((await run("demo-social", ["aaa"], { shots: false, media: false })).error, NOTHING_TO_CAPTURE); + assert.match((await run("demo-social", [])).error ?? "", /No post ids/); + assert.equal((await run("demo-video", ["aaa"])).error, "demo-video is not a social channel."); + assert.equal((await run("demo-nocapture", ["aaa"])).error, "No-capture fetcher cannot capture posts."); + assert.equal( + (await run("demo-social", ["aaa", "zzz", "yyy"])).error, + "2 id(s) not in demo-social's posts archive: zzz, yyy", + ); + assert.equal(calls.length, 0); +}); + +test("the fetcher gets the ids (deduped), the channel's posts-media dir and the halves asked for", async () => { + const root = await makeChannel("demo-args"); + calls = []; + answer = { outcomes: [] }; + const res = await run("demo-args", ["aaa", "bbb", "aaa"], { media: false, force: true }); + assert.equal(res.ok, true); + assert.equal(calls.length, 1); + assert.deepEqual(calls[0].ids, ["aaa", "bbb"]); + assert.equal(calls[0].outDir, path.join(root, "posts-media")); + assert.equal(calls[0].handle, "example.bsky.social"); + assert.equal(calls[0].media, false); + assert.equal(calls[0].force, true); +}); + +test("deleted and walled posts are recorded in posts-availability.json; a login wall is not", async () => { + const root = await makeChannel("demo-avail"); + await writePostAvailability(root, { + aaa: { availability: "available", checkedAt: "2026-01-01T00:00:00.000Z" }, + }); + answer = { + outcomes: [ + { id: "aaa", state: "deleted", availability: "deleted", files: 0 }, + { id: "bbb", state: "unavailable", availability: "account_unavailable", files: 0 }, + { id: "ccc", state: "login-wall", files: 0 }, + ], + needsCookies: true, + stoppedEarly: "X asked to log in. Stopped at ccc.", + }; + const res = await run("demo-avail", ["aaa", "bbb", "ccc"]); + assert.equal(res.ok, false); + assert.equal(res.needsCookies, true); + assert.match(res.error ?? "", /Stopped at ccc/); + const map = await readPostAvailability(root); + assert.equal(map.aaa.availability, "deleted"); + // The change keeps the moment it was last seen up. + assert.deepEqual(map.aaa.history, [{ availability: "available", at: "2026-01-01T00:00:00.000Z" }]); + assert.equal(map.bbb.availability, "account_unavailable"); + assert.equal(map.ccc, undefined); +}); + +test("a run stopped by the source fails the job; one cancelled by the operator does not", async () => { + await makeChannel("demo-stop"); + answer = { outcomes: [], stoppedEarly: "X is refusing pages right now." }; + const stopped = await run("demo-stop", ["aaa"]); + assert.equal(stopped.ok, false); + assert.match(stopped.error ?? "", /refusing/); + + const ac = new AbortController(); + ac.abort(); + answer = { outcomes: [], stoppedEarly: "Cancelled; the rest are left for a later run." }; + const cancelled = await run("demo-stop", ["aaa"], { signal: ac.signal }); + assert.equal(cancelled.ok, true); +}); diff --git a/common/controller/capturePosts.ts b/common/controller/capturePosts.ts @@ -0,0 +1,206 @@ +// Capture specific archived posts of a social channel: a screenshot of each +// post as its platform renders it, and its attached media, into +// `channels/<slug>/posts-media/<id>/` (social/postCapture.ts owns the layout). +// +// The fetcher does the capturing (`SocialFetcher.captureByIds`); this lands +// what it learned about each post's liveness in the channel's availability +// sidecar, `posts-availability.json`, through the same merge the deleted-post +// sweep uses — a deleted or walled post is recorded there, history appended +// only on a change, and nothing is recorded from ignorance (a login wall says +// nothing about the post). +// +// Only ids already in the channel's posts archive are captured: a capture is +// of the archive, and an id from somewhere else is refused by name. + +import path from "node:path"; +import { readChannelConfig } from "./channels"; +import type { Paths } from "../lib/paths"; +import type { PostAvailability } from "../lib/posts"; +import { isSocialChannel } from "../lib/channelConfig"; +import { + alwaysCookies, + resolveCookiePolicy, + type CookiePolicyInputs, +} from "../lib/cookiePolicy"; +import { + mergePostAvailability, + readPostAvailability, + readSeenPostIds, + writePostAvailability, +} from "../lib/posts-server"; +import { + handleFromAccountUrl, + resolveSocialFetcher, + type PostCaptureOutcome, + type SocialFetcher, +} from "../social/fetchers"; +import { postsMediaDir } from "../social/postCapture"; +import "../social/blueskyFetcher"; +import "../social/xGalleryDlFetcher"; +import "../social/xPlaywrightFetcher"; +import "../social/xNitterFetcher"; +import { + resolveXCookieSourceFor, + type XLoginSettings, +} from "../social/xBrowserLogin"; + +export type CapturePostsOptions = { + paths: Paths; + slug: string; + settings: CookiePolicyInputs & XLoginSettings; + ids: ReadonlyArray<string>; + // Take the screenshot / download the media. Both default to true. + shots?: boolean; + media?: boolean; + // Capture again what is already captured. + force?: boolean; + onLog?: (line: string) => void; + signal?: AbortSignal; + // The job's drain: stop between posts. + drain?: AbortSignal; +}; + +export type CapturePostsResult = { + ok: boolean; + outcomes: PostCaptureOutcome[]; + needsCookies?: boolean; + error?: string; +}; + +// Why this fetcher cannot capture posts, or null when it can. Shared with the +// server action, so the refusal is one sentence everywhere. +export function capturePostsProblem( + fetcher: Pick<SocialFetcher, "label" | "captureByIds"> | undefined, +): string | null { + if (!fetcher) return "This channel has no post fetcher."; + return typeof fetcher.captureByIds === "function" + ? null + : `${fetcher.label} cannot capture posts.`; +} + +export const NOTHING_TO_CAPTURE = + "Nothing to capture: both the screenshot and the media are turned off."; + +// The ids not in the channel's archive, or null when every one is. +export function strayCaptureIds( + ids: ReadonlyArray<string>, + archived: ReadonlySet<string>, +): string[] | null { + const stray = ids.filter((id) => !archived.has(id)); + return stray.length ? stray : null; +} + +export function strayIdsRefusal(slug: string, stray: string[]): string { + return `${stray.length} id(s) not in ${slug}'s posts archive: ${stray.join(", ")}`; +} + +export async function capturePosts( + opts: CapturePostsOptions, +): Promise<CapturePostsResult> { + const { paths, slug, settings, onLog } = opts; + const log = (line: string) => onLog?.(line); + const fail = (error: string): CapturePostsResult => ({ ok: false, outcomes: [], error }); + const channelRoot = path.join(paths.channelsDir, slug); + const ids = [...new Set(opts.ids)]; + + if (opts.shots === false && opts.media === false) return fail(NOTHING_TO_CAPTURE); + if (ids.length === 0) return fail("No post ids to capture."); + + const config = await readChannelConfig(paths, slug); + if (!config) return fail(`No such channel: ${slug}`); + if (!isSocialChannel(config)) return fail(`${slug} is not a social channel.`); + + const accountUrl = config.url ?? ""; + const fetcher = resolveSocialFetcher(config.postFetcher, accountUrl); + const problem = capturePostsProblem(fetcher); + if (problem || !fetcher?.captureByIds) return fail(problem ?? "This channel has no post fetcher."); + + const handle = config.socialHandle ?? handleFromAccountUrl(accountUrl) ?? ""; + if (!handle) return fail(`Could not determine an account handle for ${slug}`); + + const stray = strayCaptureIds(ids, await readSeenPostIds(channelRoot)); + if (stray) return fail(strayIdsRefusal(slug, stray)); + + const policy = resolveCookiePolicy(settings, config); + const xLogin = + fetcher.platform === "twitter" + ? await resolveXCookieSourceFor(paths, settings, policy.cookies) + : undefined; + + log( + `Capturing ${ids.length} post(s) of ${slug} via ${fetcher.label} (@${handle}):` + + [ + opts.shots === false ? "" : " screenshot", + opts.media === false ? "" : " media", + ].join("") + + (opts.force ? ", again where already captured" : "") + + ".", + ); + + const controller = new AbortController(); + let result; + try { + result = await fetcher.captureByIds({ + ids, + handle, + outDir: postsMediaDir(channelRoot), + shots: opts.shots, + media: opts.media, + force: opts.force, + cookies: alwaysCookies(policy), + cookieSource: xLogin?.source, + browserCookies: xLogin?.browserSpec, + signal: opts.signal ?? controller.signal, + drain: opts.drain, + onLog: log, + }); + } catch (err) { + const message = (err as Error).message; + log(`[error] ${message}`); + return fail(message); + } + + // What the pages said about each post, folded into the sidecar. + const observations = new Map<string, PostAvailability>(); + for (const o of result.outcomes) { + if (o.availability) observations.set(o.id, o.availability); + } + if (observations.size > 0) { + const { map, newlyDeleted, changed } = mergePostAvailability( + await readPostAvailability(channelRoot), + observations, + new Date().toISOString(), + ); + await writePostAvailability(channelRoot, map); + log( + `Availability: ${observations.size} recorded, ${changed} changed` + + (newlyDeleted.length ? `, newly deleted: ${newlyDeleted.join(", ")}` : "") + + ".", + ); + } + + const count = (state: string) => result.outcomes.filter((o) => o.state === state).length; + log( + `Done: ${count("captured")} captured, ${count("deleted")} deleted, ` + + `${count("unavailable")} unavailable, ${count("error")} failed` + + (count("login-wall") ? `, ${count("login-wall")} met a login wall` : "") + + `; ${ids.length - result.outcomes.length} not attempted or already captured.`, + ); + + // A run that stopped at a login is a failed run (the job says so); one that + // was cancelled or met deleted posts is not. + if (result.needsCookies) { + return { + ok: false, + outcomes: result.outcomes, + needsCookies: true, + error: result.stoppedEarly ?? "The capture needs an X login.", + }; + } + const stoppedByOperator = + (opts.signal ?? controller.signal).aborted || Boolean(opts.drain?.aborted); + if (result.stoppedEarly && !stoppedByOperator) { + return { ok: false, outcomes: result.outcomes, error: result.stoppedEarly }; + } + return { ok: true, outcomes: result.outcomes }; +} diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -512,6 +512,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = { replayable: true, queueKeyStrategy: "platform", }, + // A screenshot and the attached media of specific archived posts, into the + // channel's `posts-media/` (controller/capturePosts.ts). On the PLATFORM + // queue, like fetch-posts: each post is a page load and a download against + // the same source, so it serialises with the fetch and shares its backoff. + // Drainable (it stops between posts) and replayable (the ids are in the + // spec; posts already captured are skipped on a re-run). + "capture-posts": { + kind: "capture-posts", + label: "Capture posts", + drainable: true, + replayable: true, + queueKeyStrategy: "platform", + }, // The posts analogue of the video availability check: which archived posts // have since been deleted at the source. "check-post-availability": {