commit 36d71ae9a55192a54a06d36e87ca2214cc47e65b
parent 66e5133c912d8dfbe11b5545092dec8ed0557358
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 2 Oct 2026 18:54:55 -0400
gerge branch 'ops-retry-bucket-ids'
Diffstat:
6 files changed, 117 insertions(+), 5 deletions(-)
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **`pnpm ops retry-bucket` can run part of a bucket.** Its body takes `"ids"`, a list of video ids, and runs only those, as ticking them on the bucket's card does. Every id must be in the named bucket: one that is not is refused with a 400 naming it, and nothing runs. A job started with `ids` is not replayable, like a checkbox selection in the UI.
- **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen.
- **An X post fetched by gallery-dl says whether it is a reply or a repost.** gallery-dl writes a tweet's references as `reply_id`, `reply_to` and `retweet_id` (0 when unset), and the normalizer read only X's own `in_reply_to_*` and `retweeted_status` fields, so every fetched post was stored as an original post. A reply now carries the replied-to post and handle; a repost carries the original's id and author and is credited to the account that reposted it. gallery-dl's `quote_id` names the tweet quoting this one, not the one quoted, and is no longer read as a quote. Posts already fetched keep their old flags until the account is fetched again.
- **umtool can keep each report's render folder on a media drive.** With `UMTOOL_MEDIA_DIR` set, in umtool's environment (restart umtool after setting it), to a directory inside that drive, a report project's `out/` (its fetched windows, segments and finished video) is a link to the same path under that directory: a project's first build makes it there, and `umtool storage move-out <project>` (or `--all`) moves an existing one, copying it, checking the copy and only then leaving the link; `--dry-run` says how much would move, and `umtool storage move-back` brings one home. The manifest, its revisions, notes and sources stay where they are, and nothing in umtool reads a project differently. When the drive is not mounted, a build or source check refuses and says so instead of starting a new folder on the main disk; umtool never creates the media directory itself. `umtool storage` lists where each project's `out/` is. With `UMTOOL_MEDIA_DIR` unset nothing changes.
diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts
@@ -0,0 +1,38 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import { OpsInputError, optSubset } from "./_lib";
+
+// Run with:
+// pnpm -C editor exec tsx --test "app/**/*.test.ts"
+
+const BUCKET = ["a1", "b2", "c3"];
+
+test("optSubset: absent is undefined, so the caller falls back to the whole list", () => {
+ assert.equal(optSubset({}, "ids", BUCKET, "the bucket"), undefined);
+});
+
+test("optSubset: a subset comes back in the caller's order, duplicates collapsed", () => {
+ assert.deepEqual(
+ optSubset({ ids: ["c3", "a1", "c3"] }, "ids", BUCKET, "the bucket"),
+ ["c3", "a1"],
+ );
+});
+
+test("optSubset: a stray id is refused and every stray is named", () => {
+ assert.throws(
+ () => optSubset({ ids: ["a1", "zz", "yy"] }, "ids", BUCKET, "the bucket"),
+ (e: unknown) =>
+ e instanceof OpsInputError &&
+ /2 of "ids" not in the bucket: zz, yy/.test(e.message),
+ );
+});
+
+test("optSubset: an empty or non-string list is refused, never read as 'all'", () => {
+ for (const ids of [[], "a1", [1], [""]]) {
+ assert.throws(
+ () => optSubset({ ids }, "ids", BUCKET, "the bucket"),
+ OpsInputError,
+ JSON.stringify(ids),
+ );
+ }
+});
diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts
@@ -160,6 +160,30 @@ export function reqStringArray(body: OpsBody, key: string): string[] {
return (v as string[]).map((s) => s.trim());
}
+// A SUBSET OF A LIST THE CALLER CAN ALREADY ACT ON, or undefined when absent.
+// Every id must be in `of`: a route taking it narrows a gesture the UI offers
+// (a checkbox selection inside a bucket), it never widens it. A stray id is a
+// 400 naming every stray, not a silent drop — a caller that pasted a list from
+// an older report would otherwise run a smaller job than it asked for and be
+// told nothing. Duplicates collapse; order is the caller's.
+export function optSubset(
+ body: OpsBody,
+ key: string,
+ of: readonly string[],
+ ofName: string,
+): string[] | undefined {
+ if (body[key] === undefined) return undefined;
+ const wanted = [...new Set(reqStringArray(body, key))];
+ const known = new Set(of);
+ const stray = wanted.filter((v) => !known.has(v));
+ if (stray.length) {
+ throw new OpsInputError(
+ `${stray.length} of "${key}" not in ${ofName}: ${stray.join(", ")}`,
+ );
+ }
+ return wanted;
+}
+
// ONE SITE OR SEVERAL, SPELLED EITHER WAY. The two build routes disagreed —
// build-site took `siteIds` (a list), build-deploy took `siteId` (one) — so the
// same body worked on one and 400'd on the other, and the fix people reached
diff --git a/editor/app/api/ops/retry-bucket/route.ts b/editor/app/api/ops/retry-bucket/route.ts
@@ -8,13 +8,14 @@ import {
ops,
optBool,
optString,
+ optSubset,
reqSlug,
reqString,
} from "../_lib";
export const dynamic = "force-dynamic";
-// POST { slug, bucket, queueKey?, abortOnError?, handlingOverride?,
+// POST { slug, bucket, ids?, queueKey?, abortOnError?, handlingOverride?,
// forceCookies?, replaceAutoSubs? } -> { ok: true, jobId }
//
// THE BUCKET IS RESOLVED FROM THE SNAPSHOT HERE, and that is not new logic: the
@@ -23,12 +24,18 @@ export const dynamic = "force-dynamic";
// rather than pasting ids is the same gesture — and `bucketKey` travelling with
// it is what makes the job replayable against the CURRENT bucket, exactly as a
// clicked one is.
+//
+// `ids` NARROWS THE BUCKET, it never widens it: every id must be in the named
+// bucket, as a checkbox selection on the bucket's card is. With `ids` the job
+// carries no `bucketKey` — like the UI's ad-hoc selections it is not
+// replayable, because a replay re-reads the WHOLE current bucket.
export async function POST(request: Request) {
return ops(
request,
[
"slug",
"bucket",
+ "ids",
"queueKey",
"abortOnError",
"handlingOverride",
@@ -44,25 +51,31 @@ export async function POST(request: Request) {
`Channel "${slug}" has no report yet — run /api/ops/refresh-report first.`,
);
}
- const ids = (snapshot.buckets as Record<string, unknown>)[bucket];
- if (!Array.isArray(ids)) {
+ const bucketIds = (snapshot.buckets as Record<string, unknown>)[bucket];
+ if (!Array.isArray(bucketIds)) {
throw new OpsInputError(
`"${bucket}" is not a bucket on this channel's report — known buckets: ${Object.keys(
snapshot.buckets,
).join(", ")}`,
);
}
+ const subset = optSubset(
+ body,
+ "ids",
+ bucketIds as string[],
+ `the "${bucket}" bucket of ${slug}`,
+ );
return jobResponse(
await retryBucketAction(
slug,
- ids as string[],
+ subset ?? (bucketIds as string[]),
optString(body, "queueKey"),
optBool(body, "abortOnError"),
optString(body, "handlingOverride"),
// Replayable only for the buckets a replay can re-derive; an
// ad-hoc bucket name still runs, it just carries no spec — the
// same distinction the UI's named vs checkbox controls make.
- isReplayBucket(bucket) ? bucket : undefined,
+ !subset && isReplayBucket(bucket) ? bucket : undefined,
optBool(body, "forceCookies"),
optBool(body, "replaceAutoSubs"),
),
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -229,6 +229,37 @@ test("a traversing slug is refused at the door, on every route that takes one",
});
});
+test("retry-bucket ids must be in the bucket: a stray is refused, named, and nothing runs", async ({
+ request,
+}) => {
+ await resetData("title-filter-channel");
+ await settings();
+ const SLUG = "test-filter";
+ // A report by hand, so the bucket is known without a regen. The stray ids are
+ // in NO bucket of ANY report, so a debounced regen landing between this write
+ // and the call cannot change the answer.
+ await writeFile(
+ resolvePath(`test-transcripts/channels/${SLUG}/snapshot.json`),
+ JSON.stringify({ buckets: { downloadedNoTranscript: ["in-bucket-1"] } }),
+ );
+ const before = await listJobIds();
+ const stray = await ops(request, "retry-bucket", {
+ slug: SLUG,
+ bucket: "downloadedNoTranscript",
+ ids: ["stray-aaa", "stray-bbb"],
+ });
+ expect(stray.status).toBe(400);
+ expect(stray.body.error).toMatch(/2 of "ids" not in .*: stray-aaa, stray-bbb/);
+ // An empty list is not "the whole bucket".
+ const empty = await ops(request, "retry-bucket", {
+ slug: SLUG,
+ bucket: "downloadedNoTranscript",
+ ids: [],
+ });
+ expect(empty.status).toBe(400);
+ expect(await listJobIds()).toEqual(before);
+});
+
test("channel-config round-trips a download filter and refuses a bad regex", async ({
page,
request,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -305,6 +305,11 @@ export function usage() {
'build-site, build-deploy and deploy-site all take "siteId" (one) or',
' "siteIds" (a list).',
"",
+ 'retry-bucket runs one bucket of a channel\'s report as one job, past any',
+ ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and',
+ " every one must be in the bucket (a stray id is refused, named); a job run",
+ " with ids is not replayable, as a checkbox selection in the UI is not.",
+ "",
'"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare',
" Pages PREVIEW instead of production: the same bundle goes to a branch",
" alias, https://<branch>.<project>.pages.dev, and the live site is left",