commit fa2f521343c6c9a2b4a94e7b9871d66926ea8c4b
parent 36d71ae9a55192a54a06d36e87ca2214cc47e65b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 2 Oct 2026 19:19:27 -0400
Merge ops-transcribe-bucket (pnpm ops transcribe-bucket: the channel page's Transcribe-downloaded button over HTTP, with an optional ids subset of the bucket; retry-bucket skips audio already on disk)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 79 insertions(+), 1 deletion(-)
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **`pnpm ops transcribe-bucket` transcribes a channel's downloaded-but-untranscribed videos**, as the channel page's **Transcribe N downloaded** button does, on the transcription queue. `"ids"` runs only some of them; each must be in the bucket, and a stray id is refused by name. `retry-bucket` is not the way to do this: it retries downloads, and counts a video whose audio is on disk as complete.
- **`pnpm ops retry-bucket` can run part of a bucket.** Its body takes `"ids"`, a list of video ids, and runs only those, as ticking them on the bucket's card does. Every id must be in the named bucket: one that is not is refused with a 400 naming it, and nothing runs. A job started with `ids` is not replayable, like a checkbox selection in the UI.
- **An X post fetch keeps what it has read when it is cancelled, times out or fails part-way, and the next fetch picks up where it stopped.** The gallery-dl fetcher used to receive an account's posts all at once when gallery-dl finished, so a long fetch that X's rate limit held past the 30-minute limit — or one you cancelled — ended with nothing saved. gallery-dl now hands over each post as it reads it: the fetch saves posts every 200 posts or every minute along with gallery-dl's own resume point, and the next fetch of the channel continues from that point instead of starting again. A fetch of new posts stops once it reaches 100 already-archived posts in a row instead of reading the whole timeline, and a fetch of an account's history may run for up to 3 hours (new-post fetches keep the 30-minute limit). gallery-dl's rate-limit waits now appear in the job's log as they happen.
- **An X post fetched by gallery-dl says whether it is a reply or a repost.** gallery-dl writes a tweet's references as `reply_id`, `reply_to` and `retweet_id` (0 when unset), and the normalizer read only X's own `in_reply_to_*` and `retweeted_status` fields, so every fetched post was stored as an original post. A reply now carries the replied-to post and handle; a repost carries the original's id and author and is credited to the account that reposted it. gallery-dl's `quote_id` names the tweet quoting this one, not the one quoted, and is no longer read as a quote. Posts already fetched keep their old flags until the account is fetched again.
diff --git a/editor/app/api/ops/transcribe-bucket/route.ts b/editor/app/api/ops/transcribe-bucket/route.ts
@@ -0,0 +1,63 @@
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channels";
+import { AUDIO_FORMAT_VALUES } from "yt-dlp-transcript-common/lib/channelConfig";
+import { transcribeBucketAction } from "../../../channels/[slug]/whisperActions";
+import {
+ jobResponse,
+ OpsInputError,
+ oneOf,
+ ops,
+ optBool,
+ optString,
+ optSubset,
+ reqSlug,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { slug, ids?, queueKey?, audioFormat?, strictAudioFormat? } -> { ok: true, jobId }
+//
+// The channel page's "Transcribe N downloaded" button, over HTTP: the
+// `downloadedNoTranscript` bucket from the snapshot, handed to the same action
+// on the transcription queue. retry-bucket is not this — it is the DOWNLOAD
+// retry, and its prefilter counts a video with audio on disk as complete, so
+// it would skip every video in this bucket.
+//
+// `ids` NARROWS THE BUCKET, it never widens it (optSubset). With `ids` the job
+// carries no bucket key and is not replayable, as an ad-hoc selection is not.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["slug", "ids", "queueKey", "audioFormat", "strictAudioFormat"],
+ async (body) => {
+ const slug = reqSlug(body, "slug");
+ const snapshot = await readChannelSnapshot(getPaths(), slug);
+ if (!snapshot) {
+ throw new OpsInputError(
+ `Channel "${slug}" has no report yet — run /api/ops/refresh-report first.`,
+ );
+ }
+ const bucketIds = snapshot.buckets.downloadedNoTranscript ?? [];
+ const subset = optSubset(
+ body,
+ "ids",
+ bucketIds,
+ `the "downloadedNoTranscript" bucket of ${slug}`,
+ );
+ const audioFormat =
+ body.audioFormat === undefined
+ ? undefined
+ : oneOf(body, "audioFormat", AUDIO_FORMAT_VALUES);
+ return jobResponse(
+ await transcribeBucketAction(
+ slug,
+ subset ?? bucketIds,
+ optString(body, "queueKey"),
+ audioFormat,
+ optBool(body, "strictAudioFormat"),
+ subset ? undefined : "downloadedNoTranscript",
+ ),
+ );
+ },
+ );
+}
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -229,7 +229,7 @@ test("a traversing slug is refused at the door, on every route that takes one",
});
});
-test("retry-bucket ids must be in the bucket: a stray is refused, named, and nothing runs", async ({
+test("retry-bucket and transcribe-bucket ids must be in the bucket: a stray is refused, named, and nothing runs", async ({
request,
}) => {
await resetData("title-filter-channel");
@@ -250,6 +250,13 @@ test("retry-bucket ids must be in the bucket: a stray is refused, named, and not
});
expect(stray.status).toBe(400);
expect(stray.body.error).toMatch(/2 of "ids" not in .*: stray-aaa, stray-bbb/);
+ // transcribe-bucket narrows the same bucket by the same rule.
+ const strayT = await ops(request, "transcribe-bucket", {
+ slug: SLUG,
+ ids: ["stray-ccc"],
+ });
+ expect(strayT.status).toBe(400);
+ expect(strayT.body.error).toMatch(/1 of "ids" not in .*: stray-ccc/);
// An empty list is not "the whole bucket".
const empty = await ops(request, "retry-bucket", {
slug: SLUG,
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -97,6 +97,9 @@ const ACTIONS = [
"sync",
"download-missing",
"retry-bucket",
+ // The TRANSCRIBE half of a channel's "downloaded, not transcribed" bucket
+ // (retry-bucket is the download retry and skips audio already on disk).
+ "transcribe-bucket",
"build-index",
"build-deploy",
"build-site",
@@ -310,6 +313,10 @@ export function usage() {
" every one must be in the bucket (a stray id is refused, named); a job run",
" with ids is not replayable, as a checkbox selection in the UI is not.",
"",
+ 'transcribe-bucket transcribes a channel\'s "downloaded, not transcribed"',
+ ' bucket on the transcription queue, as the channel page\'s Transcribe button',
+ ' does: {"slug"}. "ids": [...] narrows it the same way as on retry-bucket.',
+ "",
'"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare',
" Pages PREVIEW instead of production: the same bundle goes to a branch",
" alias, https://<branch>.<project>.pages.dev, and the live site is left",