commit 119d590e9f0da5bed0ea652be72f695deecaa358
parent 7f6aaf786c96f9b0b8b01e021106824faf554644
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 24 Sep 2026 19:48:48 -0400
channels: the transcribe station queues by id
Review F1: the captionless half was still a whisper-all SCAN, which
takes every downloaded-and-untranscribed dir — including a video
downloaded before it went private, members-only or deleted.
downloadedNoTranscript does not filter those ids (downloadedAutoSubsOnly
does), so the figure left y out and the button transcribed it: 2 on the
label, 3 transcribed.
With a snapshot the no-transcript half is now transcribeBucketAction over
exactly the ids the figure counts (replayable as downloadedNoTranscript,
the channel page's own bucket job). The scan stays only for a channel
with no report, where there are no ids to name. KIND_FOR.transcribe
gains whisper-bucket-downloaded-no-transcript, which also closes an older
gap: the station could queue a batch beside the channel page's own bucket
job. The reviewer's walk is a unit test ([x,y] + [a], y private → [x],
[a], figure 2); the e2e checks no no-transcript batch rode along.
Review F2: a half that is refused while the other queues is logged
(console.warn). StreamActionResult's ok arm has no message to carry it.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
3 files changed, 86 insertions(+), 29 deletions(-)
diff --git a/common/views/channelGroupSections.test.ts b/common/views/channelGroupSections.test.ts
@@ -296,6 +296,28 @@ test("transcribe drops excluded ids from BOTH buckets", () => {
});
});
+test("an excluded download is neither counted nor named for the batch (the review's walk)", () => {
+ // x, y downloaded with no transcript; a downloaded with auto-captions only;
+ // y went private after it was downloaded. The figure is 2 and the two id
+ // lists the station queues BY ID are exactly [x] and [a] — never y.
+ const snapshot = snapshotOf({
+ buckets: normalizeBuckets({
+ downloadedNoTranscript: ["x", "y"],
+ downloadedAutoSubsOnly: ["a"],
+ }),
+ excludedFromDownload: { membersOnly: [], deleted: [], private: ["y"] },
+ });
+ const sections = build(
+ siteOf({ channels: [{ slug: "yt" }] }),
+ [channel("yt", { handling: "youtube" }, snapshot)],
+ );
+ assert.equal(sections[0].transcribe.total, 2);
+ assert.deepEqual(transcribeStationIds(snapshot), {
+ missing: ["x"],
+ autoSubs: ["a"],
+ });
+});
+
test("a social channel is not transcribe-eligible, and says why", () => {
const { brief } = channel("posts", { sourceKind: "social", handling: "transcribe" });
assert.deepEqual(stationWorkFor("transcribe", brief, LANE_ON), {
diff --git a/editor/app/channels/groupActions.ts b/editor/app/channels/groupActions.ts
@@ -23,6 +23,7 @@ import { queueForSlugs } from "./lib/queueForSlugs";
import { downloadMissingAction, syncAction } from "./[slug]/pipelineActions";
import {
transcribeAutoSubsBucketAction,
+ transcribeBucketAction,
transcribeMissingAction,
} from "./[slug]/whisperActions";
import { backfillChannelAction } from "./[slug]/backfillActions";
@@ -50,49 +51,79 @@ export type GroupOpResult = {
};
// The job kinds each station enqueues, for the "already running" dedupe. The job
-// registry has no dedupe of its own. Transcribe names TWO: its button can queue
-// either or both, and the dedupe spans them — pressing it again while only the
-// auto-captions job runs reads "already running", not a second whisper-all.
+// registry has no dedupe of its own. Transcribe names THREE: its button queues
+// one or both bucket jobs (or the scan, for a channel with no report), and the
+// dedupe spans them all — pressing it again while any one runs, including the
+// channel page's own bucket job, reads "already running", not a second batch.
const KIND_FOR: Record<StationId, readonly string[]> = {
sync: ["sync"],
download: ["download-missing"],
- transcribe: ["whisper-all", "whisper-bucket-auto-subs"],
+ transcribe: [
+ "whisper-all",
+ "whisper-bucket-downloaded-no-transcript",
+ "whisper-bucket-auto-subs",
+ ],
digest: ["digest-channel-local"],
speakers: ["backfill-channel"],
};
// THE TRANSCRIBE STATION QUEUES WHAT ITS FIGURE COUNTS — `transcribeStationIds`,
-// the fold `stationWorkFor` sums. Two batches, because one cannot cover both:
-// `runWhisperBatch`'s default scan drains `downloadedNoTranscript`, and its
-// replace-auto-captions mode needs an ASR track per id. The channel page runs
-// the same two jobs (TranscribeStage's two sections); a combined job kind would
-// need its own replay spec and /jobs label and mirror nothing. Handling is not
-// consulted — a youtube channel's captionless download is whisper work too.
+// the fold `stationWorkFor` sums — BY ID. Two batches, because one cannot cover
+// both: the no-transcript half is `transcribeBucketAction` over its ids (the
+// channel page's own bucket control, replayable as `downloadedNoTranscript`),
+// and the auto-captions half is `transcribeAutoSubsBucketAction`, whose
+// replace mode needs an ASR track per id. By id matters: a video downloaded
+// before it went private, members-only or deleted is still on disk and still
+// in `downloadedNoTranscript`, and a directory SCAN would transcribe it — one
+// more than the figure said. The channel page runs the same jobs; a combined
+// job kind would need its own replay spec and /jobs label and mirror nothing.
+// Handling is not consulted — a youtube channel's captionless download is
+// whisper work too.
//
-// Both kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a time"
-// note above stays true. `queued` counts CHANNELS; `jobIds` carries one id per
-// channel — the whisper-all job's when both were queued, and the auto-captions
-// job's stream is cancelled here exactly as queueForSlugs cancels the other.
+// The SCAN (`transcribeMissingAction`, a whisper-all with no ids) is used ONLY
+// when the channel has never reported: with no snapshot there are no ids to
+// name, and the scan finds what there is, as the station always did.
+//
+// All three kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a
+// time" note above stays true. `queued` counts CHANNELS; `jobIds` carries one
+// id per channel — the no-transcript job's when both halves were queued, and
+// the other stream is cancelled here exactly as queueForSlugs cancels the one
+// returned. A half that fails while the other queues cannot ride on an ok
+// result (StreamActionResult's ok arm has no message), so it is logged.
async function transcribeChannel(
slug: string,
brief: ChannelBrief | undefined,
): Promise<StreamActionResult> {
- const ids = brief?.snapshot ? transcribeStationIds(brief.snapshot) : null;
- const autoSubsResult =
- ids && ids.autoSubs.length > 0
+ if (!brief?.snapshot) return transcribeMissingAction(slug);
+ const ids = transcribeStationIds(brief.snapshot);
+ const halves = [
+ ids.missing.length > 0
+ ? await transcribeBucketAction(
+ slug,
+ ids.missing,
+ undefined,
+ undefined,
+ undefined,
+ "downloadedNoTranscript",
+ )
+ : null,
+ ids.autoSubs.length > 0
? await transcribeAutoSubsBucketAction(slug, ids.autoSubs)
- : null;
- // No snapshot: the default scan finds what there is, as it always did.
- const missingResult =
- !ids || ids.missing.length > 0 || !autoSubsResult
- ? await transcribeMissingAction(slug)
- : null;
- if (missingResult?.ok) {
- if (autoSubsResult?.ok) void autoSubsResult.stream.cancel();
- return missingResult;
+ : null,
+ ].filter((r): r is StreamActionResult => r !== null);
+ // Both empty is skipped upstream as "nothing to do"; say so if reached.
+ if (halves.length === 0) return { ok: false, error: "nothing to do", info: true };
+ const ok = halves.find((r) => r.ok);
+ if (!ok) return halves[0];
+ for (const r of halves) {
+ if (r === ok) continue;
+ if (r.ok) void r.stream.cancel();
+ else
+ console.warn(
+ `[group transcribe] ${slug}: one half queued, the other refused: ${r.error}`,
+ );
}
- if (autoSubsResult?.ok) return autoSubsResult;
- return (autoSubsResult ?? missingResult)!;
+ return ok;
}
const RUN_FOR: Record<
diff --git a/editor/e2e/channel-groups.spec.ts b/editor/e2e/channel-groups.spec.ts
@@ -274,7 +274,11 @@ test("a youtube group's Transcribe counts and queues its auto-caption-only video
const rows = jobRowByKind(page, "whisper-bucket-auto-subs");
await expect(rows).toHaveCount(1);
await expect(rows).toContainText("slow-b");
- // Nothing was captionless, so no whisper-all rode along.
+ // Nothing was captionless, so neither the by-id no-transcript batch nor the
+ // scan rode along.
+ await expect(
+ jobRowByKind(page, "whisper-bucket-downloaded-no-transcript"),
+ ).toHaveCount(0);
await expect(jobRowByKind(page, "whisper-all")).toHaveCount(0);
// Let the fake whisper finish before the next spec's resetData.