Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 119d590e9f0da5bed0ea652be72f695deecaa358
parent 7f6aaf786c96f9b0b8b01e021106824faf554644
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 24 Sep 2026 19:48:48 -0400

channels: the transcribe station queues by id

Review F1: the captionless half was still a whisper-all SCAN, which
takes every downloaded-and-untranscribed dir — including a video
downloaded before it went private, members-only or deleted.
downloadedNoTranscript does not filter those ids (downloadedAutoSubsOnly
does), so the figure left y out and the button transcribed it: 2 on the
label, 3 transcribed.

With a snapshot the no-transcript half is now transcribeBucketAction over
exactly the ids the figure counts (replayable as downloadedNoTranscript,
the channel page's own bucket job). The scan stays only for a channel
with no report, where there are no ids to name. KIND_FOR.transcribe
gains whisper-bucket-downloaded-no-transcript, which also closes an older
gap: the station could queue a batch beside the channel page's own bucket
job. The reviewer's walk is a unit test ([x,y] + [a], y private → [x],
[a], figure 2); the e2e checks no no-transcript batch rode along.

Review F2: a half that is refused while the other queues is logged
(console.warn). StreamActionResult's ok arm has no message to carry it.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/views/channelGroupSections.test.ts | 22++++++++++++++++++++++
Meditor/app/channels/groupActions.ts | 87+++++++++++++++++++++++++++++++++++++++++++++++++++++--------------------------
Meditor/e2e/channel-groups.spec.ts | 6+++++-
3 files changed, 86 insertions(+), 29 deletions(-)

diff --git a/common/views/channelGroupSections.test.ts b/common/views/channelGroupSections.test.ts @@ -296,6 +296,28 @@ test("transcribe drops excluded ids from BOTH buckets", () => { }); }); +test("an excluded download is neither counted nor named for the batch (the review's walk)", () => { + // x, y downloaded with no transcript; a downloaded with auto-captions only; + // y went private after it was downloaded. The figure is 2 and the two id + // lists the station queues BY ID are exactly [x] and [a] — never y. + const snapshot = snapshotOf({ + buckets: normalizeBuckets({ + downloadedNoTranscript: ["x", "y"], + downloadedAutoSubsOnly: ["a"], + }), + excludedFromDownload: { membersOnly: [], deleted: [], private: ["y"] }, + }); + const sections = build( + siteOf({ channels: [{ slug: "yt" }] }), + [channel("yt", { handling: "youtube" }, snapshot)], + ); + assert.equal(sections[0].transcribe.total, 2); + assert.deepEqual(transcribeStationIds(snapshot), { + missing: ["x"], + autoSubs: ["a"], + }); +}); + test("a social channel is not transcribe-eligible, and says why", () => { const { brief } = channel("posts", { sourceKind: "social", handling: "transcribe" }); assert.deepEqual(stationWorkFor("transcribe", brief, LANE_ON), { diff --git a/editor/app/channels/groupActions.ts b/editor/app/channels/groupActions.ts @@ -23,6 +23,7 @@ import { queueForSlugs } from "./lib/queueForSlugs"; import { downloadMissingAction, syncAction } from "./[slug]/pipelineActions"; import { transcribeAutoSubsBucketAction, + transcribeBucketAction, transcribeMissingAction, } from "./[slug]/whisperActions"; import { backfillChannelAction } from "./[slug]/backfillActions"; @@ -50,49 +51,79 @@ export type GroupOpResult = { }; // The job kinds each station enqueues, for the "already running" dedupe. The job -// registry has no dedupe of its own. Transcribe names TWO: its button can queue -// either or both, and the dedupe spans them — pressing it again while only the -// auto-captions job runs reads "already running", not a second whisper-all. +// registry has no dedupe of its own. Transcribe names THREE: its button queues +// one or both bucket jobs (or the scan, for a channel with no report), and the +// dedupe spans them all — pressing it again while any one runs, including the +// channel page's own bucket job, reads "already running", not a second batch. const KIND_FOR: Record<StationId, readonly string[]> = { sync: ["sync"], download: ["download-missing"], - transcribe: ["whisper-all", "whisper-bucket-auto-subs"], + transcribe: [ + "whisper-all", + "whisper-bucket-downloaded-no-transcript", + "whisper-bucket-auto-subs", + ], digest: ["digest-channel-local"], speakers: ["backfill-channel"], }; // THE TRANSCRIBE STATION QUEUES WHAT ITS FIGURE COUNTS — `transcribeStationIds`, -// the fold `stationWorkFor` sums. Two batches, because one cannot cover both: -// `runWhisperBatch`'s default scan drains `downloadedNoTranscript`, and its -// replace-auto-captions mode needs an ASR track per id. The channel page runs -// the same two jobs (TranscribeStage's two sections); a combined job kind would -// need its own replay spec and /jobs label and mirror nothing. Handling is not -// consulted — a youtube channel's captionless download is whisper work too. +// the fold `stationWorkFor` sums — BY ID. Two batches, because one cannot cover +// both: the no-transcript half is `transcribeBucketAction` over its ids (the +// channel page's own bucket control, replayable as `downloadedNoTranscript`), +// and the auto-captions half is `transcribeAutoSubsBucketAction`, whose +// replace mode needs an ASR track per id. By id matters: a video downloaded +// before it went private, members-only or deleted is still on disk and still +// in `downloadedNoTranscript`, and a directory SCAN would transcribe it — one +// more than the figure said. The channel page runs the same jobs; a combined +// job kind would need its own replay spec and /jobs label and mirror nothing. +// Handling is not consulted — a youtube channel's captionless download is +// whisper work too. // -// Both kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a time" -// note above stays true. `queued` counts CHANNELS; `jobIds` carries one id per -// channel — the whisper-all job's when both were queued, and the auto-captions -// job's stream is cancelled here exactly as queueForSlugs cancels the other. +// The SCAN (`transcribeMissingAction`, a whisper-all with no ids) is used ONLY +// when the channel has never reported: with no snapshot there are no ids to +// name, and the scan finds what there is, as the station always did. +// +// All three kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a +// time" note above stays true. `queued` counts CHANNELS; `jobIds` carries one +// id per channel — the no-transcript job's when both halves were queued, and +// the other stream is cancelled here exactly as queueForSlugs cancels the one +// returned. A half that fails while the other queues cannot ride on an ok +// result (StreamActionResult's ok arm has no message), so it is logged. async function transcribeChannel( slug: string, brief: ChannelBrief | undefined, ): Promise<StreamActionResult> { - const ids = brief?.snapshot ? transcribeStationIds(brief.snapshot) : null; - const autoSubsResult = - ids && ids.autoSubs.length > 0 + if (!brief?.snapshot) return transcribeMissingAction(slug); + const ids = transcribeStationIds(brief.snapshot); + const halves = [ + ids.missing.length > 0 + ? await transcribeBucketAction( + slug, + ids.missing, + undefined, + undefined, + undefined, + "downloadedNoTranscript", + ) + : null, + ids.autoSubs.length > 0 ? await transcribeAutoSubsBucketAction(slug, ids.autoSubs) - : null; - // No snapshot: the default scan finds what there is, as it always did. - const missingResult = - !ids || ids.missing.length > 0 || !autoSubsResult - ? await transcribeMissingAction(slug) - : null; - if (missingResult?.ok) { - if (autoSubsResult?.ok) void autoSubsResult.stream.cancel(); - return missingResult; + : null, + ].filter((r): r is StreamActionResult => r !== null); + // Both empty is skipped upstream as "nothing to do"; say so if reached. + if (halves.length === 0) return { ok: false, error: "nothing to do", info: true }; + const ok = halves.find((r) => r.ok); + if (!ok) return halves[0]; + for (const r of halves) { + if (r === ok) continue; + if (r.ok) void r.stream.cancel(); + else + console.warn( + `[group transcribe] ${slug}: one half queued, the other refused: ${r.error}`, + ); } - if (autoSubsResult?.ok) return autoSubsResult; - return (autoSubsResult ?? missingResult)!; + return ok; } const RUN_FOR: Record< diff --git a/editor/e2e/channel-groups.spec.ts b/editor/e2e/channel-groups.spec.ts @@ -274,7 +274,11 @@ test("a youtube group's Transcribe counts and queues its auto-caption-only video const rows = jobRowByKind(page, "whisper-bucket-auto-subs"); await expect(rows).toHaveCount(1); await expect(rows).toContainText("slow-b"); - // Nothing was captionless, so no whisper-all rode along. + // Nothing was captionless, so neither the by-id no-transcript batch nor the + // scan rode along. + await expect( + jobRowByKind(page, "whisper-bucket-downloaded-no-transcript"), + ).toHaveCount(0); await expect(jobRowByKind(page, "whisper-all")).toHaveCount(0); // Let the fake whisper finish before the next spec's resetData.