Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit e4d508d23248cf9f548325a47407393338be6723
parent 50e8496a93301150cec10e0c2429b7df463ffab9
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 24 Sep 2026 18:57:49 -0400

channels: the transcribe station counts what its button queues

The group Transcribe station refused every youtube-handling channel
("a youtube-handling channel never runs whisper"). Handling never
decided it: buckets are decided by files, the runner drains
downloadedNoTranscript for every channel, and the channel page already
replaces auto-captions for any handling. A youtube video that came down
with no captions is whisper work like any other.

stationWorkFor("transcribe") drops the handling branch: a non-social
channel is always eligible, its work is |downloadedNoTranscript| +
|downloadedAutoSubsOnly| after the download exclusions, read through
one fold, transcribeStationIds. The group action queues exactly that:
transcribeAutoSubsBucketAction over the auto-caption ids, and
transcribeMissingAction when there are captionless ones (or no report).
KIND_FOR lists both kinds, so the dedupe spans them; both run on
TRANSCRIPTION_QUEUE. The strings name no method: "No channel in this
group has downloaded audio to transcribe." / "...each takes minutes."

Tests: the unit file's youtube case inverted plus four new ones (auto-
caption-only, empty, exclusions over both buckets, social's reason); the
e2e station reads "Transcribe —" on an unreported youtube group, and a
new case seeds one ASR video with audio into slow-b and sees
"Transcribe 1" queue one whisper-bucket-auto-subs job and no whisper-all.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/views/channelGroupSections.test.ts | 72+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Mcommon/views/channelGroupSections.ts | 39+++++++++++++++++++++++++++------------
Meditor/app/channels/components/ChannelGroupLine.tsx | 8+++++---
Meditor/app/channels/groupActions.ts | 71+++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------
Meditor/e2e/channel-groups.spec.ts | 88+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
5 files changed, 236 insertions(+), 42 deletions(-)

diff --git a/common/views/channelGroupSections.test.ts b/common/views/channelGroupSections.test.ts @@ -13,11 +13,14 @@ import { normalizeBuckets } from "./pipeline/stageStatus"; import { buildChannelGroupSections, slugsInGroup, + stationWorkFor, + transcribeStationIds, type ChannelGroupSection, } from "./channelGroupSections"; -// Run from this directory: -// cd editor/app/channels/lib && ../../../../node_modules/.bin/tsx --test channelGroupSections.test.ts +// Run with the rest of the common suite (`pnpm --filter yt-dlp-transcript-common +// test`), or alone from common/: +// pnpm exec tsx --test views/channelGroupSections.test.ts // Snapshot shape as WRITTEN TO DISK — same trick as channelFlow.test.ts. function snapshotOf(patch: Partial<ChannelSnapshot> = {}): ChannelSnapshot { @@ -224,7 +227,7 @@ test("download excludes members-only/deleted/private ids", () => { assert.equal(sections[0].download.total, 1); }); -test("a youtube-handling channel is not transcribe-eligible but is download-eligible", () => { +test("a youtube-handling channel IS transcribe-eligible: missing auto-subs is whisper work", () => { const sections = build( siteOf({ channels: [{ slug: "yt" }, { slug: "whisper" }] }), [ @@ -241,10 +244,65 @@ test("a youtube-handling channel is not transcribe-eligible but is download-elig const { download, transcribe } = sections[0]; assert.deepEqual(download.eligible.sort(), ["whisper", "yt"]); assert.equal(download.total, 2); - assert.deepEqual(transcribe.eligible, ["whisper"]); - // The youtube channel's two awaiting-whisper videos are NOT in the figure — - // whisper will never run on them. - assert.equal(transcribe.total, 1); + // Handling does not decide it — the files do. The youtube channel's two + // downloaded videos with no transcript at all are what its button queues. + assert.deepEqual(transcribe.eligible.sort(), ["whisper", "yt"]); + assert.equal(transcribe.total, 3); +}); + +test("the transcribe figure counts downloaded auto-caption-only videos too", () => { + const sections = build( + siteOf({ channels: [{ slug: "yt" }] }), + [ + channel("yt", { handling: "youtube" }, snapshotOf({ + buckets: normalizeBuckets({ downloadedAutoSubsOnly: ["a"] }), + })), + ], + ); + assert.deepEqual(sections[0].transcribe.eligible, ["yt"]); + assert.equal(sections[0].transcribe.total, 1); +}); + +test("a youtube channel with nothing to transcribe is eligible at 0 (skipped at click time)", () => { + const sections = build( + siteOf({ channels: [{ slug: "yt" }] }), + [channel("yt", { handling: "youtube" })], + ); + assert.deepEqual(sections[0].transcribe.eligible, ["yt"]); + assert.equal(sections[0].transcribe.total, 0); + assert.deepEqual(sections[0].transcribe.unknown, []); +}); + +test("transcribe drops excluded ids from BOTH buckets", () => { + const snapshot = snapshotOf({ + buckets: normalizeBuckets({ + downloadedNoTranscript: ["keep-1", "gone-1"], + downloadedAutoSubsOnly: ["keep-2", "gone-2"], + }), + excludedFromDownload: { + membersOnly: ["gone-1"], + deleted: [], + private: ["gone-2"], + }, + }); + const sections = build( + siteOf({ channels: [{ slug: "yt" }] }), + [channel("yt", { handling: "youtube" }, snapshot)], + ); + assert.equal(sections[0].transcribe.total, 2); + assert.deepEqual(transcribeStationIds(snapshot), { + missing: ["keep-1"], + autoSubs: ["keep-2"], + }); +}); + +test("a social channel is not transcribe-eligible, and says why", () => { + const { brief } = channel("posts", { sourceKind: "social", handling: "transcribe" }); + assert.deepEqual(stationWorkFor("transcribe", brief, LANE_ON), { + eligible: false, + work: 0, + reason: "social account", + }); }); test("a social channel is eligible for sync only", () => { diff --git a/common/views/channelGroupSections.ts b/common/views/channelGroupSections.ts @@ -105,6 +105,23 @@ export function laneOffFor( return false; } +// The two id lists the transcribe station counts and queues, after the same +// download exclusions every other station honours. Disjoint by construction: +// an ASR VTT makes a video "transcribed" for `downloadedNoTranscript`, so a +// video is in at most one of them. They are two lists because they are two +// batches — `runWhisperBatch`'s default scan drains the first, its +// replace-auto-captions mode needs an ASR track per id for the second. +export function transcribeStationIds( + snapshot: NonNullable<ChannelBrief["snapshot"]>, +): { missing: string[]; autoSubs: string[] } { + const excluded = excludedDownloadIdSet(snapshot); + const buckets = normalizeBuckets(snapshot.buckets); + return { + missing: buckets.downloadedNoTranscript.filter((id) => !excluded.has(id)), + autoSubs: buckets.downloadedAutoSubsOnly.filter((id) => !excluded.has(id)), + }; +} + export type StationChannelWork = { // Whether the operation applies to this channel at all. eligible: boolean; @@ -168,19 +185,17 @@ export function stationWorkFor( } if (station === "transcribe") { - // A `youtube`-handling channel never runs whisper, so counting it would - // inflate the figure on a button that would skip it anyway. - if (config.handling !== "transcribe") { - return { eligible: false, work: 0, reason: "not set to transcribe" }; - } + // HANDLING DOES NOT DECIDE THIS. Buckets are decided by FILES + // (channelSnapshot.ts), and a `youtube`-handling channel whose video came + // down with no captions at all is in `downloadedNoTranscript` exactly like a + // `transcribe` one — the runner drains that bucket for every channel. What + // the station counts is what its button queues: those, plus the videos whose + // only transcript is YouTube's auto-captions and whose audio is on disk + // (`downloadedAutoSubsOnly`, the channel page's replace-auto-captions half). + // `transcribeStationIds` is the one fold; the group action runs off it too. if (!snapshot) return { eligible: true, work: null }; - const excluded = excludedDownloadIdSet(snapshot); - return { - eligible: true, - work: normalizeBuckets(snapshot.buckets).downloadedNoTranscript.filter( - (id) => !excluded.has(id), - ).length, - }; + const { missing, autoSubs } = transcribeStationIds(snapshot); + return { eligible: true, work: missing.length + autoSubs.length }; } if (station === "digest") { diff --git a/editor/app/channels/components/ChannelGroupLine.tsx b/editor/app/channels/components/ChannelGroupLine.tsx @@ -60,10 +60,12 @@ const STATIONS: Station[] = [ id: "transcribe", label: "Transcribe", done: "Transcribed", - notEligible: - "No channel in this group is set to transcribe — a youtube-handling channel never runs whisper.", + // Reachable only when every member is a social account: handling does not + // decide eligibility (channelGroupSections.ts), the files do. No method is + // named — which engine runs is the transcription settings' business. + notEligible: "No channel in this group has downloaded audio to transcribe.", confirm: (group, count) => - `Transcribe ${count} downloaded video(s) across every channel in "${group}"? They run strictly one at a time on the transcription queue, and whisper is slow.`, + `Transcribe ${count} downloaded video(s) across every channel in "${group}"? They run one at a time on the transcription queue, and each takes minutes.`, run: transcribeChannelGroupAction, }, { diff --git a/editor/app/channels/groupActions.ts b/editor/app/channels/groupActions.ts @@ -1,7 +1,10 @@ "use server"; import { revalidatePath } from "next/cache"; -import { listChannelBriefs } from "yt-dlp-transcript-common/controller/channels"; +import { + listChannelBriefs, + type ChannelBrief, +} from "yt-dlp-transcript-common/controller/channels"; import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand"; import { activeSlugsForKinds } from "yt-dlp-transcript-common/jobs/syncJobs"; import { isValidGroupId } from "yt-dlp-transcript-common/lib/channelGroups"; @@ -13,11 +16,15 @@ import { laneOffFor, slugsInGroup, stationWorkFor, + transcribeStationIds, type StationId, } from "yt-dlp-transcript-common/views/channelGroupSections"; import { queueForSlugs } from "./lib/queueForSlugs"; import { downloadMissingAction, syncAction } from "./[slug]/pipelineActions"; -import { transcribeMissingAction } from "./[slug]/whisperActions"; +import { + transcribeAutoSubsBucketAction, + transcribeMissingAction, +} from "./[slug]/whisperActions"; import { backfillChannelAction } from "./[slug]/backfillActions"; import { runOperationChannelJob } from "yt-dlp-transcript-common/controller/operationJobs"; import { DIGEST_OPERATION_ID } from "yt-dlp-transcript-common/lib/operations"; @@ -42,26 +49,62 @@ export type GroupOpResult = { skipped: { slug: string; reason: string }[]; }; -// The job kind each station enqueues, for the "already running" dedupe. The job -// registry has no dedupe of its own. -const KIND_FOR: Record<StationId, string> = { - sync: "sync", - download: "download-missing", - transcribe: "whisper-all", - digest: "digest-channel-local", - speakers: "backfill-channel", +// The job kinds each station enqueues, for the "already running" dedupe. The job +// registry has no dedupe of its own. Transcribe names TWO: its button can queue +// either or both, and the dedupe spans them — pressing it again while only the +// auto-captions job runs reads "already running", not a second whisper-all. +const KIND_FOR: Record<StationId, readonly string[]> = { + sync: ["sync"], + download: ["download-missing"], + transcribe: ["whisper-all", "whisper-bucket-auto-subs"], + digest: ["digest-channel-local"], + speakers: ["backfill-channel"], }; +// THE TRANSCRIBE STATION QUEUES WHAT ITS FIGURE COUNTS — `transcribeStationIds`, +// the fold `stationWorkFor` sums. Two batches, because one cannot cover both: +// `runWhisperBatch`'s default scan drains `downloadedNoTranscript`, and its +// replace-auto-captions mode needs an ASR track per id. The channel page runs +// the same two jobs (TranscribeStage's two sections); a combined job kind would +// need its own replay spec and /jobs label and mirror nothing. Handling is not +// consulted — a youtube channel's captionless download is whisper work too. +// +// Both kinds resolve to TRANSCRIPTION_QUEUE, so the "strictly one at a time" +// note above stays true. `queued` counts CHANNELS; `jobIds` carries one id per +// channel — the whisper-all job's when both were queued, and the auto-captions +// job's stream is cancelled here exactly as queueForSlugs cancels the other. +async function transcribeChannel( + slug: string, + brief: ChannelBrief | undefined, +): Promise<StreamActionResult> { + const ids = brief?.snapshot ? transcribeStationIds(brief.snapshot) : null; + const autoSubsResult = + ids && ids.autoSubs.length > 0 + ? await transcribeAutoSubsBucketAction(slug, ids.autoSubs) + : null; + // No snapshot: the default scan finds what there is, as it always did. + const missingResult = + !ids || ids.missing.length > 0 || !autoSubsResult + ? await transcribeMissingAction(slug) + : null; + if (missingResult?.ok) { + if (autoSubsResult?.ok) void autoSubsResult.stream.cancel(); + return missingResult; + } + if (autoSubsResult?.ok) return autoSubsResult; + return (autoSubsResult ?? missingResult)!; +} + const RUN_FOR: Record< StationId, - (slug: string) => Promise<StreamActionResult> + (slug: string, brief: ChannelBrief | undefined) => Promise<StreamActionResult> > = { sync: (slug) => syncAction(slug, undefined), // runPipelineAction already refuses with {ok:false, error} for a paused // install, a per-platform 429 cooldown and low disk, so all three land in // `skipped` for free. download: (slug) => downloadMissingAction(slug), - transcribe: (slug) => transcribeMissingAction(slug), + transcribe: (slug, brief) => transcribeChannel(slug, brief), // THE LOCAL LANE, PINNED. The metered lane is behind a settings gate and a // spend cap, so it is never what a group button starts — which is why this // names the lane explicitly instead of letting the dispatcher pick the @@ -132,7 +175,7 @@ async function runGroupStation( members.has(b.slug), ); const briefBySlug = new Map(briefs.map((b) => [b.slug, b])); - const active = activeSlugsForKinds([KIND_FOR[station]]); + const active = activeSlugsForKinds(KIND_FOR[station]); const outcome = await queueForSlugs( briefs.map((b) => b.slug), @@ -153,7 +196,7 @@ async function runGroupStation( if (station !== "sync" && work.work === 0) return "nothing to do"; return null; }, - run: RUN_FOR[station], + run: (slug) => RUN_FOR[station](slug, briefBySlug.get(slug)), }, ); return { group, ...outcome }; diff --git a/editor/e2e/channel-groups.spec.ts b/editor/e2e/channel-groups.spec.ts @@ -1,9 +1,12 @@ import { test, expect, type Page } from "@playwright/test"; +import { mkdir, writeFile } from "node:fs/promises"; import { channelStage, generateReport, jobRowByKind, + pathExists, resetData, + resolvePath, writeChannelConfig, writeSite, } from "./helpers"; @@ -147,18 +150,19 @@ test("a group's Sync queues that group's channels and nothing else", async ({ await expect(syncRows).toContainText("slow-b"); }); -test("a station with no eligible channel is disabled and says why", async ({ +test("transcribe is open to a youtube channel; a station with no eligible channel is disabled and says why", async ({ page, }) => { await seed(); await page.goto("/channels?site=alpha"); - // Both fixture channels are handling: "youtube", and a youtube channel never - // runs whisper — so counting it would inflate the figure on a button that - // would skip it anyway. + // Both fixture channels are handling: "youtube" — and that no longer shuts + // the station: a youtube video that came down with no captions is whisper + // work like any other. Neither channel has reported, so the figure is the + // honest unknown, as the download station's is below. const transcribe = page.getByLabel("transcribe group News"); - await expect(transcribe).toBeDisabled(); - await expect(transcribe).toHaveAttribute("title", /never runs whisper/); + await expect(transcribe).toBeEnabled(); + await expect(transcribe).toHaveText("Transcribe —"); // The default test settings enable no operation on the backfill lane, so the // speakers station reads "off" — NOT 0 (which reads as finished) and not — @@ -206,3 +210,75 @@ test("a group figure reads — until a channel reports, then a real number", asy ) .toMatch(/Download 5$/); }); + +// An auto-caption-only video, shaped like real yt-dlp --write-auto-subs output +// (the provenance sniff reads cue settings and inline word timings), with its +// audio on disk — i.e. the `downloadedAutoSubsOnly` bucket. +const ASR_VTT = `WEBVTT +Kind: captions +Language: en + +00:00:00.030 --> 00:00:03.919 align:start position:0% +so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to + +00:00:03.919 --> 00:00:03.929 align:start position:0% +so today we're going to + +00:00:03.929 --> 00:00:07.070 align:start position:0% +so today we're going to +talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing +`; + +test("a youtube group's Transcribe counts and queues its auto-caption-only videos", async ({ + page, +}) => { + test.setTimeout(120_000); + await seed(); + const id = "asrgrp0001"; + const dataRel = `test-transcripts/channels/slow-b/data/${id}`; + const dir = resolvePath(dataRel); + await mkdir(dir, { recursive: true }); + await writeFile(`${dir}/transcript.en.vtt`, ASR_VTT); + await writeFile( + `${dir}/metadata.info.json`, + JSON.stringify({ + id, + title: `Synthetic ${id}`, + upload_date: "20240101", + duration: 60, + extractor_key: "Youtube", + webpage_url: `https://www.youtube.com/watch?v=${id}`, + subtitles: {}, + automatic_captions: { en: [{ ext: "vtt", url: "fake://subs" }] }, + }), + ); + await writeFile(`${dir}/audio.mp3`, `fake audio ${id}\n`); + await writeFile( + resolvePath("test-transcripts/channels/slow-b/playlist"), + `https://www.youtube.com/watch?v=${id}\n`, + ); + await generateReport(page, "slow-b"); + + await page.goto("/channels?site=alpha"); + const transcribe = page.getByLabel("transcribe group News"); + await expect(transcribe).toHaveText("Transcribe 1"); + await expect(transcribe).toBeEnabled(); + page.once("dialog", (d) => void d.accept()); + await transcribe.click(); + await expect(page.getByLabel("transcribe group News result")).toContainText( + /Queued 1 . skipped 0/, + { timeout: 15_000 }, + ); + + await page.goto("/jobs"); + const rows = jobRowByKind(page, "whisper-bucket-auto-subs"); + await expect(rows).toHaveCount(1); + await expect(rows).toContainText("slow-b"); + // Nothing was captionless, so no whisper-all rode along. + await expect(jobRowByKind(page, "whisper-all")).toHaveCount(0); + + // Let the fake whisper finish before the next spec's resetData. + await expect + .poll(() => pathExists(`${dataRel}/transcript.json`), { timeout: 60_000 }) + .toBe(true); +});