Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit dd7f18d2c5bde8350eb76df2a77f1cf5253a6c31
parent 9226d6badc815ee6e224b663b2af55a884402e6a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 25 Jul 2026 22:41:21 -0400

Merge branch 'feat/replace-auto-captions'

Diffstat:
Mcommon/controller/autoRunner.ts | 77++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------
Mcommon/controller/channelSnapshot.ts | 67++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Acommon/controller/purgeSupersededAutoSubs.ts | 110+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/transcribeOneFromQueue.ts | 20++++++++++++++++++--
Mcommon/controller/whisperBatch.ts | 26+++++++++++++++++++++++++-
Mcommon/jobs/autoQueuePolicy.test.ts | 133+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/autoQueuePolicy.ts | 54++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/jobs/jobKinds.test.ts | 24++++++++++++++++++++++++
Mcommon/jobs/jobSpec.test.ts | 5++++-
Mcommon/jobs/jobSpec.ts | 10+++++++++-
Acommon/lib/subtitleProvenance.test.ts | 133+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/subtitleProvenance.ts | 156+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/videoStatus.ts | 8++++++++
Mcommon/ytdlp/runYtdlp.ts | 34+++++++++++++++++++++++++++-------
Meditor/app/auto-queue/actions.ts | 9++++++++-
Meditor/app/auto-queue/components/AutoQueueView.tsx | 1+
Meditor/app/auto-queue/components/PolicyTreeEditor.tsx | 39++++++++++++++++++++++++++++++++++++++-
Meditor/app/auto-queue/page.tsx | 10++++++----
Meditor/app/channels/[slug]/components/RetryBucketControl.tsx | 22++++++++++++++++++----
Meditor/app/channels/[slug]/components/stages/CleanupStage.tsx | 93+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/components/stages/TranscribeStage.tsx | 155+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/lib/stageStatus.ts | 49+++++++++++++++++++++++++++++++++++++++++++------
Meditor/app/channels/[slug]/pipelineActions.ts | 19++++++++++++++++++-
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 64+++++++++++++++++++++++++++++++++++++++++++++++++++++++---------
Meditor/app/channels/[slug]/videos/[id]/page.tsx | 18+++++++++++++++++-
Meditor/app/channels/[slug]/whisperActions.ts | 87+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/e2e/auto-subs-replace.spec.ts | 425+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
27 files changed, 1801 insertions(+), 47 deletions(-)

diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -1,3 +1,4 @@ +import path from "node:path"; import { readdir } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import { getPaths } from "../lib/paths"; @@ -17,7 +18,8 @@ import { type WorkPick, buildPendingByLeaf, selectNextWork, - bucketsForKind, + defaultBucketsForPolicy, + selectableBucketsForKind, } from "../jobs/autoQueuePolicy"; import { type AutoQueueKind, @@ -33,6 +35,8 @@ import { } from "../jobs/platformBackoff"; import { type DownloadFailureClass } from "../lib/availability"; import { resolveCookiePolicy } from "../lib/cookiePolicy"; +import { isAutoSubsOnly } from "../lib/subtitleProvenance"; +import { readVideoFiles } from "../lib/videoStatus"; import { type DownloadOutcomeStatus } from "../lib/downloadOutcome"; import { downloadQueueKey } from "../lib/queueKeys"; import { readChannelConfig } from "./channels"; @@ -165,7 +169,12 @@ async function buildChannelWork( const snap = await readChannelSnapshot(paths, slug); if (!snap) continue; const buckets: Record<string, string[]> = {}; - for (const name of bucketsForKind(kind)) { + // Project every bucket a leaf could be pointed at — including the opt-in + // auto-caption ones. A leaf that names a bucket the runner never projected + // would silently find no work; projecting them here costs nothing when no + // leaf (and no policy switch) asks for them, because buildPendingByLeaf + // only walks the buckets its `defaultBuckets` / `match.bucket` name. + for (const name of selectableBucketsForKind(kind)) { // `undownloadedIds` lives at the snapshot top level; every other bucket // is under snap.buckets. const ids = @@ -190,7 +199,11 @@ export async function computeLeafPendingCounts( const policy = getSettings().autoQueue[kind]; const meta = await listChannelMeta(paths); const { channels } = await buildChannelWork(paths, kind, meta); - const pending = buildPendingByLeaf(policy.root, channels, bucketsForKind(kind)); + const pending = buildPendingByLeaf( + policy.root, + channels, + defaultBucketsForPolicy(kind, policy), + ); const counts: Record<string, number> = {}; for (const [leafId, ids] of Object.entries(pending)) counts[leafId] = ids.length; return counts; @@ -229,7 +242,6 @@ async function runLoop( signal: AbortSignal, ctx: JobRunContext, ): Promise<void> { - const defaultBuckets = bucketsForKind(kind); const tracker = makeTaskTracker(ctx, onLog); const state = await readAutoQueueState(paths); const kindState = state[kind]; @@ -347,7 +359,13 @@ async function runLoop( metaCache.map((m) => [m.slug, m.platform ?? "unknown"]), ); const { channels, owner } = await buildChannelWork(paths, kind, metaCache); - const pending = buildPendingByLeaf(policy.root, channels, defaultBuckets); + // Re-derived each iteration from the freshly-read policy, like `enabled`, so + // toggling the replace-auto-captions lane takes effect without a restart. + const pending = buildPendingByLeaf( + policy.root, + channels, + defaultBucketsForPolicy(kind, policy), + ); const exclude = new Set<string>([...live.inFlight.keys(), ...completed]); removeIds(pending, exclude); // Download only: drop videos whose platform already has an in-flight @@ -543,12 +561,39 @@ type UnitResult = { failureClass?: DownloadFailureClass; }; +// Whether this unit is an auto-captions replacement: the video's only +// transcript is a YouTube ASR VTT. Decided from disk state (one readdir + a 4 KB +// VTT head read) rather than plumbed from the pick, so it stays correct no +// matter which leaf/bucket selected the video — including the default union +// under `replaceAutoSubs`. Both branches below need it: the transcription gate +// would otherwise skip the video as "already transcribed", and the download +// branch would fetch subtitles instead of audio. +async function isAutoSubsUnit( + paths: Paths, + channelSlug: string, + videoId: string, +): Promise<boolean> { + const dir = path.join(paths.channelsDir, channelSlug, "data", videoId); + const files = await readVideoFiles(dir, { checkUntranscribable: true }); + return isAutoSubsOnly(dir, files); +} + // Run a single unit of work. Returns the outcome for counters/backoff. Hard // cancel (signal) aborts an in-flight unit; drain is handled by the loop (it // stops launching new units), so the drain signal is intentionally NOT // forwarded here. async function launchUnit(args: LaunchArgs): Promise<UnitResult> { + const replaceAutoSubs = await isAutoSubsUnit( + args.paths, + args.channelSlug, + args.pick.videoId, + ); if (args.kind === "transcription") { + if (replaceAutoSubs) { + args.onLog( + `Auto-transcribe: ${args.pick.videoId} has only YouTube auto-captions — transcribing over them.`, + ); + } const res = await transcribeOneFromQueue({ paths: args.paths, channelSlug: args.channelSlug, @@ -571,6 +616,9 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { // unit is never interrupted — it drains, then the slot goes to the manual // waiter. Mirrors how auto-download units yield to a manual sync. background: true, + // Relax the "already transcribed" gate for an ASR-only video: only OUR + // transcript.json counts, so the auto-captions get transcribed over. + replaceAutoSubs, }); return { outcome: res.outcome }; } @@ -583,8 +631,23 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { // Marked `background` so a clicked sync preempts queued units (without // interrupting a running one). The runner's per-platform in-flight gate still // keeps this to one outstanding unit per platform so the queue isn't flooded. - const config = await readChannelConfig(args.paths, args.channelSlug); - if (!config) return { outcome: "skipped" }; + const rawConfig = await readChannelConfig(args.paths, args.channelSlug); + if (!rawConfig) return { outcome: "skipped" }; + // A `handling: "youtube"` channel's normal download passes --write-auto-subs + // --write-subs --skip-download: it would re-fetch the very auto-captions we're + // replacing and never touch the audio. Force transcribe-handling for this one + // video so it downloads audio instead; the channel's stored config is + // untouched. (This is the same override the manual bucket button passes as + // handlingOverride.) Nothing to do when the channel already downloads audio. + const config = + replaceAutoSubs && rawConfig.handling !== "transcribe" + ? { ...rawConfig, handling: "transcribe" as const } + : rawConfig; + if (config !== rawConfig) { + args.onLog( + `Auto-download: ${args.pick.videoId} has only YouTube auto-captions — downloading audio (handling override: transcribe).`, + ); + } const url = await findVideoSourceUrl( args.paths, args.channelSlug, diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -33,6 +33,7 @@ import { readChannelConfig } from "./channels"; import { computeKeptVideoIds } from "./keptVideos"; import { loadMaybeMissing } from "./quickAvailabilityCheck"; import { readTranscriptCoverage } from "./normalizeTranscript"; +import { resolveVttProvenance } from "../lib/subtitleProvenance"; import { isIncompleteTranscript } from "../lib/transcriptCoverage"; export type AvailabilitySnapshot = { @@ -109,6 +110,28 @@ export type ChannelSnapshot = { // user can re-download with a different format (e.g. Original). Optional: // older snapshots lack it; readers must default to []. shortAudio: string[]; + // Videos whose ONLY transcript is YouTube's speech recognition (an English + // VTT that sniffs as `asr` — see ../lib/subtitleProvenance) and that have no + // audio on disk yet. The opt-in "replace auto-captions" lane's FIRST step: + // they need an audio download before our own engine can transcribe them. + // Mirrors noTranscript (no audio yet) and, like it, excludes videos that are + // excluded from download, untranscribable, or terminally failed. A VTT whose + // provenance can't be proven machine-generated is never listed. Optional: + // older snapshots lack it; readers must default to []. + autoSubsOnly: string[]; + // Same candidate rule as autoSubsOnly, but audio IS on disk — the lane's + // SECOND step, consumed by the opt-in auto-transcribe lane and the channel + // page's "YouTube auto-captions only" section. Mirrors + // downloadedNoTranscript. Optional: older snapshots lack it; readers must + // default to []. + downloadedAutoSubsOnly: string[]; + // Videos that now have OUR transcript (transcript.json) while the superseded + // English ASR VTT is still on disk as a backup. Never cleaned automatically — + // this is the inventory behind the Cleanup stage's manual purge button (and + // the ready-made worklist for a future AI-vs-YouTube comparison). Excludes + // do-not-clean–marked dirs so the count matches what the purge would remove. + // Optional: older snapshots lack it; readers must default to []. + supersededAutoSubs: string[]; // Undownloaded playlist videos whose effective availability says browser // cookies could recover them (AUTH_RETRY_CLASSES: needs_auth, members_only, // private). Populated in EVERY cookie mode — members_only/private are @@ -319,6 +342,14 @@ export async function generateChannelSnapshot( isVideoTranscribed(files) && !files.isUntranscribable ? await readTranscriptCoverage(dir) : null; + // Where the English VTT came from (YouTube ASR vs a human-authored + // track). Only videos that HAVE such a VTT pay the 4 KB head read — + // both the auto-subs work lane (no whisper yet) and the superseded + // backup inventory (whisper already won) need it. Same conditional + // per-video sidecar read pattern as the cues.json coverage read above. + const vttProvenance = files.ytVttFile + ? await resolveVttProvenance(dir, files.ytVttFile) + : null; return { id, files, @@ -330,6 +361,7 @@ export async function generateChannelSnapshot( excludedFromTruncatedCheck, outcome, coverage, + vttProvenance, }; }), ), @@ -399,12 +431,22 @@ export async function generateChannelSnapshot( const skippedByFilter: string[] = []; const incompleteTranscript: string[] = []; const shortAudio: string[] = []; + const autoSubsOnly: string[] = []; + const downloadedAutoSubsOnly: string[] = []; + const supersededAutoSubs: string[] = []; let transcribedWithAudioBytes = 0; let multipleAudioFormatsBytes = 0; let foreignAudioBytes = 0; let transcribed = 0; let downloaded = 0; - for (const { id, files, audioSizes, outcome, coverage } of perVideo) { + for (const { + id, + files, + audioSizes, + outcome, + coverage, + vttProvenance, + } of perVideo) { if (isVideoTranscribed(files)) transcribed++; if (isVideoDownloaded(files)) downloaded++; // A download that completed but stayed malformed after one re-download. The @@ -513,6 +555,26 @@ export async function generateChannelSnapshot( ) { nonStandardVtt.push(id); } + // --- Replace-auto-captions lane ----------------------------------------- + // A transcript that is ONLY YouTube's speech recognition. isVideoTranscribed + // counts any English VTT, so without these buckets such a video is invisible + // to every transcribe path forever. Conservative by construction: a VTT whose + // provenance we can't prove ("unknown") is never a candidate, so a human + // caption track is never scheduled for replacement. The corrupt-full-source + // and failed-short-audio `continue`s above already excluded terminal videos. + const asrVtt = files.hasYtVtt && vttProvenance === "asr"; + if (asrVtt && files.hasWhisper) { + // Our transcript won; the old ASR VTT lingers as a backup. do-not-clean + // dirs are excluded so the count matches what the manual purge removes. + if (!doNotCleanIds.has(id)) supersededAutoSubs.push(id); + } else if ( + asrVtt && + !files.isUntranscribable && + !excludedById.has(id) + ) { + if (files.audioFiles.length > 0) downloadedAutoSubsOnly.push(id); + else autoSubsOnly.push(id); + } if (files.isUntranscribable) { untranscribable.push(id); continue; @@ -615,6 +677,9 @@ export async function generateChannelSnapshot( skippedByFilter: skippedByFilter.sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), + autoSubsOnly: autoSubsOnly.sort(), + downloadedAutoSubsOnly: downloadedAutoSubsOnly.sort(), + supersededAutoSubs: supersededAutoSubs.sort(), needsCookies: needsCookies.sort(), }, undownloadedIds, diff --git a/common/controller/purgeSupersededAutoSubs.ts b/common/controller/purgeSupersededAutoSubs.ts @@ -0,0 +1,110 @@ +import path from "node:path"; +import fs from "fs-extra"; +import type { Paths } from "../lib/paths"; +import { isDoNotClean } from "../lib/doNotClean-server"; +import { readVttProvenance } from "../lib/subtitleProvenance"; +import { + WHISPER_FILENAME, + isEnglishVtt, + listTranscriptVtts, +} from "../lib/videoStatus"; + +const { pathExists, readdir, remove } = fs; + +// Delete the YouTube auto-caption VTTs that our own transcript has superseded — +// the manual counterpart to the `supersededAutoSubs` snapshot bucket, and the +// one irreversible step in the replace-auto-captions lane. It is therefore +// never wired into the auto-queue: the backup only disappears on an explicit +// click, so an AI-vs-YouTube comparison stays possible until then. +// +// Deliberately narrow. A track is removed only when ALL of these hold: +// - the dir has transcript.json (our transcript already won the index pick), +// - the track is an ENGLISH one (the en / en-orig / en-US / en-en-* family) — +// foreign-language tracks are separate content and are left alone, +// - a fresh 4 KB provenance sniff still says "asr" — a manual or unclassifiable +// track is never touched, +// - the dir is not marked do-not-clean (same shield as the Clean-audio sweep). +// +// Recovering a purged track means re-running "Download missing subs". + +export type PurgeSupersededAutoSubsOptions = { + channelSlug: string; + paths: Paths; + // When set, restrict the sweep to these ids (the bucket the button showed). + // Omitted = walk every video dir, like the Clean-audio sweep. + ids?: string[]; + onLog?: (msg: string) => void; + signal?: AbortSignal; +}; + +export type PurgeSupersededAutoSubsResult = { + inspected: number; + cleanedDirs: number; + removedFiles: number; + skipped: number; +}; + +export async function purgeSupersededAutoSubs({ + channelSlug, + paths, + ids, + onLog, + signal, +}: PurgeSupersededAutoSubsOptions): Promise<PurgeSupersededAutoSubsResult> { + const log = onLog ?? ((m: string) => console.log(m)); + const dataDir = path.join(paths.channelsDir, channelSlug, "data"); + if (!(await pathExists(dataDir))) { + log(`No data directory for ${channelSlug}`); + return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 }; + } + const onDisk = await readdir(dataDir); + const dirs = ids + ? ids.filter((id) => onDisk.includes(id)) + : onDisk; + + let cleanedDirs = 0; + let removedFiles = 0; + let skipped = 0; + + for (const id of dirs) { + if (signal?.aborted) break; + const videoDir = path.join(dataDir, id); + const entries = await readdir(videoDir).catch(() => [] as string[]); + // No transcript of our own yet — nothing has been superseded, so the VTT is + // still this video's only transcript. Never touch it. + if (!entries.includes(WHISPER_FILENAME)) continue; + const englishVtts = listTranscriptVtts(entries).filter(isEnglishVtt); + if (englishVtts.length === 0) continue; + if (await isDoNotClean(videoDir)) { + log(`Skipped ${id} (marked do not clean)`); + skipped++; + continue; + } + let removedHere = 0; + for (const name of englishVtts) { + const provenance = await readVttProvenance(videoDir, name); + if (provenance !== "asr") { + log(`Kept ${id}/${name} (${provenance} captions — not auto-generated)`); + skipped++; + continue; + } + await remove(path.join(videoDir, name)); + log(`Removed ${id}/${name}`); + removedFiles++; + removedHere++; + } + if (removedHere > 0) cleanedDirs++; + } + + const skippedNote = skipped > 0 ? ` Skipped ${skipped}.` : ""; + log( + `Purged ${removedFiles} superseded auto-caption file(s) from ${cleanedDirs} of ${dirs.length} video dir(s).${skippedNote}`, + ); + + return { + inspected: dirs.length, + cleanedDirs, + removedFiles, + skipped, + }; +} diff --git a/common/controller/transcribeOneFromQueue.ts b/common/controller/transcribeOneFromQueue.ts @@ -52,6 +52,12 @@ export type TranscribeOneOptions = { // Forwarded to the worker-pool acquire: auto-runner units pass true so they // yield a free slot to any manual (foreground) transcription waiting on it. background?: boolean; + // Replace-auto-captions lane: the caller has established that this video's + // ONLY transcript is a YouTube ASR VTT (isAutoSubsOnly), so the usual + // "already transcribed" gate — which counts any English VTT — must not skip + // it. With this set, only OUR transcript.json counts as done. The VTT stays + // on disk; pickIndexTranscript prefers transcript.json once whisper writes it. + replaceAutoSubs?: boolean; }; const skip = (): TranscribeOneResult => ({ attempted: false, outcome: "skipped" }); @@ -69,6 +75,7 @@ export async function transcribeOneFromQueue({ signal, drainSignal, background, + replaceAutoSubs = false, }: TranscribeOneOptions): Promise<TranscribeOneResult> { const log = onLog ?? ((m: string) => console.log(m)); const channelDir = path.join(paths.channelsDir, channelSlug); @@ -88,13 +95,22 @@ export async function transcribeOneFromQueue({ // channel with only VTT auto-subs end up attempted -> fail with "no audio file // found" -> get added to failed-transcriptions on every run. const videoFiles = await readVideoFiles(videoPath); - if (isVideoTranscribed(videoFiles)) { + const alreadyTranscribed = replaceAutoSubs + ? videoFiles.hasWhisper + : isVideoTranscribed(videoFiles); + if (alreadyTranscribed) { log(`Transcription for ${videoId} already exists`); return skip(); } // No real audio file: a download/source problem, not a transcription // candidate. Skip (don't fail) so a re-download lets it transcribe later. - if (!isVideoDownloaded(videoFiles)) { + // isVideoDownloaded counts a VTT as an artifact, so the replace-auto-captions + // lane must insist on actual audio — otherwise an ASR-only video with no audio + // would reach whisper and fail with "no audio file found". + const downloaded = replaceAutoSubs + ? videoFiles.audioFiles.length > 0 + : isVideoDownloaded(videoFiles); + if (!downloaded) { log(`Skipping ${videoId}: not downloaded (no audio file)`); return skip(); } diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts @@ -7,6 +7,7 @@ import { isVideoTranscribed, readVideoFiles, } from "../lib/videoStatus"; +import { isAutoSubsOnly } from "../lib/subtitleProvenance"; import { transcribeOneFromQueue } from "./transcribeOneFromQueue"; import { runPool } from "../jobs/concurrentRunner"; import { pruneFailedTranscriptions } from "./failedTranscriptions"; @@ -48,6 +49,13 @@ export type WhisperBatchOptions = { // When provided, each transcription is tracked as a per-operation task with // its own parsed progress bar on the Active Jobs screen. tracker?: TaskTracker; + // Replace-auto-captions lane (always paired with an explicit `ids` set from + // the downloadedAutoSubsOnly bucket): treat only transcript.json as "already + // transcribed", so videos whose sole transcript is a YouTube ASR VTT are + // transcribed rather than skipped. Per-video provenance is re-checked here, so + // a stale bucket entry (or a hand-passed id) can't clobber a manual caption + // track. + replaceAutoSubs?: boolean; }; export type WhisperBatchResult = { @@ -74,6 +82,7 @@ export async function runWhisperBatch({ signal, drainSignal, tracker, + replaceAutoSubs = false, }: WhisperBatchOptions): Promise<WhisperBatchResult> { const log = onLog ?? ((m: string) => console.log(m)); const channelDir = path.join(paths.channelsDir, channelSlug); @@ -137,7 +146,21 @@ export async function runWhisperBatch({ for (const id of candidateIds) { if (failedSet.has(id)) continue; const vp = path.join(dataDir, id); - const files = await readVideoFiles(vp); + const files = await readVideoFiles(vp, { checkUntranscribable: true }); + if (replaceAutoSubs) { + // Replace-auto-captions lane: only OUR transcript counts as done, and the + // VTT never counts as a downloaded artifact (real audio is required). + // Re-verify provenance per video so a stale bucket entry can't schedule a + // human-authored caption track for replacement. + if (files.hasWhisper) continue; + if (files.audioFiles.length === 0) continue; + if (!(await isAutoSubsOnly(vp, files))) { + log(`Skipping ${id}: transcript is not YouTube auto-captions.`); + continue; + } + fullItems.push(id); + continue; + } // Already transcribed if whisper ran (transcript.json) OR yt-dlp wrote an // English VTT (transcript.en.vtt or a regional/auto fallback like en-US). if (isVideoTranscribed(files)) continue; @@ -213,6 +236,7 @@ export async function runWhisperBatch({ onLog: log, signal: runSignal, drainSignal, + replaceAutoSubs, }); if (res.attempted) attempted++; if (res.outcome === "transcribed") succeededCount++; diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts @@ -7,6 +7,8 @@ import { buildPendingByLeaf, bucketsForKind, defaultAutoQueue, + defaultBucketsForPolicy, + selectableBucketsForKind, emptyAutoQueueRuntime, flattenLeaves, sanitizeAutoQueue, @@ -404,3 +406,134 @@ test("sanitizeAutoQueue: duplicate ids are de-duplicated", () => { ]; assert.equal(new Set(ids).size, ids.length, "all ids unique after sanitize"); }); + +// --- Replace-auto-captions opt-in lane -------------------------------------- + +test("the default union is unchanged by the opt-in buckets", () => { + // Every existing auto-queue setup must keep behaving exactly as before, so the + // defaults stay byte-identical and the opt-in buckets live beside them. + assert.deepEqual( + [...bucketsForKind("transcription")], + ["downloadedNoTranscript", "failedListed"], + ); + assert.deepEqual( + [...bucketsForKind("download")], + ["partialDownloads", "undownloadedIds"], + ); + assert.deepEqual( + [...defaultBucketsForPolicy("transcription", { replaceAutoSubs: false })], + [...bucketsForKind("transcription")], + ); + assert.deepEqual( + [...defaultBucketsForPolicy("download", {})], + [...bucketsForKind("download")], + ); +}); + +test("selectableBucketsForKind offers defaults plus the opt-in buckets", () => { + assert.deepEqual( + [...selectableBucketsForKind("transcription")], + ["downloadedNoTranscript", "failedListed", "downloadedAutoSubsOnly"], + ); + assert.deepEqual( + [...selectableBucketsForKind("download")], + ["partialDownloads", "undownloadedIds", "autoSubsOnly"], + ); +}); + +test("replaceAutoSubs appends the opt-in bucket at the TAIL (lowest priority)", () => { + const buckets = defaultBucketsForPolicy("transcription", { + replaceAutoSubs: true, + }); + assert.equal(buckets[buckets.length - 1], "downloadedAutoSubsOnly"); + + const channels: ChannelWork[] = [ + { + slug: "cornbreadman", + platform: "youtube", + buckets: { + downloadedNoTranscript: ["c1"], + failedListed: ["cf1"], + downloadedAutoSubsOnly: ["ca1"], + }, + }, + ]; + const root: AutoQueueGroup = { + id: "root", + mode: "strict", + children: [{ id: "L", match: { type: "all" } }], + }; + // Real work is claimed first; the auto-caption candidate lands last. + assert.deepEqual(buildPendingByLeaf(root, channels, buckets).L, [ + "c1", + "cf1", + "ca1", + ]); + // With the lane off, the auto-caption candidate isn't picked up at all. + assert.deepEqual( + buildPendingByLeaf( + root, + channels, + defaultBucketsForPolicy("transcription", { replaceAutoSubs: false }), + ).L, + ["c1", "cf1"], + ); +}); + +test("a leaf can target an opt-in bucket without the runner-wide switch", () => { + const channels: ChannelWork[] = [ + { + slug: "cornbreadman", + platform: "youtube", + buckets: { + downloadedNoTranscript: ["c1"], + downloadedAutoSubsOnly: ["ca1"], + }, + }, + { + slug: "hasanabi", + platform: "twitch", + buckets: { downloadedNoTranscript: ["h1"], downloadedAutoSubsOnly: ["ha1"] }, + }, + ]; + const root: AutoQueueGroup = { + id: "root", + mode: "strict", + children: [ + { + id: "auto-subs-cornbreadman", + match: { + type: "channel", + value: "cornbreadman", + bucket: "downloadedAutoSubsOnly", + }, + }, + { id: "rest", match: { type: "all" } }, + ], + }; + const pending = buildPendingByLeaf( + root, + channels, + defaultBucketsForPolicy("transcription", { replaceAutoSubs: false }), + ); + // Per-channel opt-in: only cornbreadman's candidate is claimed, and hasanabi's + // never enters the queue. + assert.deepEqual(pending["auto-subs-cornbreadman"], ["ca1"]); + assert.deepEqual(pending.rest, ["c1", "h1"]); +}); + +test("sanitizeAutoQueue defaults replaceAutoSubs to false", () => { + assert.equal(defaultAutoQueue().transcription.replaceAutoSubs, false); + assert.equal(sanitizeAutoQueue({}).download.replaceAutoSubs, false); + // Only an explicit `true` turns the lane on. + assert.equal( + sanitizeAutoQueue({ transcription: { replaceAutoSubs: "yes" } }) + .transcription.replaceAutoSubs, + false, + ); + assert.equal( + sanitizeAutoQueue({ transcription: { replaceAutoSubs: true } }).transcription + .replaceAutoSubs, + true, + ); +}); diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts @@ -68,6 +68,15 @@ export type AutoQueuePolicy = { // Overall ceiling on concurrent in-flight workers for this runner. null = no // runner-level cap (the worker pool / platform queues are the real throttle). maxWorkers: number | null; + // Opt in to the lowest-priority "replace YouTube auto-captions" lane: append + // this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the + // tail of the default union, so videos whose only transcript is YouTube ASR + // get re-done with our own engine whenever nothing more important is pending. + // Default false — the corpus-wide cost is large (an audio download plus a + // transcription per video). A leaf can also target the bucket by name for + // per-channel opt-in without flipping this switch. Optional: settings written + // before this field existed lack it; the sanitizer defaults it to false. + replaceAutoSubs?: boolean; root: AutoQueueGroup; }; @@ -83,6 +92,18 @@ export type AutoQueueSettings = { export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const; export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const; +// Buckets a runner will NOT draw from unless asked. Replacing YouTube's +// auto-captions with our own transcript costs an audio download plus a +// transcription per video, on a corpus where ASR-only videos outnumber +// manually-captioned ones ~100:1 — so it is never part of the default union. +// Two ways in, both explicit: a leaf can target the bucket by name (per-channel +// opt-in, see selectableBucketsForKind), or the runner's `replaceAutoSubs` +// switch appends it to the TAIL of the default union (see +// defaultBucketsForPolicy) — strictly lowest priority, since buildPendingByLeaf +// walks buckets in list order and claiming is first-match-wins. +export const TRANSCRIBE_OPT_IN_BUCKETS = ["downloadedAutoSubsOnly"] as const; +export const DOWNLOAD_OPT_IN_BUCKETS = ["autoSubsOnly"] as const; + // Local kind type — do NOT import AutoQueueKind from autoQueueState.ts, which // already imports from this module (the reverse edge would be a cycle). export function bucketsForKind( @@ -91,6 +112,35 @@ export function bucketsForKind( return kind === "transcription" ? TRANSCRIBE_BUCKETS : DOWNLOAD_BUCKETS; } +export function optInBucketsForKind( + kind: "transcription" | "download", +): readonly string[] { + return kind === "transcription" + ? TRANSCRIBE_OPT_IN_BUCKETS + : DOWNLOAD_OPT_IN_BUCKETS; +} + +// Everything a leaf may be pointed at for this kind: the default union plus the +// opt-in buckets. Backs the editor's bucket dropdown AND the runner's snapshot +// projection — a leaf can only find ids in a bucket the runner projected. +export function selectableBucketsForKind( + kind: "transcription" | "download", +): readonly string[] { + return [...bucketsForKind(kind), ...optInBucketsForKind(kind)]; +} + +// The bucket list a leaf with NO explicit bucket draws from. Identical to +// bucketsForKind unless the policy opted into auto-caption replacement, which +// appends the opt-in buckets at the tail so real work always drains first. +export function defaultBucketsForPolicy( + kind: "transcription" | "download", + policy: Pick<AutoQueuePolicy, "replaceAutoSubs">, +): readonly string[] { + return policy.replaceAutoSubs + ? [...bucketsForKind(kind), ...optInBucketsForKind(kind)] + : bucketsForKind(kind); +} + export const AUTO_QUEUE_MAX_WORKERS_MAX = 64; // --- Runtime fairness state (persisted best-effort by autoQueueState.ts) ---- @@ -348,6 +398,7 @@ export function defaultAutoQueuePolicy(): AutoQueuePolicy { return { enabled: false, maxWorkers: null, + replaceAutoSubs: false, root: { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] }, }; } @@ -365,6 +416,9 @@ function sanitizePolicy(value: unknown): AutoQueuePolicy { return { enabled: r.enabled === true, maxWorkers: clampMaxWorkers(r.maxWorkers), + // Opt-in only: anything but an explicit `true` (including a missing field on + // a pre-existing settings.json) leaves the lane off. + replaceAutoSubs: r.replaceAutoSubs === true, root: sanitizeRoot(r.root, seen), }; } diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts @@ -44,6 +44,30 @@ const OLD_LABELS: Record<string, string> = { sync: "Sync", }; +// Kinds added after the Phase 1 snapshot above. Pinned here so their label and +// drainability are asserted too, without pretending they were part of the +// original consolidation. +const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = { + // Replace-auto-captions lane: a whisper batch (drains like every other batch) + // plus the manual VTT purge (a fast file sweep — nothing to drain). + "whisper-bucket-auto-subs": { + label: "Replace auto-captions", + drainable: true, + }, + "purge-superseded-auto-subs": { + label: "Purge superseded auto-captions", + drainable: false, + }, +}; + +test("added kinds carry their pinned label and drainability", () => { + for (const [kind, meta] of Object.entries(ADDED_KINDS)) { + assert.equal(jobKindLabel(kind), meta.label, `${kind} label`); + assert.equal(isDrainableKind(kind), meta.drainable, `${kind} drainability`); + assert.notEqual(getJobKind(kind), undefined, `${kind} should be registered`); + } +}); + test("drainable kinds match the old DRAINABLE_KINDS set exactly", () => { const drainSet = new Set(OLD_DRAINABLE); for (const kind of OLD_DRAINABLE) { diff --git a/common/jobs/jobSpec.test.ts b/common/jobs/jobSpec.test.ts @@ -15,7 +15,7 @@ test("parses a minimal spec and rejects malformed input", () => { assert.equal(parseJobSpec({ kind: "sync" }), null); }); -test("accepts every replay bucket, including needsCookies", () => { +test("accepts every replay bucket, including the auto-caption lane", () => { for (const bucket of [ "partialDownloads", "noTranscript", @@ -23,6 +23,9 @@ test("accepts every replay bucket, including needsCookies", () => { "incompleteTranscript", "shortAudio", "needsCookies", + "autoSubsOnly", + "downloadedAutoSubsOnly", + "supersededAutoSubs", ]) { const spec = parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket }); assert.equal(spec?.bucket, bucket); diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts @@ -21,7 +21,12 @@ export type ReplayBucket = | "downloadedNoTranscript" | "incompleteTranscript" | "shortAudio" - | "needsCookies"; + | "needsCookies" + // Replace-auto-captions lane: ASR-only without audio (fetch audio), ASR-only + // with audio (transcribe), and the kept-VTT backups (manual purge). + | "autoSubsOnly" + | "downloadedAutoSubsOnly" + | "supersededAutoSubs"; export type JobSpec = { kind: string; @@ -42,6 +47,9 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([ "incompleteTranscript", "shortAudio", "needsCookies", + "autoSubsOnly", + "downloadedAutoSubsOnly", + "supersededAutoSubs", ]); // Defensive parse for a spec read back from JSON (a sidecar or the bookmarks diff --git a/common/lib/subtitleProvenance.test.ts b/common/lib/subtitleProvenance.test.ts @@ -0,0 +1,133 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + provenanceFromMetadata, + readVttProvenance, + resolveVttProvenance, + sniffVttProvenance, + vttTrackName, +} from "./subtitleProvenance"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/subtitleProvenance.test.ts + +// Shaped after real yt-dlp --write-auto-subs output: every cue carries +// `align:start position:0%` and inline word timings. +const ASR_HEAD = `WEBVTT +Kind: captions +Language: en + +00:00:00.030 --> 00:00:03.919 align:start position:0% + +so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to + +00:00:03.919 --> 00:00:03.929 align:start position:0% +so today we're going to + +00:00:03.929 --> 00:00:07.070 align:start position:0% +so today we're going to +talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing +`; + +// Shaped after a human-uploaded/manual track: plain cues, no cue settings, no +// inline word timings. +const MANUAL_HEAD = `WEBVTT +Kind: captions +Language: en + +00:00:01.000 --> 00:00:04.000 +So today we're going to talk about the whole thing. + +00:00:04.000 --> 00:00:08.500 +It's a long story, but bear with me. + +00:00:08.500 --> 00:00:12.000 +Here we go. +`; + +test("sniffVttProvenance classifies YouTube ASR output as asr", () => { + assert.equal(sniffVttProvenance(ASR_HEAD), "asr"); +}); + +test("sniffVttProvenance classifies a manual caption track as manual", () => { + assert.equal(sniffVttProvenance(MANUAL_HEAD), "manual"); +}); + +test("sniffVttProvenance detects asr from word timings alone", () => { + // Some ASR tracks are served without the align/position cue settings but keep + // the inline word timings — still machine-generated. + const head = ASR_HEAD.replace(/ align:start position:\d+%/g, ""); + assert.equal(sniffVttProvenance(head), "asr"); +}); + +test("sniffVttProvenance returns unknown for too little evidence", () => { + assert.equal(sniffVttProvenance(""), "unknown"); + assert.equal(sniffVttProvenance("not a vtt at all"), "unknown"); + assert.equal( + sniffVttProvenance("WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nOnly one cue.\n"), + "unknown", + ); +}); + +test("sniffVttProvenance returns unknown when markers are present but sparse", () => { + const head = `WEBVTT + +00:00:00.000 --> 00:00:02.000 align:start position:0% +one + +00:00:02.000 --> 00:00:04.000 +two + +00:00:04.000 --> 00:00:06.000 +three + +00:00:06.000 --> 00:00:08.000 +four +`; + assert.equal(sniffVttProvenance(head), "unknown"); +}); + +test("provenanceFromMetadata prefers subtitles over automatic_captions", () => { + const meta = { + subtitles: { en: [{ ext: "vtt" }] }, + automatic_captions: { en: [{ ext: "vtt" }], "en-orig": [{ ext: "vtt" }] }, + }; + assert.equal(provenanceFromMetadata(meta, "en"), "manual"); + assert.equal(provenanceFromMetadata(meta, "en-orig"), "asr"); + assert.equal(provenanceFromMetadata(meta, "fr"), "unknown"); + assert.equal(provenanceFromMetadata(null, "en"), "unknown"); + assert.equal(provenanceFromMetadata({}, "en"), "unknown"); +}); + +test("vttTrackName extracts the language segment", () => { + assert.equal(vttTrackName("transcript.en.vtt"), "en"); + assert.equal(vttTrackName("transcript.en-US.vtt"), "en-US"); + assert.equal(vttTrackName("transcript.json"), null); + assert.equal(vttTrackName("transcript.cues.json"), null); +}); + +test("readVttProvenance sniffs only the head of a file on disk", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "subprov-")); + await writeFile(path.join(dir, "transcript.en.vtt"), ASR_HEAD); + await writeFile(path.join(dir, "transcript.en-US.vtt"), MANUAL_HEAD); + assert.equal(await readVttProvenance(dir, "transcript.en.vtt"), "asr"); + assert.equal(await readVttProvenance(dir, "transcript.en-US.vtt"), "manual"); + assert.equal(await readVttProvenance(dir, "missing.vtt"), "unknown"); +}); + +test("resolveVttProvenance falls back to metadata for an undecidable file", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "subprov-")); + // One cue only: the sniff can't judge it. + await writeFile( + path.join(dir, "transcript.en.vtt"), + "WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhi\n", + ); + assert.equal(await resolveVttProvenance(dir, "transcript.en.vtt"), "unknown"); + await writeFile( + path.join(dir, "metadata.info.json"), + JSON.stringify({ automatic_captions: { en: [{ ext: "vtt" }] } }), + ); + assert.equal(await resolveVttProvenance(dir, "transcript.en.vtt"), "asr"); +}); diff --git a/common/lib/subtitleProvenance.ts b/common/lib/subtitleProvenance.ts @@ -0,0 +1,156 @@ +import path from "node:path"; +import { createReadStream } from "node:fs"; +import { readFile } from "node:fs/promises"; +import type { VideoFiles } from "./videoStatus"; + +// Where a VTT transcript came from: YouTube's speech recognition ("asr", what +// yt-dlp downloads under --write-auto-subs) or a human-authored/uploaded track +// ("manual", --write-subs). "unknown" means we could not tell, and is treated as +// manual everywhere it matters — we never replace a transcript we can't prove is +// machine-generated. +// +// The authoritative answer lives in metadata.info.json (`subtitles` vs +// `automatic_captions`), but those files average ~490 KB, so parsing one per +// video per snapshot regen is not viable across a 77k-video corpus. Instead we +// sniff the first few KB of the VTT itself: YouTube's ASR cues carry +// `align:start position:N%` cue settings and inline `<hh:mm:ss.mmm>` word +// timings, and manual tracks carry neither. Measured on real data across 5 +// channels (12 manual + 12 ASR): ASR files marked 96–100% of cues, manual files +// 0%. The metadata parse is kept as the tie-breaker for the rare "unknown". + +export type SubtitleProvenance = "asr" | "manual" | "unknown"; + +// One read of this many bytes per English-VTT-having video per snapshot regen. +// Big enough to hold several cues even for long, densely-marked ASR output. +export const PROVENANCE_SNIFF_BYTES = 4096; + +// Below this many cues in the sniffed window there isn't enough evidence to +// call it either way (a 2-cue stub could be anything). +const MIN_CUES = 2; + +// Fraction of cues that must carry ASR fingerprints for an "asr" verdict. +const ASR_RATIO = 0.5; + +const CUE_ARROW_RE = /-->/g; +// yt-dlp writes YouTube's ASR cue settings verbatim: "align:start position:0%". +const ASR_ALIGN_RE = /align:start position:\d+%/g; +// Inline per-word timing tags, e.g. "<00:00:03.919>" — ASR-only in practice. +const WORD_TIMING_RE = /<\d{2}:\d{2}:\d{2}\.\d{3}>/g; + +function count(re: RegExp, text: string): number { + // Each call needs its own lastIndex reset — the /g regexes are module-level. + re.lastIndex = 0; + let n = 0; + while (re.exec(text) !== null) n++; + return n; +} + +// Pure classifier over the first PROVENANCE_SNIFF_BYTES of a VTT file. Kept +// separate from the I/O so it is directly unit-testable. +export function sniffVttProvenance(head: string): SubtitleProvenance { + const cues = count(CUE_ARROW_RE, head); + if (cues < MIN_CUES) return "unknown"; + const aligned = count(ASR_ALIGN_RE, head); + const worded = count(WORD_TIMING_RE, head); + if (aligned / cues >= ASR_RATIO || worded / cues >= ASR_RATIO) return "asr"; + // No ASR fingerprint at all on a file with real cues: a human-authored track. + if (aligned === 0 && worded === 0) return "manual"; + // Some markers but not enough to clear the bar — refuse to guess. + return "unknown"; +} + +// Read only the head of the VTT (createReadStream start/end), never the whole +// file: a 3-hour ASR transcript is multiple MB. +export async function readVttProvenance( + videoDir: string, + vttFile: string, +): Promise<SubtitleProvenance> { + let head: string; + try { + head = await readHead(path.join(videoDir, vttFile), PROVENANCE_SNIFF_BYTES); + } catch { + return "unknown"; + } + return sniffVttProvenance(head); +} + +function readHead(file: string, bytes: number): Promise<string> { + return new Promise((resolve, reject) => { + const stream = createReadStream(file, { + encoding: "utf8", + start: 0, + end: bytes - 1, + }); + let out = ""; + stream.on("data", (chunk) => { + out += chunk; + }); + stream.on("error", reject); + stream.on("end", () => resolve(out)); + }); +} + +// The language/track segment of a transcript sidecar filename: +// "transcript.en-US.vtt" -> "en-US". Null for anything else. +export function vttTrackName(filename: string): string | null { + const m = filename.match(/^transcript\.([^.]+)\.vtt$/); + return m ? m[1] : null; +} + +type CaptionMetadata = { + subtitles?: Record<string, unknown>; + automatic_captions?: Record<string, unknown>; +}; + +// The authoritative check, from an already-parsed metadata.info.json. Mirrors +// the shape read by runYtdlp/downloadOneManaged: `subtitles` is what YouTube +// calls manual/uploaded captions, `automatic_captions` is ASR. A track listed in +// both counts as manual (the manual file is what yt-dlp would have written). +export function provenanceFromMetadata( + meta: unknown, + track: string, +): SubtitleProvenance { + if (!meta || typeof meta !== "object") return "unknown"; + const m = meta as CaptionMetadata; + const subs = m.subtitles; + const auto = m.automatic_captions; + const has = (rec: Record<string, unknown> | undefined): boolean => + !!rec && typeof rec === "object" && Object.prototype.hasOwnProperty.call(rec, track); + if (has(subs)) return "manual"; + if (has(auto)) return "asr"; + return "unknown"; +} + +// Sniff first; fall back to the (expensive) metadata parse only when the sniff +// can't decide. That keeps the 490 KB read rare while still classifying the odd +// stub file correctly. +export async function resolveVttProvenance( + videoDir: string, + vttFile: string, +): Promise<SubtitleProvenance> { + const sniffed = await readVttProvenance(videoDir, vttFile); + if (sniffed !== "unknown") return sniffed; + const track = vttTrackName(vttFile); + if (!track) return "unknown"; + try { + const raw = await readFile(path.join(videoDir, "metadata.info.json"), "utf8"); + return provenanceFromMetadata(JSON.parse(raw), track); + } catch { + return "unknown"; + } +} + +// True when this video's ONLY transcript is YouTube ASR — the work-lane +// candidate rule, shared by the snapshot buckets, the whisper gate +// (transcribeOneFromQueue) and the auto-runner's download override so all four +// agree on what "auto-captions only" means. `files` is the already-read dir +// listing; only the 4 KB VTT sniff is done here. +export async function isAutoSubsOnly( + videoDir: string, + files: VideoFiles, +): Promise<boolean> { + if (!files.ytVttFile || files.hasWhisper || files.isUntranscribable) { + return false; + } + return (await resolveVttProvenance(videoDir, files.ytVttFile)) === "asr"; +} diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -156,6 +156,14 @@ function englishVttRank(track: string): number { if (/^en-en(?:-|$)/.test(track)) return 3; // auto-translated en→en variants return 2; // regional/manual en-US, en-GB, … } +// Whether a transcript sidecar filename is one of the ENGLISH VTT tracks +// resolvePrimaryVtt considers (en, en-orig, en-US, en-en-*, …). Foreign-language +// tracks — including translations like es-en-US — return false. Exported for the +// superseded-auto-caption purge, which must leave non-English tracks alone. +export function isEnglishVtt(name: string): boolean { + return EN_VTT_RE.test(name); +} + export function resolvePrimaryVtt(entries: string[]): string | null { let best: { name: string; rank: number } | null = null; for (const e of entries) { diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -14,6 +14,7 @@ import { checkDiskSpace } from "../lib/diskSpace"; import { formatBytes } from "../lib/format"; import { detectPlatform } from "../lib/platform"; import { isRealAudioFile } from "../lib/videoStatus"; +import { readVttProvenance } from "../lib/subtitleProvenance"; import type { Paths } from "../lib/paths"; import { EXCLUDED_FROM_DOWNLOAD, @@ -111,6 +112,15 @@ export type RunYtdlpOpts = { // mutating the channel config on disk. Useful for retrying old "youtube" // videos as "transcribe". handlingOverride?: ChannelHandling; + // retry-bucket only (the "YouTube auto-captions only" bucket): these videos + // ALREADY have an English VTT, which destinationExists() normally reads as + // "already downloaded" — the run would prefilter every id away. With this set, + // an English VTT that still sniffs as YouTube ASR no longer counts as a + // destination, so the audio download proceeds. A manual (or unclassifiable) + // caption track still suppresses the download, so a human transcript is never + // the reason we re-fetch. Paired with handlingOverride: "transcribe" by the + // caller, since a youtube-handling download would fetch subs, not audio. + replaceAutoSubs?: boolean; // Called when a per-video download fails with a rate-limit (HTTP 429) or // network error, so the caller can record the SHARED per-platform cooldown // (see common/jobs/downloadBackoff.ts). This entangles manual sync/download @@ -531,7 +541,11 @@ async function downloadPlaylistManaged( tofetch.push(url); continue; } - if (await destinationExists(dataDir, dirId, effectiveHandling)) { + if ( + await destinationExists(dataDir, dirId, effectiveHandling, { + replaceAutoSubs: opts.replaceAutoSubs, + }) + ) { alreadyComplete++; continue; } @@ -1107,15 +1121,21 @@ export async function destinationExists( dataDir: string, id: string, handling: ChannelHandling, + opts: { replaceAutoSubs?: boolean } = {}, ): Promise<boolean> { const dir = path.join(dataDir, id); const entries = await readdir(dir).catch(() => [] as string[]); - // Either transcript counts as "already done" — channels can be hybrid. - if ( - entries.includes("transcript.en.vtt") || - entries.includes("transcript.json") - ) { - return true; + // Our own transcript always counts as "already done". + if (entries.includes("transcript.json")) return true; + // A VTT normally counts too — channels can be hybrid. The replace-auto-captions + // run is the one exception, and only for a track that still sniffs as YouTube + // ASR: that is precisely the transcript we are here to replace, so it must not + // suppress the audio download the replacement needs. + if (entries.includes("transcript.en.vtt")) { + const replaceable = + opts.replaceAutoSubs === true && + (await readVttProvenance(dir, "transcript.en.vtt")) === "asr"; + if (!replaceable) return true; } // Transcribe channels treat raw audio as a download in progress so we don't // re-fetch it before whisper runs. YouTube channels expect a .vtt; an audio diff --git a/editor/app/auto-queue/actions.ts b/editor/app/auto-queue/actions.ts @@ -26,7 +26,13 @@ export type SaveResult = { ok: true } | { ok: false; error: string }; // its next iteration (getSettings reads from disk), so it stops on its own. export async function saveAutoQueueAction( kind: AutoQueueKind, - input: { enabled: boolean; maxWorkers: number | null; root: AutoQueueGroup }, + input: { + enabled: boolean; + maxWorkers: number | null; + // Opt in to the lowest-priority replace-auto-captions lane (default false). + replaceAutoSubs: boolean; + root: AutoQueueGroup; + }, ): Promise<SaveResult> { const current = getSettings(); const next: SiteSettings = { @@ -36,6 +42,7 @@ export async function saveAutoQueueAction( [kind]: { enabled: input.enabled, maxWorkers: input.maxWorkers, + replaceAutoSubs: input.replaceAutoSubs === true, root: input.root, }, }, diff --git a/editor/app/auto-queue/components/AutoQueueView.tsx b/editor/app/auto-queue/components/AutoQueueView.tsx @@ -183,6 +183,7 @@ function KindPanel({ kind={kind} initialEnabled={status.policy.enabled} initialMaxWorkers={status.policy.maxWorkers} + initialReplaceAutoSubs={status.policy.replaceAutoSubs === true} initialRoot={status.policy.root} channels={channels} platforms={platforms} diff --git a/editor/app/auto-queue/components/PolicyTreeEditor.tsx b/editor/app/auto-queue/components/PolicyTreeEditor.tsx @@ -127,6 +127,7 @@ export function PolicyTreeEditor({ kind, initialEnabled, initialMaxWorkers, + initialReplaceAutoSubs, initialRoot, channels, platforms, @@ -135,6 +136,7 @@ export function PolicyTreeEditor({ kind: AutoQueueKind; initialEnabled: boolean; initialMaxWorkers: number | null; + initialReplaceAutoSubs: boolean; initialRoot: AutoQueueGroup; channels: { slug: string; name: string | null }[]; platforms: string[]; @@ -143,6 +145,9 @@ export function PolicyTreeEditor({ const [root, setRoot] = useState<AutoQueueGroup>(initialRoot); const [enabled, setEnabled] = useState(initialEnabled); const [maxWorkers, setMaxWorkers] = useState<number | null>(initialMaxWorkers); + const [replaceAutoSubs, setReplaceAutoSubs] = useState( + initialReplaceAutoSubs, + ); const [saving, setSaving] = useState(false); const [result, setResult] = useState<SaveResult | null>(null); @@ -160,7 +165,14 @@ export function PolicyTreeEditor({ setSaving(true); setResult(null); try { - setResult(await saveAutoQueueAction(kind, { enabled, maxWorkers, root })); + setResult( + await saveAutoQueueAction(kind, { + enabled, + maxWorkers, + replaceAutoSubs, + root, + }), + ); } catch (e) { setResult({ ok: false, error: (e as Error).message }); } finally { @@ -198,6 +210,31 @@ export function PolicyTreeEditor({ </label> </div> + {/* Lowest-priority lane, off by default. Appending the opt-in bucket to + the TAIL of the default union is what makes it lowest priority: pending + work is claimed bucket-by-bucket in list order. */} + <label className="flex items-start gap-2 text-sm"> + <input + type="checkbox" + checked={replaceAutoSubs} + onChange={(e) => setReplaceAutoSubs(e.target.checked)} + aria-label={`replace YouTube auto-captions for auto-${kind}`} + className="mt-1" + /> + <span> + Replace YouTube auto-captions + <span className="block text-xs text-muted-foreground"> + When nothing else is pending,{" "} + {kind === "transcription" + ? "transcribe videos whose only transcript is YouTube's speech recognition (and whose audio is already downloaded)" + : "download audio for videos whose only transcript is YouTube's speech recognition"} + . Strictly lowest priority, and off by default — every candidate + costs a download plus a transcription. Rules below can also target + the bucket directly for per-channel opt-in. + </span> + </span> + </label> + <NodeEditor node={root} depth={0} diff --git a/editor/app/auto-queue/page.tsx b/editor/app/auto-queue/page.tsx @@ -3,7 +3,7 @@ import Link from "next/link"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { listChannels } from "yt-dlp-transcript-common/controller/channels"; import { PLATFORM_VALUES } from "yt-dlp-transcript-common/lib/platform"; -import { bucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy"; +import { selectableBucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy"; import { buildAutoQueueStatusPayload } from "./status"; import { AutoQueueView } from "./components/AutoQueueView"; @@ -12,10 +12,12 @@ export const dynamic = "force-dynamic"; export const metadata: Metadata = { title: "Auto-queue" }; // Buckets each runner can draw from — derived from the policy engine's single -// source of truth (bucketsForKind) so the picker can't drift from the runner. +// source of truth (selectableBucketsForKind) so the picker can't drift from the +// runner. Includes the opt-in auto-caption buckets: a leaf that names one gets +// per-channel opt-in without flipping the runner-wide switch. const BUCKETS_BY_KIND = { - transcription: [...bucketsForKind("transcription")], - download: [...bucketsForKind("download")], + transcription: [...selectableBucketsForKind("transcription")], + download: [...selectableBucketsForKind("download")], }; export default async function AutoQueuePage() { diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx @@ -20,6 +20,13 @@ type Props = { // Needs-cookies bucket: force cookie mode "always" for the run and label the // button "Download with cookies" so the manual cookie path is explicit. forceCookies?: boolean; + // Auto-captions-only bucket: these videos have a VTT but no audio, so the run + // needs BOTH the transcribe handling (a youtube-handling download would fetch + // subs, not audio) and the destination-exists bypass. The handling override is + // preselected rather than forced, so it stays visible and adjustable. + replaceAutoSubs?: boolean; + // Overrides the default "Retry (n)" button text. + buttonLabel?: string; }; export function RetryBucketControl({ @@ -30,9 +37,13 @@ export function RetryBucketControl({ existingQueues, bucketKey, forceCookies, + replaceAutoSubs, + buttonLabel, }: Props) { const [queue, setQueue] = useState(defaultQueueKey); - const [handlingOverride, setHandlingOverride] = useState(""); + const [handlingOverride, setHandlingOverride] = useState( + replaceAutoSubs ? "transcribe" : "", + ); const [abortOnError, setAbortOnError] = useState(false); if (ids.length === 0) return null; @@ -52,13 +63,16 @@ export function RetryBucketControl({ handlingOverride || undefined, bucketKey, forceCookies, + replaceAutoSubs, ) } cancelAction={cancelJobAction} buttonLabel={ - forceCookies - ? `Download with cookies (${ids.length})` - : `Retry (${ids.length})` + buttonLabel + ? `${buttonLabel} (${ids.length})` + : forceCookies + ? `Download with cookies (${ids.length})` + : `Retry (${ids.length})` } runningLabel="Retrying…" label={`Retry ${actionLabel}`} diff --git a/editor/app/channels/[slug]/components/stages/CleanupStage.tsx b/editor/app/channels/[slug]/components/stages/CleanupStage.tsx @@ -9,6 +9,7 @@ import { checkKeptDeletedAction, cleanAudioAction, cleanExtraAudioFormatsAction, + purgeSupersededAutoSubsAction, removeWrongFormatAudioAction, } from "../../whisperActions"; import { persistKeptAction } from "../../persistActions"; @@ -20,6 +21,9 @@ type Props = { existingQueues: string[]; multipleAudioFormatIds: string[]; foreignAudioIds: string[]; + // Videos where our transcript won and the superseded YouTube ASR VTT is still + // on disk as a backup. Never cleaned automatically — only by the purge below. + supersededAutoSubsIds: string[]; transcodeApplies: boolean; // Estimated bytes each cleanup would reclaim, as of the last report. transcribedAudioBytes: number; @@ -41,6 +45,7 @@ export function CleanupStage({ existingQueues, multipleAudioFormatIds, foreignAudioIds, + supersededAutoSubsIds, transcodeApplies, transcribedAudioBytes, extraFormatsBytes, @@ -90,6 +95,14 @@ export function CleanupStage({ } /> </div> + {supersededAutoSubsIds.length > 0 && ( + <SupersededAutoSubsSection + slug={slug} + ids={supersededAutoSubsIds} + existingQueues={existingQueues} + defaultQueueKey={defaultQueueKey} + /> + )} {transcodeApplies && ( <div className="flex flex-col gap-2"> <Heading @@ -124,6 +137,86 @@ export function CleanupStage({ ); } +// The only thing that deletes a kept auto-caption backup. Deliberately manual: +// nothing auto-queues it, so an AI-vs-YouTube comparison stays possible for as +// long as you want it. The sweep re-sniffs every file and removes only English +// tracks that still read as YouTube ASR; do-not-clean dirs are skipped. +function SupersededAutoSubsSection({ + slug, + ids, + existingQueues, + defaultQueueKey, +}: { + slug: string; + ids: string[]; + existingQueues: string[]; + defaultQueueKey: string; +}) { + const [queue, setQueue] = useState(defaultQueueKey); + const [confirmText, setConfirmText] = useState(""); + const armed = confirmText === "purge"; + return ( + <div + aria-label="superseded auto captions section" + className="flex flex-col gap-2 rounded border border-border p-3" + > + <div> + <h3 className="text-base font-semibold"> + Superseded auto-captions ({ids.length}) + </h3> + <p className="text-sm text-muted-foreground"> + These videos now have our own <code>transcript.json</code>, which wins + everywhere, while YouTube&apos;s original auto-caption VTT is still on + disk as a backup. Purging is irreversible — recovering a track means + re-running <em>Download missing subs</em>. Only English tracks that + still read as auto-generated are removed; manual captions and + foreign-language tracks are left alone, as are dirs marked{" "} + <em>do not clean</em>. + </p> + </div> + <VideoIdList + slug={slug} + ids={ids} + ariaLabel="superseded auto captions list" + emptyAriaLabel="superseded auto captions empty" + emptyMessage="None" + itemAriaLabel={(id) => `superseded auto captions ${id}`} + /> + <div className="flex flex-col gap-2"> + <label className="flex items-center gap-2 text-xs text-muted-foreground"> + Type + <span className="font-mono">purge</span> + to confirm + <input + type="text" + value={confirmText} + onChange={(e) => setConfirmText(e.target.value)} + aria-label="confirm purge superseded auto captions" + className="rounded border border-border bg-card px-2 py-0.5 font-mono" + /> + </label> + <StreamActionLog + trigger={() => purgeSupersededAutoSubsAction(slug, queue)} + cancelAction={cancelJobAction} + buttonLabel={`Purge superseded auto-captions (${ids.length})`} + runningLabel="Purging…" + label="Purge superseded auto-captions" + disabled={!armed} + extraControls={ + <QueueControl + value={queue} + onChange={setQueue} + defaultQueueKey={defaultQueueKey} + existingQueues={existingQueues} + actionLabel="Purge superseded auto-captions" + /> + } + /> + </div> + </div> + ); +} + function WrongFormatAudioSection({ slug, ids, diff --git a/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx b/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx @@ -16,9 +16,11 @@ import { import { cancelJobAction } from "../../../../jobs/actions"; import { clearFailedTranscriptionsAction, + transcribeAutoSubsBucketAction, transcribeBucketAction, transcribeMissingAction, } from "../../whisperActions"; +import { RetryBucketControl } from "../RetryBucketControl"; import { VideoIdList } from "../VideoIdList"; type AudioFormatChoice = AudioFormat | "any"; @@ -28,7 +30,14 @@ type Props = { existingQueues: string[]; failedVideoIds: string[]; downloadedNoTranscriptIds: string[]; + // Replace-auto-captions lane: videos whose only transcript is YouTube ASR, + // split by whether the audio our engine needs is on disk yet. + autoSubsOnlyIds: string[]; + downloadedAutoSubsOnlyIds: string[]; defaultQueueKey: string; + // Queue key for the download half of the lane (the platform queue a sync uses), + // which is not the transcription queue the rest of this stage runs on. + downloadQueueKey: string; missingShard: ShardConfigSummary | null; }; @@ -37,11 +46,15 @@ export function TranscribeStage({ existingQueues, failedVideoIds, downloadedNoTranscriptIds, + autoSubsOnlyIds, + downloadedAutoSubsOnlyIds, defaultQueueKey, + downloadQueueKey, missingShard, }: Props) { const channelQueueKey = `channel:${slug}`; const bucketCount = downloadedNoTranscriptIds.length; + const autoSubsCount = autoSubsOnlyIds.length + downloadedAutoSubsOnlyIds.length; return ( <div className="flex flex-col gap-6"> @@ -60,6 +73,16 @@ export function TranscribeStage({ missingShard={missingShard} demoted={bucketCount > 0} /> + {autoSubsCount > 0 && ( + <AutoSubsSection + slug={slug} + noAudioIds={autoSubsOnlyIds} + withAudioIds={downloadedAutoSubsOnlyIds} + existingQueues={existingQueues} + transcribeQueueKey={defaultQueueKey} + downloadQueueKey={downloadQueueKey} + /> + )} <FailedTranscriptionsSection slug={slug} ids={failedVideoIds} @@ -70,6 +93,138 @@ export function TranscribeStage({ ); } +// The manual face of the replace-auto-captions lane. Two steps, because a video +// walks them one restart-safe step at a time: fetch the audio (platform download +// queue), then transcribe it (worker pool). The original VTT is kept as a backup +// either way — the Cleanup stage purges those on demand. +function AutoSubsSection({ + slug, + noAudioIds, + withAudioIds, + existingQueues, + transcribeQueueKey, + downloadQueueKey, +}: { + slug: string; + noAudioIds: string[]; + withAudioIds: string[]; + existingQueues: string[]; + transcribeQueueKey: string; + downloadQueueKey: string; +}) { + const [queue, setQueue] = useState(transcribeQueueKey); + const [audioFormat, setAudioFormat] = useState<AudioFormatChoice>("any"); + const [strictFormat, setStrictFormat] = useState(false); + const total = noAudioIds.length + withAudioIds.length; + return ( + <div + aria-label="auto captions only section" + className="flex flex-col gap-3 rounded border border-border p-3" + > + <div> + <h3 className="text-base font-semibold"> + YouTube auto-captions only ({total}) + </h3> + <p className="text-sm text-muted-foreground"> + These videos have no transcript of our own — only YouTube&apos;s + speech recognition (no punctuation, rolling duplicate cues,{" "} + <code>[Music]</code> filler). Replacing them costs an audio download + plus a transcription each, so nothing happens automatically unless the{" "} + <a href="/auto-queue" className="underline"> + auto-queue + </a>{" "} + opts in. Human-written captions are never listed here. The old VTT is + kept as a backup; purge it from the Cleanup stage when you no longer + want it. + </p> + </div> + + <div className="flex flex-col gap-2"> + <h4 className="text-sm font-semibold"> + Needs audio ({noAudioIds.length}) + </h4> + <p className="text-xs text-muted-foreground"> + Step 1 — download the audio our engine transcribes from. Runs as a + <code> transcribe</code>-handling download so yt-dlp fetches audio + instead of re-fetching the subtitles. + </p> + <VideoIdList + slug={slug} + ids={noAudioIds} + ariaLabel="auto captions needing audio list" + emptyAriaLabel="auto captions needing audio empty" + emptyMessage="None" + itemAriaLabel={(id) => `auto captions needing audio ${id}`} + /> + <RetryBucketControl + slug={slug} + ids={noAudioIds} + actionLabel="auto-captions audio" + defaultQueueKey={downloadQueueKey} + existingQueues={existingQueues} + bucketKey="autoSubsOnly" + replaceAutoSubs + buttonLabel="Fetch audio" + /> + </div> + + <div className="flex flex-col gap-2"> + <h4 className="text-sm font-semibold"> + Ready to transcribe ({withAudioIds.length}) + </h4> + <p className="text-xs text-muted-foreground"> + Step 2 — run whisper over the downloaded audio. The resulting{" "} + <code>transcript.json</code> takes precedence over any VTT everywhere + (index, viewer, export). + </p> + <VideoIdList + slug={slug} + ids={withAudioIds} + ariaLabel="auto captions ready to transcribe list" + emptyAriaLabel="auto captions ready to transcribe empty" + emptyMessage="None" + itemAriaLabel={(id) => `auto captions ready to transcribe ${id}`} + /> + {withAudioIds.length > 0 && ( + <StreamActionLog + trigger={() => + transcribeAutoSubsBucketAction( + slug, + withAudioIds, + queue, + audioFormat === "any" ? undefined : audioFormat, + audioFormat !== "any" && strictFormat, + ) + } + cancelAction={cancelJobAction} + buttonLabel={`Replace auto-captions (${withAudioIds.length})`} + runningLabel="Transcribing…" + label="Replace auto-captions" + extraControls={ + <> + <QueueControl + value={queue} + onChange={setQueue} + defaultQueueKey={transcribeQueueKey} + existingQueues={existingQueues} + actionLabel="Replace auto-captions" + /> + <AudioFormatControl + actionLabel="Replace auto-captions" + value={audioFormat} + onChange={setAudioFormat} + strict={strictFormat} + onStrictChange={setStrictFormat} + /> + </> + } + /> + )} + </div> + </div> + ); +} + function BucketTranscribeSection({ slug, ids, diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -30,6 +30,9 @@ export function normalizeBuckets( skippedByFilter: raw?.skippedByFilter ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], shortAudio: raw?.shortAudio ?? [], + autoSubsOnly: raw?.autoSubsOnly ?? [], + downloadedAutoSubsOnly: raw?.downloadedAutoSubsOnly ?? [], + supersededAutoSubs: raw?.supersededAutoSubs ?? [], needsCookies: raw?.needsCookies ?? [], }; } @@ -71,7 +74,9 @@ const JOB_KIND_TO_STAGE: Record<string, StageId> = { "whisper-retry": "transcribe", "transcribe-one": "transcribe", "whisper-video": "transcribe", + "whisper-bucket-auto-subs": "transcribe", "clean-audio-transcribed": "cleanup", + "purge-superseded-auto-subs": "cleanup", "clean-extra-audio-formats": "cleanup", "remove-wrong-format-audio": "cleanup", "check-availability": "diagnostics", @@ -289,6 +294,20 @@ export function computeStageStatuses( if (failedVideoIds.length > 0) { transcribeParts.push(pluralize(failedVideoIds.length, "failed")); } + // Informational only — the replace-auto-captions lane is opt-in, so these are + // NOT counted as pending work (that would light every YouTube channel up + // amber forever). + const autoSubsCandidates = + buckets.autoSubsOnly.length + buckets.downloadedAutoSubsOnly.length; + if (autoSubsCandidates > 0) { + transcribeParts.push( + pluralize( + autoSubsCandidates, + "video with only auto-captions", + "videos with only auto-captions", + ), + ); + } const transcribe: StageStatus = { id: "transcribe", title: "Transcribe", @@ -309,6 +328,28 @@ export function computeStageStatuses( }; const cleanupRunning = runningByStage.has("cleanup"); + const cleanupParts: string[] = []; + if (cleanupPending > 0) { + cleanupParts.push( + pluralize( + cleanupPending, + "dir has extra audio formats", + "dirs have extra audio formats", + ), + ); + } + // Kept auto-caption backups. Informational (not folded into `pending`): they + // are deliberately retained until purged by hand, so they are inventory, not + // a chore. + if (buckets.supersededAutoSubs.length > 0) { + cleanupParts.push( + pluralize( + buckets.supersededAutoSubs.length, + "superseded auto-caption backup", + "superseded auto-caption backups", + ), + ); + } const cleanup: StageStatus = { id: "cleanup", title: "Cleanup", @@ -318,12 +359,8 @@ export function computeStageStatuses( defaultOpen: true, summary: cleanupRunning ? "Running…" - : cleanupPending > 0 - ? pluralize( - cleanupPending, - "dir has extra audio formats", - "dirs have extra audio formats", - ) + : cleanupParts.length > 0 + ? cleanupParts.join(" · ") : "Nothing to clean.", tone: pickTone({ running: cleanupRunning, diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -75,6 +75,10 @@ async function runPipelineAction( // "always" for this run so every yt-dlp invocation carries the configured // cookies and the defer-mode exclusion is bypassed. forceCookies?: boolean; + // retry-bucket only (the "YouTube auto-captions only" bucket): let an + // ASR-provenance VTT stop counting as an existing destination, so the audio + // download actually runs for videos that already have auto-captions. + replaceAutoSubs?: boolean; // Per-run persistence overrides (Phase 2), forwarded to runYtdlp → // downloadOneManaged for every managed download in this run. keepSourceVideoOverride?: boolean; @@ -184,6 +188,7 @@ async function runPipelineAction( bucketIds: options?.bucketIds, handlingOverride: options?.handlingOverride, forceCookies: options?.forceCookies, + replaceAutoSubs: options?.replaceAutoSubs, keepSourceVideoOverride: options?.keepSourceVideoOverride, extractImmediately: options?.extractImmediately, audioFormatOverride: options?.audioFormatOverride, @@ -322,6 +327,11 @@ export async function retryBucketAction( bucketKey?: ReplayBucket, // Needs-cookies bucket only: run with cookie mode forced to "always". forceCookies?: boolean, + // Auto-captions-only bucket: bypass the destination-exists prefilter for + // videos whose only "destination" is the YouTube ASR VTT we're replacing. + // Always paired with handlingOverride "transcribe" (a youtube-handling run + // would re-fetch subs instead of audio). + replaceAutoSubs?: boolean, ): Promise<StreamActionResult> { if (!Array.isArray(bucketIds) || bucketIds.length === 0) { return { ok: false, error: "No video IDs supplied for retry." }; @@ -341,7 +351,13 @@ export async function retryBucketAction( kind: "retry-bucket", slug, bucket: bucketKey, - params: { queueKey, abortOnError, handlingOverride, forceCookies }, + params: { + queueKey, + abortOnError, + handlingOverride, + forceCookies, + replaceAutoSubs, + }, } : undefined; return runPipelineAction( @@ -354,6 +370,7 @@ export async function retryBucketAction( handlingOverride: handling, abortOnError, forceCookies, + replaceAutoSubs, spec, }, ); diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -11,6 +11,7 @@ import { import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome"; import type { SavedVideoPointer } from "yt-dlp-transcript-common/lib/savedVideo"; import type { AvailabilityHistoryEntry } from "yt-dlp-transcript-common/lib/availability"; +import type { SubtitleProvenance } from "yt-dlp-transcript-common/lib/subtitleProvenance"; import { formatBytes, formatDuration } from "yt-dlp-transcript-common/lib/format"; import { QueueControl } from "../../../../../components/QueueControl"; import { cancelJobAction } from "../../../../../jobs/actions"; @@ -58,6 +59,9 @@ type Props = { // Resolved primary English VTT filename (transcript.en.vtt, or a regional/auto // fallback like transcript.en-US.vtt), or null when no English VTT exists. primaryVtt: string | null; + // filename -> where that subtitle track came from (YouTube ASR vs a manual + // upload). Computed server-side from a 4 KB head read per VTT. + vttProvenance?: Record<string, SubtitleProvenance>; handling: ChannelHandling; defaultQueueKey: string; existingQueues: string[]; @@ -164,12 +168,20 @@ export function VideoPanel({ prevHref, nextHref, position, + vttProvenance = {}, }: Props) { const audioFiles = files.filter((f) => isAudioFile(f.name)); const transcodeSources = files.filter((f) => isTranscodeSource(f.name)); const hasTranscriptJson = files.some((f) => f.name === WHISPER_FILENAME); const hasYtVtt = primaryVtt !== null; const hasTranscript = hasTranscriptJson || hasYtVtt; + // This video's only transcript is YouTube's speech recognition: the transcribe + // controls stay live so it can be replaced with one of our own. A manual — or + // unclassifiable — caption track still blocks them. + const autoSubsOnly = + !hasTranscriptJson && + primaryVtt !== null && + vttProvenance[primaryVtt] === "asr"; const vttTracks = files .filter((f) => isTranscriptVttName(f.name)) .map((f) => f.name) @@ -185,11 +197,13 @@ export function VideoPanel({ ? "No audio on disk — run the pipeline (VTT first, whisper if needed)." : "Audio not yet downloaded." : `${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} on disk.`; - const transcribeSummary = hasTranscript - ? "Transcript present." - : noAudio - ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript." - : "Audio ready, no transcript."; + const transcribeSummary = autoSubsOnly + ? "YouTube auto-captions only — can be replaced with an AI transcript." + : hasTranscript + ? "Transcript present." + : noAudio + ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript." + : "Audio ready, no transcript."; const transcodeSummary = transcodeSources.length > 0 ? `Convert ${transcodeSources.length} source file${transcodeSources.length === 1 ? "" : "s"} to another format.` @@ -326,7 +340,8 @@ export function VideoPanel({ videoId={videoId} file={f} existingQueues={existingQueues} - hasTranscript={hasTranscript} + hasTranscript={hasTranscript && !autoSubsOnly} + replaceAutoSubs={autoSubsOnly} /> ))} </div> @@ -354,6 +369,7 @@ export function VideoPanel({ vttTracks={vttTracks} primaryVtt={primaryVtt} hasWhisper={hasTranscriptJson} + vttProvenance={vttProvenance} /> </PipelineStageCard> )} @@ -673,23 +689,33 @@ function PerFileTranscribeRow({ file, existingQueues, hasTranscript, + replaceAutoSubs = false, }: { slug: string; videoId: string; file: VideoFile; existingQueues: string[]; hasTranscript: boolean; + // The existing transcript is YouTube ASR: transcribing is offered (and + // labelled as a replacement) rather than blocked. + replaceAutoSubs?: boolean; }) { const [transcribeQueue, setTranscribeQueue] = useState(""); return ( <div className="flex flex-col gap-2 rounded border border-border p-3"> <Heading - title={`Transcribe ${file.name}`} + title={ + replaceAutoSubs + ? `Replace auto-captions with an AI transcript (${file.name})` + : `Transcribe ${file.name}` + } desc={ hasTranscript ? "A transcript already exists for this video. Delete transcript.json (or transcript.en.vtt) on disk to re-run." - : "Run whisper-cli on this audio file. Writes transcript.json next to it." + : replaceAutoSubs + ? "This video's only transcript is YouTube's speech recognition. Run whisper-cli on this audio file to write a transcript.json, which takes precedence everywhere. The auto-caption VTT is kept on disk as a backup." + : "Run whisper-cli on this audio file. Writes transcript.json next to it." } /> {hasTranscript ? ( @@ -702,7 +728,11 @@ function PerFileTranscribeRow({ transcribeOneAction(slug, videoId, file.name, transcribeQueue) } cancelAction={cancelJobAction} - buttonLabel={`Transcribe ${file.name}`} + buttonLabel={ + replaceAutoSubs + ? `Replace auto-captions (${file.name})` + : `Transcribe ${file.name}` + } runningLabel="Transcribing…" label={`Transcribe ${file.name}`} extraControls={ @@ -841,18 +871,26 @@ function FilesList({ ); } +const PROVENANCE_LABEL: Record<SubtitleProvenance, string> = { + asr: "YouTube auto-captions", + manual: "manual captions", + unknown: "unknown source", +}; + function TranscriptSourceSection({ slug, videoId, vttTracks, primaryVtt, hasWhisper, + vttProvenance, }: { slug: string; videoId: string; vttTracks: string[]; primaryVtt: string | null; hasWhisper: boolean; + vttProvenance: Record<string, SubtitleProvenance>; }) { const [pending, startTransition] = useTransition(); const [busyFile, setBusyFile] = useState<string | null>(null); @@ -897,6 +935,14 @@ function TranscriptSourceSection({ > <span className="flex items-center gap-2 font-mono text-sm"> {name} + {vttProvenance[name] && ( + <span + aria-label={`transcript provenance ${name}`} + className="rounded bg-muted px-1.5 py-0.5 text-[10px] font-sans font-medium uppercase tracking-wide text-muted-foreground" + > + {PROVENANCE_LABEL[vttProvenance[name]]} + </span> + )} {isPrimary && ( <span aria-label={`primary transcript ${name}`} diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx @@ -11,7 +11,14 @@ import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server"; import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server"; import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; -import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus"; +import { + isTranscriptVtt, + resolvePrimaryVtt, +} from "yt-dlp-transcript-common/lib/videoStatus"; +import { + resolveVttProvenance, + type SubtitleProvenance, +} from "yt-dlp-transcript-common/lib/subtitleProvenance"; import { readTranscriptCoverage } from "yt-dlp-transcript-common/controller/normalizeTranscript"; import { isIncompleteTranscript } from "yt-dlp-transcript-common/lib/transcriptCoverage"; import { @@ -108,6 +115,14 @@ export default async function VideoDetailPage({ const excludedFromTruncatedCheck = await isExcludedFromTruncatedCheck(videoDir); const savedVideo = await loadSavedVideo(videoDir); + // Where each subtitle track came from, so the panel can say "YouTube + // auto-captions" vs "manual captions" — and offer to replace the former with a + // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance). + const vttProvenance: Record<string, SubtitleProvenance> = {}; + for (const f of dirData.files) { + if (!isTranscriptVtt(f.name)) continue; + vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name); + } const cov = await readTranscriptCoverage(videoDir); const coverage = cov ? { @@ -192,6 +207,7 @@ export default async function VideoDetailPage({ videoId={id} files={dirData.files} primaryVtt={resolvePrimaryVtt(dirData.files.map((f) => f.name))} + vttProvenance={vttProvenance} handling={config.handling} defaultQueueKey={defaultQueueKey} existingQueues={existingQueues} diff --git a/editor/app/channels/[slug]/whisperActions.ts b/editor/app/channels/[slug]/whisperActions.ts @@ -25,6 +25,7 @@ import { } from "yt-dlp-transcript-common/controller/failedTranscodings"; import { clearFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { cleanAudioFromTranscribed } from "yt-dlp-transcript-common/controller/cleanAudioFromTranscribed"; +import { purgeSupersededAutoSubs } from "yt-dlp-transcript-common/controller/purgeSupersededAutoSubs"; import { checkKeptDeleted } from "yt-dlp-transcript-common/controller/checkKeptDeleted"; import { cleanExtraAudioFormats } from "yt-dlp-transcript-common/controller/cleanExtraAudioFormats"; import { removeWrongFormatAudio } from "yt-dlp-transcript-common/controller/removeWrongFormatAudio"; @@ -160,6 +161,59 @@ export async function transcribeBucketAction( }); } +// Replace-auto-captions lane, transcribe half. Same batch machinery as +// transcribeBucketAction, with the lane flag set so runWhisperBatch treats only +// transcript.json as "already transcribed" (an English VTT no longer counts) and +// re-verifies each id's provenance before touching it. The superseded VTT is +// left on disk — the Cleanup stage's purge is the only thing that removes it. +export async function transcribeAutoSubsBucketAction( + slug: string, + ids: string[], + queueKey?: string, + audioFormat?: AudioFormat, + strictAudioFormat?: boolean, +): Promise<StreamActionResult> { + const paths = getPaths(); + const fmt = sanitizeAudioFormat(audioFormat); + const cleaned = Array.from(new Set(ids.map((id) => id.trim()).filter(Boolean))); + if (cleaned.length === 0) { + return { ok: false, error: "No video ids supplied" }; + } + return runManagedFunction({ + kind: "whisper-bucket-auto-subs", + queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey), + paths, + channelSlug: slug, + spec: { + kind: "whisper-bucket-auto-subs", + slug, + bucket: "downloadedAutoSubsOnly", + params: { queueKey, audioFormat, strictAudioFormat }, + }, + fn: async (onLog, signal, _setProgress, ctx) => { + // No setProgress here: the channel's transcriptCount already counts these + // videos (their VTT is a transcript), so a "transcripts" progress range + // would be a flat, meaningless bar. + const result = await runWhisperBatch({ + channelSlug: slug, + paths, + audioFormat: fmt, + strictAudioFormat: fmt !== undefined && strictAudioFormat === true, + ids: cleaned, + replaceAutoSubs: true, + onLog, + signal, + drainSignal: ctx.drainSignal, + tracker: makeTaskTracker(ctx, onLog), + }); + onLog( + `Replace auto-captions: ${result.succeeded} succeeded, ${result.failed} failed, ${result.skipped} skipped, ${result.attempted} attempted.`, + ); + revalidatePath(`/channels/${slug}`); + }, + }); +} + export async function clearFailedTranscriptionsAction( slug: string, queueKey?: string, @@ -368,6 +422,39 @@ export async function cleanAudioAction( }); } +// Delete the YouTube auto-caption VTTs that our own transcript superseded (the +// supersededAutoSubs bucket). Manual only — never auto-queued — and the single +// irreversible step in the lane, so it lives next to the Clean-audio sweep and +// honors the same do-not-clean marker. Scoped to English ASR-provenance tracks: +// the controller re-sniffs every file before removing it. +export async function purgeSupersededAutoSubsAction( + slug: string, + queueKey?: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + return runManagedFunction({ + kind: "purge-superseded-auto-subs", + queueKey: resolveQueueKey(channelQueueKey(slug), queueKey), + paths, + channelSlug: slug, + spec: { + kind: "purge-superseded-auto-subs", + slug, + bucket: "supersededAutoSubs", + params: { queueKey }, + }, + fn: async (onLog, signal) => { + await purgeSupersededAutoSubs({ + channelSlug: slug, + paths, + onLog, + signal, + }); + revalidatePath(`/channels/${slug}`); + }, + }); +} + // Re-probe the channel's keep-latest window for source deletion and pin any // gone videos (do-not-clean) so they survive even after rolling out of the // window. Mirrors cleanAudioAction's managed-job shape. Also driven by the sync diff --git a/editor/e2e/auto-subs-replace.spec.ts b/editor/e2e/auto-subs-replace.spec.ts @@ -0,0 +1,425 @@ +// The opt-in "replace YouTube auto-captions" lane, end to end. +// +// A video whose only transcript is YouTube's speech recognition is invisible to +// every transcribe path (isVideoTranscribed counts any English VTT). These tests +// prove the three new snapshot buckets classify such videos correctly — and only +// such videos — and that the manual controls walk one through the whole lane: +// autoSubsOnly → (fetch audio) → downloadedAutoSubsOnly → (transcribe) +// → supersededAutoSubs → (purge) → nothing. +// +// Provenance comes from a 4 KB sniff of the VTT itself (see +// common/lib/subtitleProvenance.ts), so the fixtures below use realistically +// shaped cues: ASR tracks carry `align:start position:N%` plus inline word +// timings, manual tracks carry neither. +// +// Run in default dev mode — E2E_MODE=start serves a stale build. + +import { mkdir, writeFile } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { + pathExists, + readJson, + resetData, + resolvePath, + writeSettings, +} from "./helpers"; +import { baseUrl } from "./baseUrl"; + +const SLUG = "test-auto-subs"; +const ROOT = `test-transcripts/channels/${SLUG}`; +const SNAPSHOT_REL = `${ROOT}/snapshot.json`; + +// Shaped after real yt-dlp --write-auto-subs output. +const ASR_VTT = `WEBVTT +Kind: captions +Language: en + +00:00:00.030 --> 00:00:03.919 align:start position:0% +so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to + +00:00:03.919 --> 00:00:03.929 align:start position:0% +so today we're going to + +00:00:03.929 --> 00:00:07.070 align:start position:0% +so today we're going to +talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing +`; + +// Shaped after a human-uploaded caption track: no cue settings, no word timings. +const MANUAL_VTT = `WEBVTT +Kind: captions +Language: en + +00:00:01.000 --> 00:00:04.000 +So today we're going to talk about the whole thing. + +00:00:04.000 --> 00:00:08.500 +It's a long story, but bear with me. + +00:00:08.500 --> 00:00:12.000 +Here we go. +`; + +const WHISPER_JSON = JSON.stringify({ + transcription: [{ text: "a real transcript" }], +}); + +type SeedVideo = { + id: string; + // "asr" writes an auto-caption-shaped VTT and metadata listing the track under + // automatic_captions; "manual" writes a human-shaped VTT under subtitles. + captions: "asr" | "manual"; + whisper?: boolean; + audio?: boolean; + doNotClean?: boolean; +}; + +function dataRel(videoId: string, file: string): string { + return `${ROOT}/data/${videoId}/${file}`; +} + +async function seedChannel(videos: SeedVideo[]): Promise<void> { + await mkdir(resolvePath(`${ROOT}/data`), { recursive: true }); + // A youtube-handling channel — the case the lane exists for. audioFormat is + // pinned so the forced transcribe-handling download lands on audio.mp3. + await writeFile( + resolvePath(`${ROOT}/config.json`), + JSON.stringify({ + handling: "youtube", + name: "Auto-subs test channel", + url: "https://www.youtube.com/@autosubs/videos", + audioFormat: "mp3", + }), + ); + // retry-bucket resolves each id's source URL from the stored playlist. + await writeFile( + resolvePath(`${ROOT}/playlist`), + videos.map((v) => `https://www.youtube.com/watch?v=${v.id}\n`).join(""), + ); + for (const v of videos) { + const dir = resolvePath(`${ROOT}/data/${v.id}`); + await mkdir(dir, { recursive: true }); + await writeFile( + `${dir}/transcript.en.vtt`, + v.captions === "asr" ? ASR_VTT : MANUAL_VTT, + ); + const track = { en: [{ ext: "vtt", url: "fake://subs" }] }; + await writeFile( + `${dir}/metadata.info.json`, + JSON.stringify({ + id: v.id, + title: `Synthetic ${v.id}`, + upload_date: "20240101", + duration: 60, + extractor_key: "Youtube", + webpage_url: `https://www.youtube.com/watch?v=${v.id}`, + subtitles: v.captions === "manual" ? track : {}, + automatic_captions: v.captions === "asr" ? track : {}, + }), + ); + if (v.whisper) await writeFile(`${dir}/transcript.json`, WHISPER_JSON); + if (v.audio) await writeFile(`${dir}/audio.mp3`, `fake audio ${v.id}\n`); + if (v.doNotClean) { + await writeFile( + `${dir}/do-not-clean.json`, + JSON.stringify({ setAt: new Date(0).toISOString() }), + ); + } + } + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); +} + +type Buckets = { + autoSubsOnly?: string[]; + downloadedAutoSubsOnly?: string[]; + supersededAutoSubs?: string[]; +}; + +// Snapshot regeneration is debounced (~1s) after a page visit / job finish, so +// every assertion on it polls rather than reading once. +async function expectBuckets(expected: Buckets): Promise<void> { + await expect + .poll( + async () => { + const snap = await readJson<{ buckets: Buckets }>(SNAPSHOT_REL).catch( + () => null, + ); + if (!snap) return null; + return { + autoSubsOnly: snap.buckets.autoSubsOnly ?? [], + downloadedAutoSubsOnly: snap.buckets.downloadedAutoSubsOnly ?? [], + supersededAutoSubs: snap.buckets.supersededAutoSubs ?? [], + }; + }, + { timeout: 20_000 }, + ) + .toEqual({ + autoSubsOnly: expected.autoSubsOnly ?? [], + downloadedAutoSubsOnly: expected.downloadedAutoSubsOnly ?? [], + supersededAutoSubs: expected.supersededAutoSubs ?? [], + }); +} + +test("walks an auto-caption video through fetch → transcribe → purge", async ({ + page, +}) => { + test.setTimeout(120_000); + await resetData(); + await seedChannel([ + { id: "asrvid0001", captions: "asr" }, + // Negative control: a human-captioned video must never enter the lane. + { id: "manvid0001", captions: "manual" }, + ]); + + // --- Step 0: classification ----------------------------------------------- + await page.goto(`/channels/${SLUG}`); + await expectBuckets({ autoSubsOnly: ["asrvid0001"] }); + + await expect( + page.getByRole("heading", { name: "YouTube auto-captions only (1)" }), + ).toBeVisible(); + await expect( + page + .getByLabel("auto captions needing audio list") + .getByLabel("auto captions needing audio asrvid0001"), + ).toBeVisible(); + + // --- Step 1: fetch the audio our engine transcribes from ------------------ + await page + .getByLabel("retry auto-captions audio bucket") + .getByRole("button", { name: /^Fetch audio \(1\)$/ }) + .click(); + const downloadLog = page.getByLabel("Retry auto-captions audio output"); + // The VTT must NOT prefilter the video away as "already downloaded". + await expect(downloadLog).toContainText("Prefilter: 1 missing destination", { + timeout: 60_000, + }); + await expect(downloadLog).toContainText("Managed download complete", { + timeout: 60_000, + }); + expect(await pathExists(dataRel("asrvid0001", "audio.mp3"))).toBe(true); + + // --- Step 2: transcribe over the auto-captions ---------------------------- + await page.goto(`/channels/${SLUG}`); + await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0001"] }); + + await page.reload(); + await page + .getByRole("button", { name: /^Replace auto-captions \(1\)$/ }) + .click(); + await expect(page.getByLabel("Replace auto-captions output")).toContainText( + "1 succeeded", + { timeout: 60_000 }, + ); + expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true); + // normalizeTranscript regenerates the derived cues from the new transcript. + expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe( + true, + ); + // The superseded VTT is KEPT as a backup — nothing deletes it automatically. + expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe( + true, + ); + + // --- Step 3: the backup shows up as purgeable inventory ------------------- + await page.goto(`/channels/${SLUG}`); + await expectBuckets({ supersededAutoSubs: ["asrvid0001"] }); + + await page.reload(); + await page.getByRole("button", { name: "Cleanup stage summary" }).click(); + const section = page.getByLabel("superseded auto captions section"); + await expect( + section.getByRole("heading", { name: "Superseded auto-captions (1)" }), + ).toBeVisible(); + + // --- Step 4: purge, and only then does the VTT go ------------------------ + await section + .getByLabel("confirm purge superseded auto captions") + .fill("purge"); + await section + .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ }) + .click(); + await expect( + page.getByLabel("Purge superseded auto-captions output"), + ).toContainText("Purged 1 superseded auto-caption file", { + timeout: 60_000, + }); + + expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe( + false, + ); + expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true); + expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe( + true, + ); + // The human-captioned video was never touched at any step. + expect(await pathExists(dataRel("manvid0001", "transcript.en.vtt"))).toBe( + true, + ); + + await page.goto(`/channels/${SLUG}`); + await expectBuckets({}); +}); + +test("never buckets or purges captions it can't prove are auto-generated", async ({ + page, +}) => { + test.setTimeout(90_000); + await resetData(); + await seedChannel([ + // Superseded ASR backup: the one thing the purge may remove. + { id: "asrdone0001", captions: "asr", whisper: true }, + // Manual captions alongside our transcript: not a backup, never purged. + { id: "mandone0001", captions: "manual", whisper: true }, + // Manual captions and no transcript of ours: not lane work either. + { id: "manonly0001", captions: "manual" }, + // Archived media: shielded from the purge exactly like the Clean-audio sweep. + { id: "asrkeep0001", captions: "asr", whisper: true, doNotClean: true }, + ]); + + await page.goto(`/channels/${SLUG}`); + // Only the unprotected ASR-plus-whisper video is listed as a backup, and no + // manual-caption video appears in the work lane at all. + await expectBuckets({ supersededAutoSubs: ["asrdone0001"] }); + + await page.reload(); + await page.getByRole("button", { name: "Cleanup stage summary" }).click(); + const section = page.getByLabel("superseded auto captions section"); + await section + .getByLabel("confirm purge superseded auto captions") + .fill("purge"); + await section + .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ }) + .click(); + const log = page.getByLabel("Purge superseded auto-captions output"); + await expect(log).toContainText("Purged 1 superseded auto-caption file", { + timeout: 60_000, + }); + await expect(log).toContainText("Skipped asrkeep0001 (marked do not clean)"); + await expect(log).toContainText( + "Kept mandone0001/transcript.en.vtt (manual captions", + ); + + expect(await pathExists(dataRel("asrdone0001", "transcript.en.vtt"))).toBe( + false, + ); + expect(await pathExists(dataRel("mandone0001", "transcript.en.vtt"))).toBe( + true, + ); + expect(await pathExists(dataRel("manonly0001", "transcript.en.vtt"))).toBe( + true, + ); + expect(await pathExists(dataRel("asrkeep0001", "transcript.en.vtt"))).toBe( + true, + ); +}); + +// One enabled local worker using the fake whisper engine (mirrors auto-queue.spec). +const ONE_WORKER = [ + { + id: "w1", + name: "W1", + kind: "local", + enabled: true, + priority: 0, + appId: "whisper-cpp", + config: {}, + }, +]; + +test("the auto-transcribe runner transcribes over auto-captions when opted in", async ({ + page, + request, +}) => { + test.setTimeout(120_000); + await resetData(); + await seedChannel([ + // Already has its audio, so the transcribe runner can take it directly. + { id: "asrvid0003", captions: "asr", audio: true }, + // Manual captions + audio: the runner must never pick this one up. + { id: "manvid0003", captions: "manual", audio: true }, + ]); + await page.goto(`/channels/${SLUG}`); + await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0003"] }); + + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + minFreeDiskGB: 0, + workers: ONE_WORKER, + autoQueue: { + transcription: { + enabled: true, + maxWorkers: 1, + // The switch under test: without it the runner's default union never + // reaches the auto-caption bucket. + replaceAutoSubs: true, + root: { id: "root", mode: "strict", children: [{ id: "leaf-all", match: { type: "all" } }] }, + }, + download: {}, + }, + }); + + const started = await request.post(`${baseUrl}/api/auto-queue/control`, { + data: { kind: "transcription", action: "start" }, + }); + expect(started.ok()).toBeTruthy(); + try { + await expect + .poll( + () => pathExists(dataRel("asrvid0003", "transcript.json")), + { timeout: 60_000 }, + ) + .toBe(true); + // The manual-caption video stays untouched no matter how long the runner idles. + expect(await pathExists(dataRel("manvid0003", "transcript.json"))).toBe( + false, + ); + } finally { + await request.post(`${baseUrl}/api/auto-queue/control`, { + data: { kind: "transcription", action: "stop" }, + }); + } +}); + +test("the auto-queue opt-in is off by default and persists when enabled", async ({ + page, +}) => { + await resetData(); + await seedChannel([{ id: "asrvid0002", captions: "asr", audio: true }]); + + await page.goto("/auto-queue"); + const optIn = page.getByLabel( + "replace YouTube auto-captions for auto-transcription", + ); + await expect(optIn).not.toBeChecked(); + + // The opt-in bucket is also offered per-leaf, so a single channel can join the + // lane without flipping the runner-wide switch. + await page.getByRole("button", { name: "+ Channel rule" }).first().click(); + await expect( + page.getByRole("option", { name: "downloadedAutoSubsOnly" }), + ).toHaveCount(1); + + await optIn.check(); + await page.getByRole("button", { name: "Save policy" }).first().click(); + await expect(page.getByRole("status").first()).toHaveText("Saved."); + + await expect + .poll( + async () => { + const settings = await readJson<{ + autoQueue?: { transcription?: { replaceAutoSubs?: boolean } }; + }>("test-settings.json").catch(() => null); + return settings?.autoQueue?.transcription?.replaceAutoSubs ?? null; + }, + { timeout: 10_000 }, + ) + .toBe(true); + + await page.reload(); + await expect( + page.getByLabel("replace YouTube auto-captions for auto-transcription"), + ).toBeChecked(); +});