commit 13920927ced65f30e5212050628e75339f16a7fc
parent 9226d6badc815ee6e224b663b2af55a884402e6a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 25 Jul 2026 22:41:17 -0400
Replace YouTube auto-captions with our own AI transcripts (opt-in, lowest priority)
Most of the corpus rides on YouTube ASR captions, which are noticeably worse
than what the transcription workers produce -- and they were *sticky*:
isVideoTranscribed() counts any English VTT, so a video with only auto-captions
was permanently invisible to every transcribe bucket and every transcribe job.
This adds an opt-in, strictly-lowest-priority lane that finds those videos,
downloads their audio, and transcribes them properly. pickIndexTranscript()
already prefers transcript.json over any VTT, so whisper takes over on its own
once written.
Detection: a 4 KB head sniff of the VTT itself (common/lib/subtitleProvenance.ts).
YouTube ASR marks ~96-100% of cues with `align:start position:N%` plus inline
<hh:mm:ss.mmm> word timings; manual tracks mark 0%. metadata.info.json (~490 KB)
is parsed only to break an "unknown" tie, so the per-regen cost stays one small
read per English-VTT-having video. Measured on angryjoeshow: 4214 asr / 11
manual / 0 unknown across 4225 VTTs, ~0.35 ms each.
Three snapshot buckets carry the work one restart-safe step at a time:
autoSubsOnly (needs audio) -> downloadedAutoSubsOnly (needs whisper) ->
supersededAutoSubs (done; the old VTT is kept as a backup).
Nothing is automatic by default. The auto-queue gains a per-runner
`replaceAutoSubs` switch that appends the opt-in bucket to the TAIL of the
default union, so real work always drains first; a leaf can also target the
bucket by name for per-channel opt-in. TRANSCRIBE_BUCKETS / DOWNLOAD_BUCKETS are
byte-identical to before -- pinned by a test -- so existing setups are untouched.
Two gates had to be relaxed, both narrowly and both provenance-checked:
- destinationExists() treats an existing VTT as "already downloaded" and would
have prefiltered every candidate away.
- transcribeOneFromQueue / runWhisperBatch skip anything isVideoTranscribed(),
which is true for every ASR-only video -- so the transcribe half would have
silently skipped 100% of candidates.
Both now take a replaceAutoSubs option that narrows "already done" to
transcript.json alone, and each re-verifies provenance per video, so a stale
bucket entry can never schedule a human-authored caption track for replacement.
The original VTT is never deleted automatically. purge-superseded-auto-subs is
manual-only, never auto-queued, scoped to English tracks that still sniff as
asr, and honors do-not-clean -- so a later AI-vs-YouTube comparison stays
possible (supersededAutoSubs is the ready-made worklist).
Verification: 9 new provenance unit tests, 5 new policy tests, extended jobSpec /
jobKinds pins, and editor/e2e/auto-subs-replace.spec.ts -- the full lane (fetch
audio -> transcribe -> superseded -> purge, VTT surviving until the click), the
negative cases (manual captions and do-not-clean never touched), the auto-runner
transcribing over auto-captions, and the opt-in round-trip.
Note: the jobKinds entries, replay handlers, channel-page wiring and CHANGELOG
entry for this feature were absorbed into bb3a217 (social posts) by a concurrent
working-tree commit; they are not repeated here.
Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Diffstat:
27 files changed, 1801 insertions(+), 47 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -1,3 +1,4 @@
+import path from "node:path";
import { readdir } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import { getPaths } from "../lib/paths";
@@ -17,7 +18,8 @@ import {
type WorkPick,
buildPendingByLeaf,
selectNextWork,
- bucketsForKind,
+ defaultBucketsForPolicy,
+ selectableBucketsForKind,
} from "../jobs/autoQueuePolicy";
import {
type AutoQueueKind,
@@ -33,6 +35,8 @@ import {
} from "../jobs/platformBackoff";
import { type DownloadFailureClass } from "../lib/availability";
import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import { isAutoSubsOnly } from "../lib/subtitleProvenance";
+import { readVideoFiles } from "../lib/videoStatus";
import { type DownloadOutcomeStatus } from "../lib/downloadOutcome";
import { downloadQueueKey } from "../lib/queueKeys";
import { readChannelConfig } from "./channels";
@@ -165,7 +169,12 @@ async function buildChannelWork(
const snap = await readChannelSnapshot(paths, slug);
if (!snap) continue;
const buckets: Record<string, string[]> = {};
- for (const name of bucketsForKind(kind)) {
+ // Project every bucket a leaf could be pointed at — including the opt-in
+ // auto-caption ones. A leaf that names a bucket the runner never projected
+ // would silently find no work; projecting them here costs nothing when no
+ // leaf (and no policy switch) asks for them, because buildPendingByLeaf
+ // only walks the buckets its `defaultBuckets` / `match.bucket` name.
+ for (const name of selectableBucketsForKind(kind)) {
// `undownloadedIds` lives at the snapshot top level; every other bucket
// is under snap.buckets.
const ids =
@@ -190,7 +199,11 @@ export async function computeLeafPendingCounts(
const policy = getSettings().autoQueue[kind];
const meta = await listChannelMeta(paths);
const { channels } = await buildChannelWork(paths, kind, meta);
- const pending = buildPendingByLeaf(policy.root, channels, bucketsForKind(kind));
+ const pending = buildPendingByLeaf(
+ policy.root,
+ channels,
+ defaultBucketsForPolicy(kind, policy),
+ );
const counts: Record<string, number> = {};
for (const [leafId, ids] of Object.entries(pending)) counts[leafId] = ids.length;
return counts;
@@ -229,7 +242,6 @@ async function runLoop(
signal: AbortSignal,
ctx: JobRunContext,
): Promise<void> {
- const defaultBuckets = bucketsForKind(kind);
const tracker = makeTaskTracker(ctx, onLog);
const state = await readAutoQueueState(paths);
const kindState = state[kind];
@@ -347,7 +359,13 @@ async function runLoop(
metaCache.map((m) => [m.slug, m.platform ?? "unknown"]),
);
const { channels, owner } = await buildChannelWork(paths, kind, metaCache);
- const pending = buildPendingByLeaf(policy.root, channels, defaultBuckets);
+ // Re-derived each iteration from the freshly-read policy, like `enabled`, so
+ // toggling the replace-auto-captions lane takes effect without a restart.
+ const pending = buildPendingByLeaf(
+ policy.root,
+ channels,
+ defaultBucketsForPolicy(kind, policy),
+ );
const exclude = new Set<string>([...live.inFlight.keys(), ...completed]);
removeIds(pending, exclude);
// Download only: drop videos whose platform already has an in-flight
@@ -543,12 +561,39 @@ type UnitResult = {
failureClass?: DownloadFailureClass;
};
+// Whether this unit is an auto-captions replacement: the video's only
+// transcript is a YouTube ASR VTT. Decided from disk state (one readdir + a 4 KB
+// VTT head read) rather than plumbed from the pick, so it stays correct no
+// matter which leaf/bucket selected the video — including the default union
+// under `replaceAutoSubs`. Both branches below need it: the transcription gate
+// would otherwise skip the video as "already transcribed", and the download
+// branch would fetch subtitles instead of audio.
+async function isAutoSubsUnit(
+ paths: Paths,
+ channelSlug: string,
+ videoId: string,
+): Promise<boolean> {
+ const dir = path.join(paths.channelsDir, channelSlug, "data", videoId);
+ const files = await readVideoFiles(dir, { checkUntranscribable: true });
+ return isAutoSubsOnly(dir, files);
+}
+
// Run a single unit of work. Returns the outcome for counters/backoff. Hard
// cancel (signal) aborts an in-flight unit; drain is handled by the loop (it
// stops launching new units), so the drain signal is intentionally NOT
// forwarded here.
async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
+ const replaceAutoSubs = await isAutoSubsUnit(
+ args.paths,
+ args.channelSlug,
+ args.pick.videoId,
+ );
if (args.kind === "transcription") {
+ if (replaceAutoSubs) {
+ args.onLog(
+ `Auto-transcribe: ${args.pick.videoId} has only YouTube auto-captions — transcribing over them.`,
+ );
+ }
const res = await transcribeOneFromQueue({
paths: args.paths,
channelSlug: args.channelSlug,
@@ -571,6 +616,9 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
// unit is never interrupted — it drains, then the slot goes to the manual
// waiter. Mirrors how auto-download units yield to a manual sync.
background: true,
+ // Relax the "already transcribed" gate for an ASR-only video: only OUR
+ // transcript.json counts, so the auto-captions get transcribed over.
+ replaceAutoSubs,
});
return { outcome: res.outcome };
}
@@ -583,8 +631,23 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
// Marked `background` so a clicked sync preempts queued units (without
// interrupting a running one). The runner's per-platform in-flight gate still
// keeps this to one outstanding unit per platform so the queue isn't flooded.
- const config = await readChannelConfig(args.paths, args.channelSlug);
- if (!config) return { outcome: "skipped" };
+ const rawConfig = await readChannelConfig(args.paths, args.channelSlug);
+ if (!rawConfig) return { outcome: "skipped" };
+ // A `handling: "youtube"` channel's normal download passes --write-auto-subs
+ // --write-subs --skip-download: it would re-fetch the very auto-captions we're
+ // replacing and never touch the audio. Force transcribe-handling for this one
+ // video so it downloads audio instead; the channel's stored config is
+ // untouched. (This is the same override the manual bucket button passes as
+ // handlingOverride.) Nothing to do when the channel already downloads audio.
+ const config =
+ replaceAutoSubs && rawConfig.handling !== "transcribe"
+ ? { ...rawConfig, handling: "transcribe" as const }
+ : rawConfig;
+ if (config !== rawConfig) {
+ args.onLog(
+ `Auto-download: ${args.pick.videoId} has only YouTube auto-captions — downloading audio (handling override: transcribe).`,
+ );
+ }
const url = await findVideoSourceUrl(
args.paths,
args.channelSlug,
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -33,6 +33,7 @@ import { readChannelConfig } from "./channels";
import { computeKeptVideoIds } from "./keptVideos";
import { loadMaybeMissing } from "./quickAvailabilityCheck";
import { readTranscriptCoverage } from "./normalizeTranscript";
+import { resolveVttProvenance } from "../lib/subtitleProvenance";
import { isIncompleteTranscript } from "../lib/transcriptCoverage";
export type AvailabilitySnapshot = {
@@ -109,6 +110,28 @@ export type ChannelSnapshot = {
// user can re-download with a different format (e.g. Original). Optional:
// older snapshots lack it; readers must default to [].
shortAudio: string[];
+ // Videos whose ONLY transcript is YouTube's speech recognition (an English
+ // VTT that sniffs as `asr` — see ../lib/subtitleProvenance) and that have no
+ // audio on disk yet. The opt-in "replace auto-captions" lane's FIRST step:
+ // they need an audio download before our own engine can transcribe them.
+ // Mirrors noTranscript (no audio yet) and, like it, excludes videos that are
+ // excluded from download, untranscribable, or terminally failed. A VTT whose
+ // provenance can't be proven machine-generated is never listed. Optional:
+ // older snapshots lack it; readers must default to [].
+ autoSubsOnly: string[];
+ // Same candidate rule as autoSubsOnly, but audio IS on disk — the lane's
+ // SECOND step, consumed by the opt-in auto-transcribe lane and the channel
+ // page's "YouTube auto-captions only" section. Mirrors
+ // downloadedNoTranscript. Optional: older snapshots lack it; readers must
+ // default to [].
+ downloadedAutoSubsOnly: string[];
+ // Videos that now have OUR transcript (transcript.json) while the superseded
+ // English ASR VTT is still on disk as a backup. Never cleaned automatically —
+ // this is the inventory behind the Cleanup stage's manual purge button (and
+ // the ready-made worklist for a future AI-vs-YouTube comparison). Excludes
+ // do-not-clean–marked dirs so the count matches what the purge would remove.
+ // Optional: older snapshots lack it; readers must default to [].
+ supersededAutoSubs: string[];
// Undownloaded playlist videos whose effective availability says browser
// cookies could recover them (AUTH_RETRY_CLASSES: needs_auth, members_only,
// private). Populated in EVERY cookie mode — members_only/private are
@@ -319,6 +342,14 @@ export async function generateChannelSnapshot(
isVideoTranscribed(files) && !files.isUntranscribable
? await readTranscriptCoverage(dir)
: null;
+ // Where the English VTT came from (YouTube ASR vs a human-authored
+ // track). Only videos that HAVE such a VTT pay the 4 KB head read —
+ // both the auto-subs work lane (no whisper yet) and the superseded
+ // backup inventory (whisper already won) need it. Same conditional
+ // per-video sidecar read pattern as the cues.json coverage read above.
+ const vttProvenance = files.ytVttFile
+ ? await resolveVttProvenance(dir, files.ytVttFile)
+ : null;
return {
id,
files,
@@ -330,6 +361,7 @@ export async function generateChannelSnapshot(
excludedFromTruncatedCheck,
outcome,
coverage,
+ vttProvenance,
};
}),
),
@@ -399,12 +431,22 @@ export async function generateChannelSnapshot(
const skippedByFilter: string[] = [];
const incompleteTranscript: string[] = [];
const shortAudio: string[] = [];
+ const autoSubsOnly: string[] = [];
+ const downloadedAutoSubsOnly: string[] = [];
+ const supersededAutoSubs: string[] = [];
let transcribedWithAudioBytes = 0;
let multipleAudioFormatsBytes = 0;
let foreignAudioBytes = 0;
let transcribed = 0;
let downloaded = 0;
- for (const { id, files, audioSizes, outcome, coverage } of perVideo) {
+ for (const {
+ id,
+ files,
+ audioSizes,
+ outcome,
+ coverage,
+ vttProvenance,
+ } of perVideo) {
if (isVideoTranscribed(files)) transcribed++;
if (isVideoDownloaded(files)) downloaded++;
// A download that completed but stayed malformed after one re-download. The
@@ -513,6 +555,26 @@ export async function generateChannelSnapshot(
) {
nonStandardVtt.push(id);
}
+ // --- Replace-auto-captions lane -----------------------------------------
+ // A transcript that is ONLY YouTube's speech recognition. isVideoTranscribed
+ // counts any English VTT, so without these buckets such a video is invisible
+ // to every transcribe path forever. Conservative by construction: a VTT whose
+ // provenance we can't prove ("unknown") is never a candidate, so a human
+ // caption track is never scheduled for replacement. The corrupt-full-source
+ // and failed-short-audio `continue`s above already excluded terminal videos.
+ const asrVtt = files.hasYtVtt && vttProvenance === "asr";
+ if (asrVtt && files.hasWhisper) {
+ // Our transcript won; the old ASR VTT lingers as a backup. do-not-clean
+ // dirs are excluded so the count matches what the manual purge removes.
+ if (!doNotCleanIds.has(id)) supersededAutoSubs.push(id);
+ } else if (
+ asrVtt &&
+ !files.isUntranscribable &&
+ !excludedById.has(id)
+ ) {
+ if (files.audioFiles.length > 0) downloadedAutoSubsOnly.push(id);
+ else autoSubsOnly.push(id);
+ }
if (files.isUntranscribable) {
untranscribable.push(id);
continue;
@@ -615,6 +677,9 @@ export async function generateChannelSnapshot(
skippedByFilter: skippedByFilter.sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
+ autoSubsOnly: autoSubsOnly.sort(),
+ downloadedAutoSubsOnly: downloadedAutoSubsOnly.sort(),
+ supersededAutoSubs: supersededAutoSubs.sort(),
needsCookies: needsCookies.sort(),
},
undownloadedIds,
diff --git a/common/controller/purgeSupersededAutoSubs.ts b/common/controller/purgeSupersededAutoSubs.ts
@@ -0,0 +1,110 @@
+import path from "node:path";
+import fs from "fs-extra";
+import type { Paths } from "../lib/paths";
+import { isDoNotClean } from "../lib/doNotClean-server";
+import { readVttProvenance } from "../lib/subtitleProvenance";
+import {
+ WHISPER_FILENAME,
+ isEnglishVtt,
+ listTranscriptVtts,
+} from "../lib/videoStatus";
+
+const { pathExists, readdir, remove } = fs;
+
+// Delete the YouTube auto-caption VTTs that our own transcript has superseded —
+// the manual counterpart to the `supersededAutoSubs` snapshot bucket, and the
+// one irreversible step in the replace-auto-captions lane. It is therefore
+// never wired into the auto-queue: the backup only disappears on an explicit
+// click, so an AI-vs-YouTube comparison stays possible until then.
+//
+// Deliberately narrow. A track is removed only when ALL of these hold:
+// - the dir has transcript.json (our transcript already won the index pick),
+// - the track is an ENGLISH one (the en / en-orig / en-US / en-en-* family) —
+// foreign-language tracks are separate content and are left alone,
+// - a fresh 4 KB provenance sniff still says "asr" — a manual or unclassifiable
+// track is never touched,
+// - the dir is not marked do-not-clean (same shield as the Clean-audio sweep).
+//
+// Recovering a purged track means re-running "Download missing subs".
+
+export type PurgeSupersededAutoSubsOptions = {
+ channelSlug: string;
+ paths: Paths;
+ // When set, restrict the sweep to these ids (the bucket the button showed).
+ // Omitted = walk every video dir, like the Clean-audio sweep.
+ ids?: string[];
+ onLog?: (msg: string) => void;
+ signal?: AbortSignal;
+};
+
+export type PurgeSupersededAutoSubsResult = {
+ inspected: number;
+ cleanedDirs: number;
+ removedFiles: number;
+ skipped: number;
+};
+
+export async function purgeSupersededAutoSubs({
+ channelSlug,
+ paths,
+ ids,
+ onLog,
+ signal,
+}: PurgeSupersededAutoSubsOptions): Promise<PurgeSupersededAutoSubsResult> {
+ const log = onLog ?? ((m: string) => console.log(m));
+ const dataDir = path.join(paths.channelsDir, channelSlug, "data");
+ if (!(await pathExists(dataDir))) {
+ log(`No data directory for ${channelSlug}`);
+ return { inspected: 0, cleanedDirs: 0, removedFiles: 0, skipped: 0 };
+ }
+ const onDisk = await readdir(dataDir);
+ const dirs = ids
+ ? ids.filter((id) => onDisk.includes(id))
+ : onDisk;
+
+ let cleanedDirs = 0;
+ let removedFiles = 0;
+ let skipped = 0;
+
+ for (const id of dirs) {
+ if (signal?.aborted) break;
+ const videoDir = path.join(dataDir, id);
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ // No transcript of our own yet — nothing has been superseded, so the VTT is
+ // still this video's only transcript. Never touch it.
+ if (!entries.includes(WHISPER_FILENAME)) continue;
+ const englishVtts = listTranscriptVtts(entries).filter(isEnglishVtt);
+ if (englishVtts.length === 0) continue;
+ if (await isDoNotClean(videoDir)) {
+ log(`Skipped ${id} (marked do not clean)`);
+ skipped++;
+ continue;
+ }
+ let removedHere = 0;
+ for (const name of englishVtts) {
+ const provenance = await readVttProvenance(videoDir, name);
+ if (provenance !== "asr") {
+ log(`Kept ${id}/${name} (${provenance} captions — not auto-generated)`);
+ skipped++;
+ continue;
+ }
+ await remove(path.join(videoDir, name));
+ log(`Removed ${id}/${name}`);
+ removedFiles++;
+ removedHere++;
+ }
+ if (removedHere > 0) cleanedDirs++;
+ }
+
+ const skippedNote = skipped > 0 ? ` Skipped ${skipped}.` : "";
+ log(
+ `Purged ${removedFiles} superseded auto-caption file(s) from ${cleanedDirs} of ${dirs.length} video dir(s).${skippedNote}`,
+ );
+
+ return {
+ inspected: dirs.length,
+ cleanedDirs,
+ removedFiles,
+ skipped,
+ };
+}
diff --git a/common/controller/transcribeOneFromQueue.ts b/common/controller/transcribeOneFromQueue.ts
@@ -52,6 +52,12 @@ export type TranscribeOneOptions = {
// Forwarded to the worker-pool acquire: auto-runner units pass true so they
// yield a free slot to any manual (foreground) transcription waiting on it.
background?: boolean;
+ // Replace-auto-captions lane: the caller has established that this video's
+ // ONLY transcript is a YouTube ASR VTT (isAutoSubsOnly), so the usual
+ // "already transcribed" gate — which counts any English VTT — must not skip
+ // it. With this set, only OUR transcript.json counts as done. The VTT stays
+ // on disk; pickIndexTranscript prefers transcript.json once whisper writes it.
+ replaceAutoSubs?: boolean;
};
const skip = (): TranscribeOneResult => ({ attempted: false, outcome: "skipped" });
@@ -69,6 +75,7 @@ export async function transcribeOneFromQueue({
signal,
drainSignal,
background,
+ replaceAutoSubs = false,
}: TranscribeOneOptions): Promise<TranscribeOneResult> {
const log = onLog ?? ((m: string) => console.log(m));
const channelDir = path.join(paths.channelsDir, channelSlug);
@@ -88,13 +95,22 @@ export async function transcribeOneFromQueue({
// channel with only VTT auto-subs end up attempted -> fail with "no audio file
// found" -> get added to failed-transcriptions on every run.
const videoFiles = await readVideoFiles(videoPath);
- if (isVideoTranscribed(videoFiles)) {
+ const alreadyTranscribed = replaceAutoSubs
+ ? videoFiles.hasWhisper
+ : isVideoTranscribed(videoFiles);
+ if (alreadyTranscribed) {
log(`Transcription for ${videoId} already exists`);
return skip();
}
// No real audio file: a download/source problem, not a transcription
// candidate. Skip (don't fail) so a re-download lets it transcribe later.
- if (!isVideoDownloaded(videoFiles)) {
+ // isVideoDownloaded counts a VTT as an artifact, so the replace-auto-captions
+ // lane must insist on actual audio — otherwise an ASR-only video with no audio
+ // would reach whisper and fail with "no audio file found".
+ const downloaded = replaceAutoSubs
+ ? videoFiles.audioFiles.length > 0
+ : isVideoDownloaded(videoFiles);
+ if (!downloaded) {
log(`Skipping ${videoId}: not downloaded (no audio file)`);
return skip();
}
diff --git a/common/controller/whisperBatch.ts b/common/controller/whisperBatch.ts
@@ -7,6 +7,7 @@ import {
isVideoTranscribed,
readVideoFiles,
} from "../lib/videoStatus";
+import { isAutoSubsOnly } from "../lib/subtitleProvenance";
import { transcribeOneFromQueue } from "./transcribeOneFromQueue";
import { runPool } from "../jobs/concurrentRunner";
import { pruneFailedTranscriptions } from "./failedTranscriptions";
@@ -48,6 +49,13 @@ export type WhisperBatchOptions = {
// When provided, each transcription is tracked as a per-operation task with
// its own parsed progress bar on the Active Jobs screen.
tracker?: TaskTracker;
+ // Replace-auto-captions lane (always paired with an explicit `ids` set from
+ // the downloadedAutoSubsOnly bucket): treat only transcript.json as "already
+ // transcribed", so videos whose sole transcript is a YouTube ASR VTT are
+ // transcribed rather than skipped. Per-video provenance is re-checked here, so
+ // a stale bucket entry (or a hand-passed id) can't clobber a manual caption
+ // track.
+ replaceAutoSubs?: boolean;
};
export type WhisperBatchResult = {
@@ -74,6 +82,7 @@ export async function runWhisperBatch({
signal,
drainSignal,
tracker,
+ replaceAutoSubs = false,
}: WhisperBatchOptions): Promise<WhisperBatchResult> {
const log = onLog ?? ((m: string) => console.log(m));
const channelDir = path.join(paths.channelsDir, channelSlug);
@@ -137,7 +146,21 @@ export async function runWhisperBatch({
for (const id of candidateIds) {
if (failedSet.has(id)) continue;
const vp = path.join(dataDir, id);
- const files = await readVideoFiles(vp);
+ const files = await readVideoFiles(vp, { checkUntranscribable: true });
+ if (replaceAutoSubs) {
+ // Replace-auto-captions lane: only OUR transcript counts as done, and the
+ // VTT never counts as a downloaded artifact (real audio is required).
+ // Re-verify provenance per video so a stale bucket entry can't schedule a
+ // human-authored caption track for replacement.
+ if (files.hasWhisper) continue;
+ if (files.audioFiles.length === 0) continue;
+ if (!(await isAutoSubsOnly(vp, files))) {
+ log(`Skipping ${id}: transcript is not YouTube auto-captions.`);
+ continue;
+ }
+ fullItems.push(id);
+ continue;
+ }
// Already transcribed if whisper ran (transcript.json) OR yt-dlp wrote an
// English VTT (transcript.en.vtt or a regional/auto fallback like en-US).
if (isVideoTranscribed(files)) continue;
@@ -213,6 +236,7 @@ export async function runWhisperBatch({
onLog: log,
signal: runSignal,
drainSignal,
+ replaceAutoSubs,
});
if (res.attempted) attempted++;
if (res.outcome === "transcribed") succeededCount++;
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -7,6 +7,8 @@ import {
buildPendingByLeaf,
bucketsForKind,
defaultAutoQueue,
+ defaultBucketsForPolicy,
+ selectableBucketsForKind,
emptyAutoQueueRuntime,
flattenLeaves,
sanitizeAutoQueue,
@@ -404,3 +406,134 @@ test("sanitizeAutoQueue: duplicate ids are de-duplicated", () => {
];
assert.equal(new Set(ids).size, ids.length, "all ids unique after sanitize");
});
+
+// --- Replace-auto-captions opt-in lane --------------------------------------
+
+test("the default union is unchanged by the opt-in buckets", () => {
+ // Every existing auto-queue setup must keep behaving exactly as before, so the
+ // defaults stay byte-identical and the opt-in buckets live beside them.
+ assert.deepEqual(
+ [...bucketsForKind("transcription")],
+ ["downloadedNoTranscript", "failedListed"],
+ );
+ assert.deepEqual(
+ [...bucketsForKind("download")],
+ ["partialDownloads", "undownloadedIds"],
+ );
+ assert.deepEqual(
+ [...defaultBucketsForPolicy("transcription", { replaceAutoSubs: false })],
+ [...bucketsForKind("transcription")],
+ );
+ assert.deepEqual(
+ [...defaultBucketsForPolicy("download", {})],
+ [...bucketsForKind("download")],
+ );
+});
+
+test("selectableBucketsForKind offers defaults plus the opt-in buckets", () => {
+ assert.deepEqual(
+ [...selectableBucketsForKind("transcription")],
+ ["downloadedNoTranscript", "failedListed", "downloadedAutoSubsOnly"],
+ );
+ assert.deepEqual(
+ [...selectableBucketsForKind("download")],
+ ["partialDownloads", "undownloadedIds", "autoSubsOnly"],
+ );
+});
+
+test("replaceAutoSubs appends the opt-in bucket at the TAIL (lowest priority)", () => {
+ const buckets = defaultBucketsForPolicy("transcription", {
+ replaceAutoSubs: true,
+ });
+ assert.equal(buckets[buckets.length - 1], "downloadedAutoSubsOnly");
+
+ const channels: ChannelWork[] = [
+ {
+ slug: "cornbreadman",
+ platform: "youtube",
+ buckets: {
+ downloadedNoTranscript: ["c1"],
+ failedListed: ["cf1"],
+ downloadedAutoSubsOnly: ["ca1"],
+ },
+ },
+ ];
+ const root: AutoQueueGroup = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "L", match: { type: "all" } }],
+ };
+ // Real work is claimed first; the auto-caption candidate lands last.
+ assert.deepEqual(buildPendingByLeaf(root, channels, buckets).L, [
+ "c1",
+ "cf1",
+ "ca1",
+ ]);
+ // With the lane off, the auto-caption candidate isn't picked up at all.
+ assert.deepEqual(
+ buildPendingByLeaf(
+ root,
+ channels,
+ defaultBucketsForPolicy("transcription", { replaceAutoSubs: false }),
+ ).L,
+ ["c1", "cf1"],
+ );
+});
+
+test("a leaf can target an opt-in bucket without the runner-wide switch", () => {
+ const channels: ChannelWork[] = [
+ {
+ slug: "cornbreadman",
+ platform: "youtube",
+ buckets: {
+ downloadedNoTranscript: ["c1"],
+ downloadedAutoSubsOnly: ["ca1"],
+ },
+ },
+ {
+ slug: "hasanabi",
+ platform: "twitch",
+ buckets: { downloadedNoTranscript: ["h1"], downloadedAutoSubsOnly: ["ha1"] },
+ },
+ ];
+ const root: AutoQueueGroup = {
+ id: "root",
+ mode: "strict",
+ children: [
+ {
+ id: "auto-subs-cornbreadman",
+ match: {
+ type: "channel",
+ value: "cornbreadman",
+ bucket: "downloadedAutoSubsOnly",
+ },
+ },
+ { id: "rest", match: { type: "all" } },
+ ],
+ };
+ const pending = buildPendingByLeaf(
+ root,
+ channels,
+ defaultBucketsForPolicy("transcription", { replaceAutoSubs: false }),
+ );
+ // Per-channel opt-in: only cornbreadman's candidate is claimed, and hasanabi's
+ // never enters the queue.
+ assert.deepEqual(pending["auto-subs-cornbreadman"], ["ca1"]);
+ assert.deepEqual(pending.rest, ["c1", "h1"]);
+});
+
+test("sanitizeAutoQueue defaults replaceAutoSubs to false", () => {
+ assert.equal(defaultAutoQueue().transcription.replaceAutoSubs, false);
+ assert.equal(sanitizeAutoQueue({}).download.replaceAutoSubs, false);
+ // Only an explicit `true` turns the lane on.
+ assert.equal(
+ sanitizeAutoQueue({ transcription: { replaceAutoSubs: "yes" } })
+ .transcription.replaceAutoSubs,
+ false,
+ );
+ assert.equal(
+ sanitizeAutoQueue({ transcription: { replaceAutoSubs: true } }).transcription
+ .replaceAutoSubs,
+ true,
+ );
+});
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -68,6 +68,15 @@ export type AutoQueuePolicy = {
// Overall ceiling on concurrent in-flight workers for this runner. null = no
// runner-level cap (the worker pool / platform queues are the real throttle).
maxWorkers: number | null;
+ // Opt in to the lowest-priority "replace YouTube auto-captions" lane: append
+ // this kind's opt-in buckets (autoSubsOnly / downloadedAutoSubsOnly) to the
+ // tail of the default union, so videos whose only transcript is YouTube ASR
+ // get re-done with our own engine whenever nothing more important is pending.
+ // Default false — the corpus-wide cost is large (an audio download plus a
+ // transcription per video). A leaf can also target the bucket by name for
+ // per-channel opt-in without flipping this switch. Optional: settings written
+ // before this field existed lack it; the sanitizer defaults it to false.
+ replaceAutoSubs?: boolean;
root: AutoQueueGroup;
};
@@ -83,6 +92,18 @@ export type AutoQueueSettings = {
export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const;
export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const;
+// Buckets a runner will NOT draw from unless asked. Replacing YouTube's
+// auto-captions with our own transcript costs an audio download plus a
+// transcription per video, on a corpus where ASR-only videos outnumber
+// manually-captioned ones ~100:1 — so it is never part of the default union.
+// Two ways in, both explicit: a leaf can target the bucket by name (per-channel
+// opt-in, see selectableBucketsForKind), or the runner's `replaceAutoSubs`
+// switch appends it to the TAIL of the default union (see
+// defaultBucketsForPolicy) — strictly lowest priority, since buildPendingByLeaf
+// walks buckets in list order and claiming is first-match-wins.
+export const TRANSCRIBE_OPT_IN_BUCKETS = ["downloadedAutoSubsOnly"] as const;
+export const DOWNLOAD_OPT_IN_BUCKETS = ["autoSubsOnly"] as const;
+
// Local kind type — do NOT import AutoQueueKind from autoQueueState.ts, which
// already imports from this module (the reverse edge would be a cycle).
export function bucketsForKind(
@@ -91,6 +112,35 @@ export function bucketsForKind(
return kind === "transcription" ? TRANSCRIBE_BUCKETS : DOWNLOAD_BUCKETS;
}
+export function optInBucketsForKind(
+ kind: "transcription" | "download",
+): readonly string[] {
+ return kind === "transcription"
+ ? TRANSCRIBE_OPT_IN_BUCKETS
+ : DOWNLOAD_OPT_IN_BUCKETS;
+}
+
+// Everything a leaf may be pointed at for this kind: the default union plus the
+// opt-in buckets. Backs the editor's bucket dropdown AND the runner's snapshot
+// projection — a leaf can only find ids in a bucket the runner projected.
+export function selectableBucketsForKind(
+ kind: "transcription" | "download",
+): readonly string[] {
+ return [...bucketsForKind(kind), ...optInBucketsForKind(kind)];
+}
+
+// The bucket list a leaf with NO explicit bucket draws from. Identical to
+// bucketsForKind unless the policy opted into auto-caption replacement, which
+// appends the opt-in buckets at the tail so real work always drains first.
+export function defaultBucketsForPolicy(
+ kind: "transcription" | "download",
+ policy: Pick<AutoQueuePolicy, "replaceAutoSubs">,
+): readonly string[] {
+ return policy.replaceAutoSubs
+ ? [...bucketsForKind(kind), ...optInBucketsForKind(kind)]
+ : bucketsForKind(kind);
+}
+
export const AUTO_QUEUE_MAX_WORKERS_MAX = 64;
// --- Runtime fairness state (persisted best-effort by autoQueueState.ts) ----
@@ -348,6 +398,7 @@ export function defaultAutoQueuePolicy(): AutoQueuePolicy {
return {
enabled: false,
maxWorkers: null,
+ replaceAutoSubs: false,
root: { id: "root", mode: "strict", weight: 1, maxWorkers: null, children: [] },
};
}
@@ -365,6 +416,9 @@ function sanitizePolicy(value: unknown): AutoQueuePolicy {
return {
enabled: r.enabled === true,
maxWorkers: clampMaxWorkers(r.maxWorkers),
+ // Opt-in only: anything but an explicit `true` (including a missing field on
+ // a pre-existing settings.json) leaves the lane off.
+ replaceAutoSubs: r.replaceAutoSubs === true,
root: sanitizeRoot(r.root, seen),
};
}
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -44,6 +44,30 @@ const OLD_LABELS: Record<string, string> = {
sync: "Sync",
};
+// Kinds added after the Phase 1 snapshot above. Pinned here so their label and
+// drainability are asserted too, without pretending they were part of the
+// original consolidation.
+const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = {
+ // Replace-auto-captions lane: a whisper batch (drains like every other batch)
+ // plus the manual VTT purge (a fast file sweep — nothing to drain).
+ "whisper-bucket-auto-subs": {
+ label: "Replace auto-captions",
+ drainable: true,
+ },
+ "purge-superseded-auto-subs": {
+ label: "Purge superseded auto-captions",
+ drainable: false,
+ },
+};
+
+test("added kinds carry their pinned label and drainability", () => {
+ for (const [kind, meta] of Object.entries(ADDED_KINDS)) {
+ assert.equal(jobKindLabel(kind), meta.label, `${kind} label`);
+ assert.equal(isDrainableKind(kind), meta.drainable, `${kind} drainability`);
+ assert.notEqual(getJobKind(kind), undefined, `${kind} should be registered`);
+ }
+});
+
test("drainable kinds match the old DRAINABLE_KINDS set exactly", () => {
const drainSet = new Set(OLD_DRAINABLE);
for (const kind of OLD_DRAINABLE) {
diff --git a/common/jobs/jobSpec.test.ts b/common/jobs/jobSpec.test.ts
@@ -15,7 +15,7 @@ test("parses a minimal spec and rejects malformed input", () => {
assert.equal(parseJobSpec({ kind: "sync" }), null);
});
-test("accepts every replay bucket, including needsCookies", () => {
+test("accepts every replay bucket, including the auto-caption lane", () => {
for (const bucket of [
"partialDownloads",
"noTranscript",
@@ -23,6 +23,9 @@ test("accepts every replay bucket, including needsCookies", () => {
"incompleteTranscript",
"shortAudio",
"needsCookies",
+ "autoSubsOnly",
+ "downloadedAutoSubsOnly",
+ "supersededAutoSubs",
]) {
const spec = parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket });
assert.equal(spec?.bucket, bucket);
diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts
@@ -21,7 +21,12 @@ export type ReplayBucket =
| "downloadedNoTranscript"
| "incompleteTranscript"
| "shortAudio"
- | "needsCookies";
+ | "needsCookies"
+ // Replace-auto-captions lane: ASR-only without audio (fetch audio), ASR-only
+ // with audio (transcribe), and the kept-VTT backups (manual purge).
+ | "autoSubsOnly"
+ | "downloadedAutoSubsOnly"
+ | "supersededAutoSubs";
export type JobSpec = {
kind: string;
@@ -42,6 +47,9 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([
"incompleteTranscript",
"shortAudio",
"needsCookies",
+ "autoSubsOnly",
+ "downloadedAutoSubsOnly",
+ "supersededAutoSubs",
]);
// Defensive parse for a spec read back from JSON (a sidecar or the bookmarks
diff --git a/common/lib/subtitleProvenance.test.ts b/common/lib/subtitleProvenance.test.ts
@@ -0,0 +1,133 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ provenanceFromMetadata,
+ readVttProvenance,
+ resolveVttProvenance,
+ sniffVttProvenance,
+ vttTrackName,
+} from "./subtitleProvenance";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/subtitleProvenance.test.ts
+
+// Shaped after real yt-dlp --write-auto-subs output: every cue carries
+// `align:start position:0%` and inline word timings.
+const ASR_HEAD = `WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.030 --> 00:00:03.919 align:start position:0%
+
+so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to
+
+00:00:03.919 --> 00:00:03.929 align:start position:0%
+so today we're going to
+
+00:00:03.929 --> 00:00:07.070 align:start position:0%
+so today we're going to
+talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing
+`;
+
+// Shaped after a human-uploaded/manual track: plain cues, no cue settings, no
+// inline word timings.
+const MANUAL_HEAD = `WEBVTT
+Kind: captions
+Language: en
+
+00:00:01.000 --> 00:00:04.000
+So today we're going to talk about the whole thing.
+
+00:00:04.000 --> 00:00:08.500
+It's a long story, but bear with me.
+
+00:00:08.500 --> 00:00:12.000
+Here we go.
+`;
+
+test("sniffVttProvenance classifies YouTube ASR output as asr", () => {
+ assert.equal(sniffVttProvenance(ASR_HEAD), "asr");
+});
+
+test("sniffVttProvenance classifies a manual caption track as manual", () => {
+ assert.equal(sniffVttProvenance(MANUAL_HEAD), "manual");
+});
+
+test("sniffVttProvenance detects asr from word timings alone", () => {
+ // Some ASR tracks are served without the align/position cue settings but keep
+ // the inline word timings — still machine-generated.
+ const head = ASR_HEAD.replace(/ align:start position:\d+%/g, "");
+ assert.equal(sniffVttProvenance(head), "asr");
+});
+
+test("sniffVttProvenance returns unknown for too little evidence", () => {
+ assert.equal(sniffVttProvenance(""), "unknown");
+ assert.equal(sniffVttProvenance("not a vtt at all"), "unknown");
+ assert.equal(
+ sniffVttProvenance("WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nOnly one cue.\n"),
+ "unknown",
+ );
+});
+
+test("sniffVttProvenance returns unknown when markers are present but sparse", () => {
+ const head = `WEBVTT
+
+00:00:00.000 --> 00:00:02.000 align:start position:0%
+one
+
+00:00:02.000 --> 00:00:04.000
+two
+
+00:00:04.000 --> 00:00:06.000
+three
+
+00:00:06.000 --> 00:00:08.000
+four
+`;
+ assert.equal(sniffVttProvenance(head), "unknown");
+});
+
+test("provenanceFromMetadata prefers subtitles over automatic_captions", () => {
+ const meta = {
+ subtitles: { en: [{ ext: "vtt" }] },
+ automatic_captions: { en: [{ ext: "vtt" }], "en-orig": [{ ext: "vtt" }] },
+ };
+ assert.equal(provenanceFromMetadata(meta, "en"), "manual");
+ assert.equal(provenanceFromMetadata(meta, "en-orig"), "asr");
+ assert.equal(provenanceFromMetadata(meta, "fr"), "unknown");
+ assert.equal(provenanceFromMetadata(null, "en"), "unknown");
+ assert.equal(provenanceFromMetadata({}, "en"), "unknown");
+});
+
+test("vttTrackName extracts the language segment", () => {
+ assert.equal(vttTrackName("transcript.en.vtt"), "en");
+ assert.equal(vttTrackName("transcript.en-US.vtt"), "en-US");
+ assert.equal(vttTrackName("transcript.json"), null);
+ assert.equal(vttTrackName("transcript.cues.json"), null);
+});
+
+test("readVttProvenance sniffs only the head of a file on disk", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "subprov-"));
+ await writeFile(path.join(dir, "transcript.en.vtt"), ASR_HEAD);
+ await writeFile(path.join(dir, "transcript.en-US.vtt"), MANUAL_HEAD);
+ assert.equal(await readVttProvenance(dir, "transcript.en.vtt"), "asr");
+ assert.equal(await readVttProvenance(dir, "transcript.en-US.vtt"), "manual");
+ assert.equal(await readVttProvenance(dir, "missing.vtt"), "unknown");
+});
+
+test("resolveVttProvenance falls back to metadata for an undecidable file", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "subprov-"));
+ // One cue only: the sniff can't judge it.
+ await writeFile(
+ path.join(dir, "transcript.en.vtt"),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:02.000\nhi\n",
+ );
+ assert.equal(await resolveVttProvenance(dir, "transcript.en.vtt"), "unknown");
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ automatic_captions: { en: [{ ext: "vtt" }] } }),
+ );
+ assert.equal(await resolveVttProvenance(dir, "transcript.en.vtt"), "asr");
+});
diff --git a/common/lib/subtitleProvenance.ts b/common/lib/subtitleProvenance.ts
@@ -0,0 +1,156 @@
+import path from "node:path";
+import { createReadStream } from "node:fs";
+import { readFile } from "node:fs/promises";
+import type { VideoFiles } from "./videoStatus";
+
+// Where a VTT transcript came from: YouTube's speech recognition ("asr", what
+// yt-dlp downloads under --write-auto-subs) or a human-authored/uploaded track
+// ("manual", --write-subs). "unknown" means we could not tell, and is treated as
+// manual everywhere it matters — we never replace a transcript we can't prove is
+// machine-generated.
+//
+// The authoritative answer lives in metadata.info.json (`subtitles` vs
+// `automatic_captions`), but those files average ~490 KB, so parsing one per
+// video per snapshot regen is not viable across a 77k-video corpus. Instead we
+// sniff the first few KB of the VTT itself: YouTube's ASR cues carry
+// `align:start position:N%` cue settings and inline `<hh:mm:ss.mmm>` word
+// timings, and manual tracks carry neither. Measured on real data across 5
+// channels (12 manual + 12 ASR): ASR files marked 96–100% of cues, manual files
+// 0%. The metadata parse is kept as the tie-breaker for the rare "unknown".
+
+export type SubtitleProvenance = "asr" | "manual" | "unknown";
+
+// One read of this many bytes per English-VTT-having video per snapshot regen.
+// Big enough to hold several cues even for long, densely-marked ASR output.
+export const PROVENANCE_SNIFF_BYTES = 4096;
+
+// Below this many cues in the sniffed window there isn't enough evidence to
+// call it either way (a 2-cue stub could be anything).
+const MIN_CUES = 2;
+
+// Fraction of cues that must carry ASR fingerprints for an "asr" verdict.
+const ASR_RATIO = 0.5;
+
+const CUE_ARROW_RE = /-->/g;
+// yt-dlp writes YouTube's ASR cue settings verbatim: "align:start position:0%".
+const ASR_ALIGN_RE = /align:start position:\d+%/g;
+// Inline per-word timing tags, e.g. "<00:00:03.919>" — ASR-only in practice.
+const WORD_TIMING_RE = /<\d{2}:\d{2}:\d{2}\.\d{3}>/g;
+
+function count(re: RegExp, text: string): number {
+ // Each call needs its own lastIndex reset — the /g regexes are module-level.
+ re.lastIndex = 0;
+ let n = 0;
+ while (re.exec(text) !== null) n++;
+ return n;
+}
+
+// Pure classifier over the first PROVENANCE_SNIFF_BYTES of a VTT file. Kept
+// separate from the I/O so it is directly unit-testable.
+export function sniffVttProvenance(head: string): SubtitleProvenance {
+ const cues = count(CUE_ARROW_RE, head);
+ if (cues < MIN_CUES) return "unknown";
+ const aligned = count(ASR_ALIGN_RE, head);
+ const worded = count(WORD_TIMING_RE, head);
+ if (aligned / cues >= ASR_RATIO || worded / cues >= ASR_RATIO) return "asr";
+ // No ASR fingerprint at all on a file with real cues: a human-authored track.
+ if (aligned === 0 && worded === 0) return "manual";
+ // Some markers but not enough to clear the bar — refuse to guess.
+ return "unknown";
+}
+
+// Read only the head of the VTT (createReadStream start/end), never the whole
+// file: a 3-hour ASR transcript is multiple MB.
+export async function readVttProvenance(
+ videoDir: string,
+ vttFile: string,
+): Promise<SubtitleProvenance> {
+ let head: string;
+ try {
+ head = await readHead(path.join(videoDir, vttFile), PROVENANCE_SNIFF_BYTES);
+ } catch {
+ return "unknown";
+ }
+ return sniffVttProvenance(head);
+}
+
+function readHead(file: string, bytes: number): Promise<string> {
+ return new Promise((resolve, reject) => {
+ const stream = createReadStream(file, {
+ encoding: "utf8",
+ start: 0,
+ end: bytes - 1,
+ });
+ let out = "";
+ stream.on("data", (chunk) => {
+ out += chunk;
+ });
+ stream.on("error", reject);
+ stream.on("end", () => resolve(out));
+ });
+}
+
+// The language/track segment of a transcript sidecar filename:
+// "transcript.en-US.vtt" -> "en-US". Null for anything else.
+export function vttTrackName(filename: string): string | null {
+ const m = filename.match(/^transcript\.([^.]+)\.vtt$/);
+ return m ? m[1] : null;
+}
+
+type CaptionMetadata = {
+ subtitles?: Record<string, unknown>;
+ automatic_captions?: Record<string, unknown>;
+};
+
+// The authoritative check, from an already-parsed metadata.info.json. Mirrors
+// the shape read by runYtdlp/downloadOneManaged: `subtitles` is what YouTube
+// calls manual/uploaded captions, `automatic_captions` is ASR. A track listed in
+// both counts as manual (the manual file is what yt-dlp would have written).
+export function provenanceFromMetadata(
+ meta: unknown,
+ track: string,
+): SubtitleProvenance {
+ if (!meta || typeof meta !== "object") return "unknown";
+ const m = meta as CaptionMetadata;
+ const subs = m.subtitles;
+ const auto = m.automatic_captions;
+ const has = (rec: Record<string, unknown> | undefined): boolean =>
+ !!rec && typeof rec === "object" && Object.prototype.hasOwnProperty.call(rec, track);
+ if (has(subs)) return "manual";
+ if (has(auto)) return "asr";
+ return "unknown";
+}
+
+// Sniff first; fall back to the (expensive) metadata parse only when the sniff
+// can't decide. That keeps the 490 KB read rare while still classifying the odd
+// stub file correctly.
+export async function resolveVttProvenance(
+ videoDir: string,
+ vttFile: string,
+): Promise<SubtitleProvenance> {
+ const sniffed = await readVttProvenance(videoDir, vttFile);
+ if (sniffed !== "unknown") return sniffed;
+ const track = vttTrackName(vttFile);
+ if (!track) return "unknown";
+ try {
+ const raw = await readFile(path.join(videoDir, "metadata.info.json"), "utf8");
+ return provenanceFromMetadata(JSON.parse(raw), track);
+ } catch {
+ return "unknown";
+ }
+}
+
+// True when this video's ONLY transcript is YouTube ASR — the work-lane
+// candidate rule, shared by the snapshot buckets, the whisper gate
+// (transcribeOneFromQueue) and the auto-runner's download override so all four
+// agree on what "auto-captions only" means. `files` is the already-read dir
+// listing; only the 4 KB VTT sniff is done here.
+export async function isAutoSubsOnly(
+ videoDir: string,
+ files: VideoFiles,
+): Promise<boolean> {
+ if (!files.ytVttFile || files.hasWhisper || files.isUntranscribable) {
+ return false;
+ }
+ return (await resolveVttProvenance(videoDir, files.ytVttFile)) === "asr";
+}
diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts
@@ -156,6 +156,14 @@ function englishVttRank(track: string): number {
if (/^en-en(?:-|$)/.test(track)) return 3; // auto-translated en→en variants
return 2; // regional/manual en-US, en-GB, …
}
+// Whether a transcript sidecar filename is one of the ENGLISH VTT tracks
+// resolvePrimaryVtt considers (en, en-orig, en-US, en-en-*, …). Foreign-language
+// tracks — including translations like es-en-US — return false. Exported for the
+// superseded-auto-caption purge, which must leave non-English tracks alone.
+export function isEnglishVtt(name: string): boolean {
+ return EN_VTT_RE.test(name);
+}
+
export function resolvePrimaryVtt(entries: string[]): string | null {
let best: { name: string; rank: number } | null = null;
for (const e of entries) {
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -14,6 +14,7 @@ import { checkDiskSpace } from "../lib/diskSpace";
import { formatBytes } from "../lib/format";
import { detectPlatform } from "../lib/platform";
import { isRealAudioFile } from "../lib/videoStatus";
+import { readVttProvenance } from "../lib/subtitleProvenance";
import type { Paths } from "../lib/paths";
import {
EXCLUDED_FROM_DOWNLOAD,
@@ -111,6 +112,15 @@ export type RunYtdlpOpts = {
// mutating the channel config on disk. Useful for retrying old "youtube"
// videos as "transcribe".
handlingOverride?: ChannelHandling;
+ // retry-bucket only (the "YouTube auto-captions only" bucket): these videos
+ // ALREADY have an English VTT, which destinationExists() normally reads as
+ // "already downloaded" — the run would prefilter every id away. With this set,
+ // an English VTT that still sniffs as YouTube ASR no longer counts as a
+ // destination, so the audio download proceeds. A manual (or unclassifiable)
+ // caption track still suppresses the download, so a human transcript is never
+ // the reason we re-fetch. Paired with handlingOverride: "transcribe" by the
+ // caller, since a youtube-handling download would fetch subs, not audio.
+ replaceAutoSubs?: boolean;
// Called when a per-video download fails with a rate-limit (HTTP 429) or
// network error, so the caller can record the SHARED per-platform cooldown
// (see common/jobs/downloadBackoff.ts). This entangles manual sync/download
@@ -531,7 +541,11 @@ async function downloadPlaylistManaged(
tofetch.push(url);
continue;
}
- if (await destinationExists(dataDir, dirId, effectiveHandling)) {
+ if (
+ await destinationExists(dataDir, dirId, effectiveHandling, {
+ replaceAutoSubs: opts.replaceAutoSubs,
+ })
+ ) {
alreadyComplete++;
continue;
}
@@ -1107,15 +1121,21 @@ export async function destinationExists(
dataDir: string,
id: string,
handling: ChannelHandling,
+ opts: { replaceAutoSubs?: boolean } = {},
): Promise<boolean> {
const dir = path.join(dataDir, id);
const entries = await readdir(dir).catch(() => [] as string[]);
- // Either transcript counts as "already done" — channels can be hybrid.
- if (
- entries.includes("transcript.en.vtt") ||
- entries.includes("transcript.json")
- ) {
- return true;
+ // Our own transcript always counts as "already done".
+ if (entries.includes("transcript.json")) return true;
+ // A VTT normally counts too — channels can be hybrid. The replace-auto-captions
+ // run is the one exception, and only for a track that still sniffs as YouTube
+ // ASR: that is precisely the transcript we are here to replace, so it must not
+ // suppress the audio download the replacement needs.
+ if (entries.includes("transcript.en.vtt")) {
+ const replaceable =
+ opts.replaceAutoSubs === true &&
+ (await readVttProvenance(dir, "transcript.en.vtt")) === "asr";
+ if (!replaceable) return true;
}
// Transcribe channels treat raw audio as a download in progress so we don't
// re-fetch it before whisper runs. YouTube channels expect a .vtt; an audio
diff --git a/editor/app/auto-queue/actions.ts b/editor/app/auto-queue/actions.ts
@@ -26,7 +26,13 @@ export type SaveResult = { ok: true } | { ok: false; error: string };
// its next iteration (getSettings reads from disk), so it stops on its own.
export async function saveAutoQueueAction(
kind: AutoQueueKind,
- input: { enabled: boolean; maxWorkers: number | null; root: AutoQueueGroup },
+ input: {
+ enabled: boolean;
+ maxWorkers: number | null;
+ // Opt in to the lowest-priority replace-auto-captions lane (default false).
+ replaceAutoSubs: boolean;
+ root: AutoQueueGroup;
+ },
): Promise<SaveResult> {
const current = getSettings();
const next: SiteSettings = {
@@ -36,6 +42,7 @@ export async function saveAutoQueueAction(
[kind]: {
enabled: input.enabled,
maxWorkers: input.maxWorkers,
+ replaceAutoSubs: input.replaceAutoSubs === true,
root: input.root,
},
},
diff --git a/editor/app/auto-queue/components/AutoQueueView.tsx b/editor/app/auto-queue/components/AutoQueueView.tsx
@@ -183,6 +183,7 @@ function KindPanel({
kind={kind}
initialEnabled={status.policy.enabled}
initialMaxWorkers={status.policy.maxWorkers}
+ initialReplaceAutoSubs={status.policy.replaceAutoSubs === true}
initialRoot={status.policy.root}
channels={channels}
platforms={platforms}
diff --git a/editor/app/auto-queue/components/PolicyTreeEditor.tsx b/editor/app/auto-queue/components/PolicyTreeEditor.tsx
@@ -127,6 +127,7 @@ export function PolicyTreeEditor({
kind,
initialEnabled,
initialMaxWorkers,
+ initialReplaceAutoSubs,
initialRoot,
channels,
platforms,
@@ -135,6 +136,7 @@ export function PolicyTreeEditor({
kind: AutoQueueKind;
initialEnabled: boolean;
initialMaxWorkers: number | null;
+ initialReplaceAutoSubs: boolean;
initialRoot: AutoQueueGroup;
channels: { slug: string; name: string | null }[];
platforms: string[];
@@ -143,6 +145,9 @@ export function PolicyTreeEditor({
const [root, setRoot] = useState<AutoQueueGroup>(initialRoot);
const [enabled, setEnabled] = useState(initialEnabled);
const [maxWorkers, setMaxWorkers] = useState<number | null>(initialMaxWorkers);
+ const [replaceAutoSubs, setReplaceAutoSubs] = useState(
+ initialReplaceAutoSubs,
+ );
const [saving, setSaving] = useState(false);
const [result, setResult] = useState<SaveResult | null>(null);
@@ -160,7 +165,14 @@ export function PolicyTreeEditor({
setSaving(true);
setResult(null);
try {
- setResult(await saveAutoQueueAction(kind, { enabled, maxWorkers, root }));
+ setResult(
+ await saveAutoQueueAction(kind, {
+ enabled,
+ maxWorkers,
+ replaceAutoSubs,
+ root,
+ }),
+ );
} catch (e) {
setResult({ ok: false, error: (e as Error).message });
} finally {
@@ -198,6 +210,31 @@ export function PolicyTreeEditor({
</label>
</div>
+ {/* Lowest-priority lane, off by default. Appending the opt-in bucket to
+ the TAIL of the default union is what makes it lowest priority: pending
+ work is claimed bucket-by-bucket in list order. */}
+ <label className="flex items-start gap-2 text-sm">
+ <input
+ type="checkbox"
+ checked={replaceAutoSubs}
+ onChange={(e) => setReplaceAutoSubs(e.target.checked)}
+ aria-label={`replace YouTube auto-captions for auto-${kind}`}
+ className="mt-1"
+ />
+ <span>
+ Replace YouTube auto-captions
+ <span className="block text-xs text-muted-foreground">
+ When nothing else is pending,{" "}
+ {kind === "transcription"
+ ? "transcribe videos whose only transcript is YouTube's speech recognition (and whose audio is already downloaded)"
+ : "download audio for videos whose only transcript is YouTube's speech recognition"}
+ . Strictly lowest priority, and off by default — every candidate
+ costs a download plus a transcription. Rules below can also target
+ the bucket directly for per-channel opt-in.
+ </span>
+ </span>
+ </label>
+
<NodeEditor
node={root}
depth={0}
diff --git a/editor/app/auto-queue/page.tsx b/editor/app/auto-queue/page.tsx
@@ -3,7 +3,7 @@ import Link from "next/link";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import { listChannels } from "yt-dlp-transcript-common/controller/channels";
import { PLATFORM_VALUES } from "yt-dlp-transcript-common/lib/platform";
-import { bucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy";
+import { selectableBucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueuePolicy";
import { buildAutoQueueStatusPayload } from "./status";
import { AutoQueueView } from "./components/AutoQueueView";
@@ -12,10 +12,12 @@ export const dynamic = "force-dynamic";
export const metadata: Metadata = { title: "Auto-queue" };
// Buckets each runner can draw from — derived from the policy engine's single
-// source of truth (bucketsForKind) so the picker can't drift from the runner.
+// source of truth (selectableBucketsForKind) so the picker can't drift from the
+// runner. Includes the opt-in auto-caption buckets: a leaf that names one gets
+// per-channel opt-in without flipping the runner-wide switch.
const BUCKETS_BY_KIND = {
- transcription: [...bucketsForKind("transcription")],
- download: [...bucketsForKind("download")],
+ transcription: [...selectableBucketsForKind("transcription")],
+ download: [...selectableBucketsForKind("download")],
};
export default async function AutoQueuePage() {
diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx
@@ -20,6 +20,13 @@ type Props = {
// Needs-cookies bucket: force cookie mode "always" for the run and label the
// button "Download with cookies" so the manual cookie path is explicit.
forceCookies?: boolean;
+ // Auto-captions-only bucket: these videos have a VTT but no audio, so the run
+ // needs BOTH the transcribe handling (a youtube-handling download would fetch
+ // subs, not audio) and the destination-exists bypass. The handling override is
+ // preselected rather than forced, so it stays visible and adjustable.
+ replaceAutoSubs?: boolean;
+ // Overrides the default "Retry (n)" button text.
+ buttonLabel?: string;
};
export function RetryBucketControl({
@@ -30,9 +37,13 @@ export function RetryBucketControl({
existingQueues,
bucketKey,
forceCookies,
+ replaceAutoSubs,
+ buttonLabel,
}: Props) {
const [queue, setQueue] = useState(defaultQueueKey);
- const [handlingOverride, setHandlingOverride] = useState("");
+ const [handlingOverride, setHandlingOverride] = useState(
+ replaceAutoSubs ? "transcribe" : "",
+ );
const [abortOnError, setAbortOnError] = useState(false);
if (ids.length === 0) return null;
@@ -52,13 +63,16 @@ export function RetryBucketControl({
handlingOverride || undefined,
bucketKey,
forceCookies,
+ replaceAutoSubs,
)
}
cancelAction={cancelJobAction}
buttonLabel={
- forceCookies
- ? `Download with cookies (${ids.length})`
- : `Retry (${ids.length})`
+ buttonLabel
+ ? `${buttonLabel} (${ids.length})`
+ : forceCookies
+ ? `Download with cookies (${ids.length})`
+ : `Retry (${ids.length})`
}
runningLabel="Retrying…"
label={`Retry ${actionLabel}`}
diff --git a/editor/app/channels/[slug]/components/stages/CleanupStage.tsx b/editor/app/channels/[slug]/components/stages/CleanupStage.tsx
@@ -9,6 +9,7 @@ import {
checkKeptDeletedAction,
cleanAudioAction,
cleanExtraAudioFormatsAction,
+ purgeSupersededAutoSubsAction,
removeWrongFormatAudioAction,
} from "../../whisperActions";
import { persistKeptAction } from "../../persistActions";
@@ -20,6 +21,9 @@ type Props = {
existingQueues: string[];
multipleAudioFormatIds: string[];
foreignAudioIds: string[];
+ // Videos where our transcript won and the superseded YouTube ASR VTT is still
+ // on disk as a backup. Never cleaned automatically — only by the purge below.
+ supersededAutoSubsIds: string[];
transcodeApplies: boolean;
// Estimated bytes each cleanup would reclaim, as of the last report.
transcribedAudioBytes: number;
@@ -41,6 +45,7 @@ export function CleanupStage({
existingQueues,
multipleAudioFormatIds,
foreignAudioIds,
+ supersededAutoSubsIds,
transcodeApplies,
transcribedAudioBytes,
extraFormatsBytes,
@@ -90,6 +95,14 @@ export function CleanupStage({
}
/>
</div>
+ {supersededAutoSubsIds.length > 0 && (
+ <SupersededAutoSubsSection
+ slug={slug}
+ ids={supersededAutoSubsIds}
+ existingQueues={existingQueues}
+ defaultQueueKey={defaultQueueKey}
+ />
+ )}
{transcodeApplies && (
<div className="flex flex-col gap-2">
<Heading
@@ -124,6 +137,86 @@ export function CleanupStage({
);
}
+// The only thing that deletes a kept auto-caption backup. Deliberately manual:
+// nothing auto-queues it, so an AI-vs-YouTube comparison stays possible for as
+// long as you want it. The sweep re-sniffs every file and removes only English
+// tracks that still read as YouTube ASR; do-not-clean dirs are skipped.
+function SupersededAutoSubsSection({
+ slug,
+ ids,
+ existingQueues,
+ defaultQueueKey,
+}: {
+ slug: string;
+ ids: string[];
+ existingQueues: string[];
+ defaultQueueKey: string;
+}) {
+ const [queue, setQueue] = useState(defaultQueueKey);
+ const [confirmText, setConfirmText] = useState("");
+ const armed = confirmText === "purge";
+ return (
+ <div
+ aria-label="superseded auto captions section"
+ className="flex flex-col gap-2 rounded border border-border p-3"
+ >
+ <div>
+ <h3 className="text-base font-semibold">
+ Superseded auto-captions ({ids.length})
+ </h3>
+ <p className="text-sm text-muted-foreground">
+ These videos now have our own <code>transcript.json</code>, which wins
+ everywhere, while YouTube's original auto-caption VTT is still on
+ disk as a backup. Purging is irreversible — recovering a track means
+ re-running <em>Download missing subs</em>. Only English tracks that
+ still read as auto-generated are removed; manual captions and
+ foreign-language tracks are left alone, as are dirs marked{" "}
+ <em>do not clean</em>.
+ </p>
+ </div>
+ <VideoIdList
+ slug={slug}
+ ids={ids}
+ ariaLabel="superseded auto captions list"
+ emptyAriaLabel="superseded auto captions empty"
+ emptyMessage="None"
+ itemAriaLabel={(id) => `superseded auto captions ${id}`}
+ />
+ <div className="flex flex-col gap-2">
+ <label className="flex items-center gap-2 text-xs text-muted-foreground">
+ Type
+ <span className="font-mono">purge</span>
+ to confirm
+ <input
+ type="text"
+ value={confirmText}
+ onChange={(e) => setConfirmText(e.target.value)}
+ aria-label="confirm purge superseded auto captions"
+ className="rounded border border-border bg-card px-2 py-0.5 font-mono"
+ />
+ </label>
+ <StreamActionLog
+ trigger={() => purgeSupersededAutoSubsAction(slug, queue)}
+ cancelAction={cancelJobAction}
+ buttonLabel={`Purge superseded auto-captions (${ids.length})`}
+ runningLabel="Purging…"
+ label="Purge superseded auto-captions"
+ disabled={!armed}
+ extraControls={
+ <QueueControl
+ value={queue}
+ onChange={setQueue}
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ actionLabel="Purge superseded auto-captions"
+ />
+ }
+ />
+ </div>
+ </div>
+ );
+}
+
function WrongFormatAudioSection({
slug,
ids,
diff --git a/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx b/editor/app/channels/[slug]/components/stages/TranscribeStage.tsx
@@ -16,9 +16,11 @@ import {
import { cancelJobAction } from "../../../../jobs/actions";
import {
clearFailedTranscriptionsAction,
+ transcribeAutoSubsBucketAction,
transcribeBucketAction,
transcribeMissingAction,
} from "../../whisperActions";
+import { RetryBucketControl } from "../RetryBucketControl";
import { VideoIdList } from "../VideoIdList";
type AudioFormatChoice = AudioFormat | "any";
@@ -28,7 +30,14 @@ type Props = {
existingQueues: string[];
failedVideoIds: string[];
downloadedNoTranscriptIds: string[];
+ // Replace-auto-captions lane: videos whose only transcript is YouTube ASR,
+ // split by whether the audio our engine needs is on disk yet.
+ autoSubsOnlyIds: string[];
+ downloadedAutoSubsOnlyIds: string[];
defaultQueueKey: string;
+ // Queue key for the download half of the lane (the platform queue a sync uses),
+ // which is not the transcription queue the rest of this stage runs on.
+ downloadQueueKey: string;
missingShard: ShardConfigSummary | null;
};
@@ -37,11 +46,15 @@ export function TranscribeStage({
existingQueues,
failedVideoIds,
downloadedNoTranscriptIds,
+ autoSubsOnlyIds,
+ downloadedAutoSubsOnlyIds,
defaultQueueKey,
+ downloadQueueKey,
missingShard,
}: Props) {
const channelQueueKey = `channel:${slug}`;
const bucketCount = downloadedNoTranscriptIds.length;
+ const autoSubsCount = autoSubsOnlyIds.length + downloadedAutoSubsOnlyIds.length;
return (
<div className="flex flex-col gap-6">
@@ -60,6 +73,16 @@ export function TranscribeStage({
missingShard={missingShard}
demoted={bucketCount > 0}
/>
+ {autoSubsCount > 0 && (
+ <AutoSubsSection
+ slug={slug}
+ noAudioIds={autoSubsOnlyIds}
+ withAudioIds={downloadedAutoSubsOnlyIds}
+ existingQueues={existingQueues}
+ transcribeQueueKey={defaultQueueKey}
+ downloadQueueKey={downloadQueueKey}
+ />
+ )}
<FailedTranscriptionsSection
slug={slug}
ids={failedVideoIds}
@@ -70,6 +93,138 @@ export function TranscribeStage({
);
}
+// The manual face of the replace-auto-captions lane. Two steps, because a video
+// walks them one restart-safe step at a time: fetch the audio (platform download
+// queue), then transcribe it (worker pool). The original VTT is kept as a backup
+// either way — the Cleanup stage purges those on demand.
+function AutoSubsSection({
+ slug,
+ noAudioIds,
+ withAudioIds,
+ existingQueues,
+ transcribeQueueKey,
+ downloadQueueKey,
+}: {
+ slug: string;
+ noAudioIds: string[];
+ withAudioIds: string[];
+ existingQueues: string[];
+ transcribeQueueKey: string;
+ downloadQueueKey: string;
+}) {
+ const [queue, setQueue] = useState(transcribeQueueKey);
+ const [audioFormat, setAudioFormat] = useState<AudioFormatChoice>("any");
+ const [strictFormat, setStrictFormat] = useState(false);
+ const total = noAudioIds.length + withAudioIds.length;
+ return (
+ <div
+ aria-label="auto captions only section"
+ className="flex flex-col gap-3 rounded border border-border p-3"
+ >
+ <div>
+ <h3 className="text-base font-semibold">
+ YouTube auto-captions only ({total})
+ </h3>
+ <p className="text-sm text-muted-foreground">
+ These videos have no transcript of our own — only YouTube's
+ speech recognition (no punctuation, rolling duplicate cues,{" "}
+ <code>[Music]</code> filler). Replacing them costs an audio download
+ plus a transcription each, so nothing happens automatically unless the{" "}
+ <a href="/auto-queue" className="underline">
+ auto-queue
+ </a>{" "}
+ opts in. Human-written captions are never listed here. The old VTT is
+ kept as a backup; purge it from the Cleanup stage when you no longer
+ want it.
+ </p>
+ </div>
+
+ <div className="flex flex-col gap-2">
+ <h4 className="text-sm font-semibold">
+ Needs audio ({noAudioIds.length})
+ </h4>
+ <p className="text-xs text-muted-foreground">
+ Step 1 — download the audio our engine transcribes from. Runs as a
+ <code> transcribe</code>-handling download so yt-dlp fetches audio
+ instead of re-fetching the subtitles.
+ </p>
+ <VideoIdList
+ slug={slug}
+ ids={noAudioIds}
+ ariaLabel="auto captions needing audio list"
+ emptyAriaLabel="auto captions needing audio empty"
+ emptyMessage="None"
+ itemAriaLabel={(id) => `auto captions needing audio ${id}`}
+ />
+ <RetryBucketControl
+ slug={slug}
+ ids={noAudioIds}
+ actionLabel="auto-captions audio"
+ defaultQueueKey={downloadQueueKey}
+ existingQueues={existingQueues}
+ bucketKey="autoSubsOnly"
+ replaceAutoSubs
+ buttonLabel="Fetch audio"
+ />
+ </div>
+
+ <div className="flex flex-col gap-2">
+ <h4 className="text-sm font-semibold">
+ Ready to transcribe ({withAudioIds.length})
+ </h4>
+ <p className="text-xs text-muted-foreground">
+ Step 2 — run whisper over the downloaded audio. The resulting{" "}
+ <code>transcript.json</code> takes precedence over any VTT everywhere
+ (index, viewer, export).
+ </p>
+ <VideoIdList
+ slug={slug}
+ ids={withAudioIds}
+ ariaLabel="auto captions ready to transcribe list"
+ emptyAriaLabel="auto captions ready to transcribe empty"
+ emptyMessage="None"
+ itemAriaLabel={(id) => `auto captions ready to transcribe ${id}`}
+ />
+ {withAudioIds.length > 0 && (
+ <StreamActionLog
+ trigger={() =>
+ transcribeAutoSubsBucketAction(
+ slug,
+ withAudioIds,
+ queue,
+ audioFormat === "any" ? undefined : audioFormat,
+ audioFormat !== "any" && strictFormat,
+ )
+ }
+ cancelAction={cancelJobAction}
+ buttonLabel={`Replace auto-captions (${withAudioIds.length})`}
+ runningLabel="Transcribing…"
+ label="Replace auto-captions"
+ extraControls={
+ <>
+ <QueueControl
+ value={queue}
+ onChange={setQueue}
+ defaultQueueKey={transcribeQueueKey}
+ existingQueues={existingQueues}
+ actionLabel="Replace auto-captions"
+ />
+ <AudioFormatControl
+ actionLabel="Replace auto-captions"
+ value={audioFormat}
+ onChange={setAudioFormat}
+ strict={strictFormat}
+ onStrictChange={setStrictFormat}
+ />
+ </>
+ }
+ />
+ )}
+ </div>
+ </div>
+ );
+}
+
function BucketTranscribeSection({
slug,
ids,
diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts
@@ -30,6 +30,9 @@ export function normalizeBuckets(
skippedByFilter: raw?.skippedByFilter ?? [],
incompleteTranscript: raw?.incompleteTranscript ?? [],
shortAudio: raw?.shortAudio ?? [],
+ autoSubsOnly: raw?.autoSubsOnly ?? [],
+ downloadedAutoSubsOnly: raw?.downloadedAutoSubsOnly ?? [],
+ supersededAutoSubs: raw?.supersededAutoSubs ?? [],
needsCookies: raw?.needsCookies ?? [],
};
}
@@ -71,7 +74,9 @@ const JOB_KIND_TO_STAGE: Record<string, StageId> = {
"whisper-retry": "transcribe",
"transcribe-one": "transcribe",
"whisper-video": "transcribe",
+ "whisper-bucket-auto-subs": "transcribe",
"clean-audio-transcribed": "cleanup",
+ "purge-superseded-auto-subs": "cleanup",
"clean-extra-audio-formats": "cleanup",
"remove-wrong-format-audio": "cleanup",
"check-availability": "diagnostics",
@@ -289,6 +294,20 @@ export function computeStageStatuses(
if (failedVideoIds.length > 0) {
transcribeParts.push(pluralize(failedVideoIds.length, "failed"));
}
+ // Informational only — the replace-auto-captions lane is opt-in, so these are
+ // NOT counted as pending work (that would light every YouTube channel up
+ // amber forever).
+ const autoSubsCandidates =
+ buckets.autoSubsOnly.length + buckets.downloadedAutoSubsOnly.length;
+ if (autoSubsCandidates > 0) {
+ transcribeParts.push(
+ pluralize(
+ autoSubsCandidates,
+ "video with only auto-captions",
+ "videos with only auto-captions",
+ ),
+ );
+ }
const transcribe: StageStatus = {
id: "transcribe",
title: "Transcribe",
@@ -309,6 +328,28 @@ export function computeStageStatuses(
};
const cleanupRunning = runningByStage.has("cleanup");
+ const cleanupParts: string[] = [];
+ if (cleanupPending > 0) {
+ cleanupParts.push(
+ pluralize(
+ cleanupPending,
+ "dir has extra audio formats",
+ "dirs have extra audio formats",
+ ),
+ );
+ }
+ // Kept auto-caption backups. Informational (not folded into `pending`): they
+ // are deliberately retained until purged by hand, so they are inventory, not
+ // a chore.
+ if (buckets.supersededAutoSubs.length > 0) {
+ cleanupParts.push(
+ pluralize(
+ buckets.supersededAutoSubs.length,
+ "superseded auto-caption backup",
+ "superseded auto-caption backups",
+ ),
+ );
+ }
const cleanup: StageStatus = {
id: "cleanup",
title: "Cleanup",
@@ -318,12 +359,8 @@ export function computeStageStatuses(
defaultOpen: true,
summary: cleanupRunning
? "Running…"
- : cleanupPending > 0
- ? pluralize(
- cleanupPending,
- "dir has extra audio formats",
- "dirs have extra audio formats",
- )
+ : cleanupParts.length > 0
+ ? cleanupParts.join(" · ")
: "Nothing to clean.",
tone: pickTone({
running: cleanupRunning,
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -75,6 +75,10 @@ async function runPipelineAction(
// "always" for this run so every yt-dlp invocation carries the configured
// cookies and the defer-mode exclusion is bypassed.
forceCookies?: boolean;
+ // retry-bucket only (the "YouTube auto-captions only" bucket): let an
+ // ASR-provenance VTT stop counting as an existing destination, so the audio
+ // download actually runs for videos that already have auto-captions.
+ replaceAutoSubs?: boolean;
// Per-run persistence overrides (Phase 2), forwarded to runYtdlp →
// downloadOneManaged for every managed download in this run.
keepSourceVideoOverride?: boolean;
@@ -184,6 +188,7 @@ async function runPipelineAction(
bucketIds: options?.bucketIds,
handlingOverride: options?.handlingOverride,
forceCookies: options?.forceCookies,
+ replaceAutoSubs: options?.replaceAutoSubs,
keepSourceVideoOverride: options?.keepSourceVideoOverride,
extractImmediately: options?.extractImmediately,
audioFormatOverride: options?.audioFormatOverride,
@@ -322,6 +327,11 @@ export async function retryBucketAction(
bucketKey?: ReplayBucket,
// Needs-cookies bucket only: run with cookie mode forced to "always".
forceCookies?: boolean,
+ // Auto-captions-only bucket: bypass the destination-exists prefilter for
+ // videos whose only "destination" is the YouTube ASR VTT we're replacing.
+ // Always paired with handlingOverride "transcribe" (a youtube-handling run
+ // would re-fetch subs instead of audio).
+ replaceAutoSubs?: boolean,
): Promise<StreamActionResult> {
if (!Array.isArray(bucketIds) || bucketIds.length === 0) {
return { ok: false, error: "No video IDs supplied for retry." };
@@ -341,7 +351,13 @@ export async function retryBucketAction(
kind: "retry-bucket",
slug,
bucket: bucketKey,
- params: { queueKey, abortOnError, handlingOverride, forceCookies },
+ params: {
+ queueKey,
+ abortOnError,
+ handlingOverride,
+ forceCookies,
+ replaceAutoSubs,
+ },
}
: undefined;
return runPipelineAction(
@@ -354,6 +370,7 @@ export async function retryBucketAction(
handlingOverride: handling,
abortOnError,
forceCookies,
+ replaceAutoSubs,
spec,
},
);
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -11,6 +11,7 @@ import {
import type { DownloadOutcomeRecord } from "yt-dlp-transcript-common/lib/downloadOutcome";
import type { SavedVideoPointer } from "yt-dlp-transcript-common/lib/savedVideo";
import type { AvailabilityHistoryEntry } from "yt-dlp-transcript-common/lib/availability";
+import type { SubtitleProvenance } from "yt-dlp-transcript-common/lib/subtitleProvenance";
import { formatBytes, formatDuration } from "yt-dlp-transcript-common/lib/format";
import { QueueControl } from "../../../../../components/QueueControl";
import { cancelJobAction } from "../../../../../jobs/actions";
@@ -58,6 +59,9 @@ type Props = {
// Resolved primary English VTT filename (transcript.en.vtt, or a regional/auto
// fallback like transcript.en-US.vtt), or null when no English VTT exists.
primaryVtt: string | null;
+ // filename -> where that subtitle track came from (YouTube ASR vs a manual
+ // upload). Computed server-side from a 4 KB head read per VTT.
+ vttProvenance?: Record<string, SubtitleProvenance>;
handling: ChannelHandling;
defaultQueueKey: string;
existingQueues: string[];
@@ -164,12 +168,20 @@ export function VideoPanel({
prevHref,
nextHref,
position,
+ vttProvenance = {},
}: Props) {
const audioFiles = files.filter((f) => isAudioFile(f.name));
const transcodeSources = files.filter((f) => isTranscodeSource(f.name));
const hasTranscriptJson = files.some((f) => f.name === WHISPER_FILENAME);
const hasYtVtt = primaryVtt !== null;
const hasTranscript = hasTranscriptJson || hasYtVtt;
+ // This video's only transcript is YouTube's speech recognition: the transcribe
+ // controls stay live so it can be replaced with one of our own. A manual — or
+ // unclassifiable — caption track still blocks them.
+ const autoSubsOnly =
+ !hasTranscriptJson &&
+ primaryVtt !== null &&
+ vttProvenance[primaryVtt] === "asr";
const vttTracks = files
.filter((f) => isTranscriptVttName(f.name))
.map((f) => f.name)
@@ -185,11 +197,13 @@ export function VideoPanel({
? "No audio on disk — run the pipeline (VTT first, whisper if needed)."
: "Audio not yet downloaded."
: `${audioFiles.length} audio file${audioFiles.length === 1 ? "" : "s"} on disk.`;
- const transcribeSummary = hasTranscript
- ? "Transcript present."
- : noAudio
- ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript."
- : "Audio ready, no transcript.";
+ const transcribeSummary = autoSubsOnly
+ ? "YouTube auto-captions only — can be replaced with an AI transcript."
+ : hasTranscript
+ ? "Transcript present."
+ : noAudio
+ ? "No audio yet — run the download pipeline (or use Audio + Whisper) to produce a transcript."
+ : "Audio ready, no transcript.";
const transcodeSummary =
transcodeSources.length > 0
? `Convert ${transcodeSources.length} source file${transcodeSources.length === 1 ? "" : "s"} to another format.`
@@ -326,7 +340,8 @@ export function VideoPanel({
videoId={videoId}
file={f}
existingQueues={existingQueues}
- hasTranscript={hasTranscript}
+ hasTranscript={hasTranscript && !autoSubsOnly}
+ replaceAutoSubs={autoSubsOnly}
/>
))}
</div>
@@ -354,6 +369,7 @@ export function VideoPanel({
vttTracks={vttTracks}
primaryVtt={primaryVtt}
hasWhisper={hasTranscriptJson}
+ vttProvenance={vttProvenance}
/>
</PipelineStageCard>
)}
@@ -673,23 +689,33 @@ function PerFileTranscribeRow({
file,
existingQueues,
hasTranscript,
+ replaceAutoSubs = false,
}: {
slug: string;
videoId: string;
file: VideoFile;
existingQueues: string[];
hasTranscript: boolean;
+ // The existing transcript is YouTube ASR: transcribing is offered (and
+ // labelled as a replacement) rather than blocked.
+ replaceAutoSubs?: boolean;
}) {
const [transcribeQueue, setTranscribeQueue] = useState("");
return (
<div className="flex flex-col gap-2 rounded border border-border p-3">
<Heading
- title={`Transcribe ${file.name}`}
+ title={
+ replaceAutoSubs
+ ? `Replace auto-captions with an AI transcript (${file.name})`
+ : `Transcribe ${file.name}`
+ }
desc={
hasTranscript
? "A transcript already exists for this video. Delete transcript.json (or transcript.en.vtt) on disk to re-run."
- : "Run whisper-cli on this audio file. Writes transcript.json next to it."
+ : replaceAutoSubs
+ ? "This video's only transcript is YouTube's speech recognition. Run whisper-cli on this audio file to write a transcript.json, which takes precedence everywhere. The auto-caption VTT is kept on disk as a backup."
+ : "Run whisper-cli on this audio file. Writes transcript.json next to it."
}
/>
{hasTranscript ? (
@@ -702,7 +728,11 @@ function PerFileTranscribeRow({
transcribeOneAction(slug, videoId, file.name, transcribeQueue)
}
cancelAction={cancelJobAction}
- buttonLabel={`Transcribe ${file.name}`}
+ buttonLabel={
+ replaceAutoSubs
+ ? `Replace auto-captions (${file.name})`
+ : `Transcribe ${file.name}`
+ }
runningLabel="Transcribing…"
label={`Transcribe ${file.name}`}
extraControls={
@@ -841,18 +871,26 @@ function FilesList({
);
}
+const PROVENANCE_LABEL: Record<SubtitleProvenance, string> = {
+ asr: "YouTube auto-captions",
+ manual: "manual captions",
+ unknown: "unknown source",
+};
+
function TranscriptSourceSection({
slug,
videoId,
vttTracks,
primaryVtt,
hasWhisper,
+ vttProvenance,
}: {
slug: string;
videoId: string;
vttTracks: string[];
primaryVtt: string | null;
hasWhisper: boolean;
+ vttProvenance: Record<string, SubtitleProvenance>;
}) {
const [pending, startTransition] = useTransition();
const [busyFile, setBusyFile] = useState<string | null>(null);
@@ -897,6 +935,14 @@ function TranscriptSourceSection({
>
<span className="flex items-center gap-2 font-mono text-sm">
{name}
+ {vttProvenance[name] && (
+ <span
+ aria-label={`transcript provenance ${name}`}
+ className="rounded bg-muted px-1.5 py-0.5 text-[10px] font-sans font-medium uppercase tracking-wide text-muted-foreground"
+ >
+ {PROVENANCE_LABEL[vttProvenance[name]]}
+ </span>
+ )}
{isPrimary && (
<span
aria-label={`primary transcript ${name}`}
diff --git a/editor/app/channels/[slug]/videos/[id]/page.tsx b/editor/app/channels/[slug]/videos/[id]/page.tsx
@@ -11,7 +11,14 @@ import { isDoNotClean } from "yt-dlp-transcript-common/lib/doNotClean-server";
import { isExcludedFromTruncatedCheck } from "yt-dlp-transcript-common/lib/excludeTruncatedCheck-server";
import { loadSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
-import { resolvePrimaryVtt } from "yt-dlp-transcript-common/lib/videoStatus";
+import {
+ isTranscriptVtt,
+ resolvePrimaryVtt,
+} from "yt-dlp-transcript-common/lib/videoStatus";
+import {
+ resolveVttProvenance,
+ type SubtitleProvenance,
+} from "yt-dlp-transcript-common/lib/subtitleProvenance";
import { readTranscriptCoverage } from "yt-dlp-transcript-common/controller/normalizeTranscript";
import { isIncompleteTranscript } from "yt-dlp-transcript-common/lib/transcriptCoverage";
import {
@@ -108,6 +115,14 @@ export default async function VideoDetailPage({
const excludedFromTruncatedCheck =
await isExcludedFromTruncatedCheck(videoDir);
const savedVideo = await loadSavedVideo(videoDir);
+ // Where each subtitle track came from, so the panel can say "YouTube
+ // auto-captions" vs "manual captions" — and offer to replace the former with a
+ // transcript of our own. One 4 KB head read per VTT (see subtitleProvenance).
+ const vttProvenance: Record<string, SubtitleProvenance> = {};
+ for (const f of dirData.files) {
+ if (!isTranscriptVtt(f.name)) continue;
+ vttProvenance[f.name] = await resolveVttProvenance(videoDir, f.name);
+ }
const cov = await readTranscriptCoverage(videoDir);
const coverage = cov
? {
@@ -192,6 +207,7 @@ export default async function VideoDetailPage({
videoId={id}
files={dirData.files}
primaryVtt={resolvePrimaryVtt(dirData.files.map((f) => f.name))}
+ vttProvenance={vttProvenance}
handling={config.handling}
defaultQueueKey={defaultQueueKey}
existingQueues={existingQueues}
diff --git a/editor/app/channels/[slug]/whisperActions.ts b/editor/app/channels/[slug]/whisperActions.ts
@@ -25,6 +25,7 @@ import {
} from "yt-dlp-transcript-common/controller/failedTranscodings";
import { clearFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions";
import { cleanAudioFromTranscribed } from "yt-dlp-transcript-common/controller/cleanAudioFromTranscribed";
+import { purgeSupersededAutoSubs } from "yt-dlp-transcript-common/controller/purgeSupersededAutoSubs";
import { checkKeptDeleted } from "yt-dlp-transcript-common/controller/checkKeptDeleted";
import { cleanExtraAudioFormats } from "yt-dlp-transcript-common/controller/cleanExtraAudioFormats";
import { removeWrongFormatAudio } from "yt-dlp-transcript-common/controller/removeWrongFormatAudio";
@@ -160,6 +161,59 @@ export async function transcribeBucketAction(
});
}
+// Replace-auto-captions lane, transcribe half. Same batch machinery as
+// transcribeBucketAction, with the lane flag set so runWhisperBatch treats only
+// transcript.json as "already transcribed" (an English VTT no longer counts) and
+// re-verifies each id's provenance before touching it. The superseded VTT is
+// left on disk — the Cleanup stage's purge is the only thing that removes it.
+export async function transcribeAutoSubsBucketAction(
+ slug: string,
+ ids: string[],
+ queueKey?: string,
+ audioFormat?: AudioFormat,
+ strictAudioFormat?: boolean,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const fmt = sanitizeAudioFormat(audioFormat);
+ const cleaned = Array.from(new Set(ids.map((id) => id.trim()).filter(Boolean)));
+ if (cleaned.length === 0) {
+ return { ok: false, error: "No video ids supplied" };
+ }
+ return runManagedFunction({
+ kind: "whisper-bucket-auto-subs",
+ queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey),
+ paths,
+ channelSlug: slug,
+ spec: {
+ kind: "whisper-bucket-auto-subs",
+ slug,
+ bucket: "downloadedAutoSubsOnly",
+ params: { queueKey, audioFormat, strictAudioFormat },
+ },
+ fn: async (onLog, signal, _setProgress, ctx) => {
+ // No setProgress here: the channel's transcriptCount already counts these
+ // videos (their VTT is a transcript), so a "transcripts" progress range
+ // would be a flat, meaningless bar.
+ const result = await runWhisperBatch({
+ channelSlug: slug,
+ paths,
+ audioFormat: fmt,
+ strictAudioFormat: fmt !== undefined && strictAudioFormat === true,
+ ids: cleaned,
+ replaceAutoSubs: true,
+ onLog,
+ signal,
+ drainSignal: ctx.drainSignal,
+ tracker: makeTaskTracker(ctx, onLog),
+ });
+ onLog(
+ `Replace auto-captions: ${result.succeeded} succeeded, ${result.failed} failed, ${result.skipped} skipped, ${result.attempted} attempted.`,
+ );
+ revalidatePath(`/channels/${slug}`);
+ },
+ });
+}
+
export async function clearFailedTranscriptionsAction(
slug: string,
queueKey?: string,
@@ -368,6 +422,39 @@ export async function cleanAudioAction(
});
}
+// Delete the YouTube auto-caption VTTs that our own transcript superseded (the
+// supersededAutoSubs bucket). Manual only — never auto-queued — and the single
+// irreversible step in the lane, so it lives next to the Clean-audio sweep and
+// honors the same do-not-clean marker. Scoped to English ASR-provenance tracks:
+// the controller re-sniffs every file before removing it.
+export async function purgeSupersededAutoSubsAction(
+ slug: string,
+ queueKey?: string,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ return runManagedFunction({
+ kind: "purge-superseded-auto-subs",
+ queueKey: resolveQueueKey(channelQueueKey(slug), queueKey),
+ paths,
+ channelSlug: slug,
+ spec: {
+ kind: "purge-superseded-auto-subs",
+ slug,
+ bucket: "supersededAutoSubs",
+ params: { queueKey },
+ },
+ fn: async (onLog, signal) => {
+ await purgeSupersededAutoSubs({
+ channelSlug: slug,
+ paths,
+ onLog,
+ signal,
+ });
+ revalidatePath(`/channels/${slug}`);
+ },
+ });
+}
+
// Re-probe the channel's keep-latest window for source deletion and pin any
// gone videos (do-not-clean) so they survive even after rolling out of the
// window. Mirrors cleanAudioAction's managed-job shape. Also driven by the sync
diff --git a/editor/e2e/auto-subs-replace.spec.ts b/editor/e2e/auto-subs-replace.spec.ts
@@ -0,0 +1,425 @@
+// The opt-in "replace YouTube auto-captions" lane, end to end.
+//
+// A video whose only transcript is YouTube's speech recognition is invisible to
+// every transcribe path (isVideoTranscribed counts any English VTT). These tests
+// prove the three new snapshot buckets classify such videos correctly — and only
+// such videos — and that the manual controls walk one through the whole lane:
+// autoSubsOnly → (fetch audio) → downloadedAutoSubsOnly → (transcribe)
+// → supersededAutoSubs → (purge) → nothing.
+//
+// Provenance comes from a 4 KB sniff of the VTT itself (see
+// common/lib/subtitleProvenance.ts), so the fixtures below use realistically
+// shaped cues: ASR tracks carry `align:start position:N%` plus inline word
+// timings, manual tracks carry neither.
+//
+// Run in default dev mode — E2E_MODE=start serves a stale build.
+
+import { mkdir, writeFile } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import {
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSettings,
+} from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+const SLUG = "test-auto-subs";
+const ROOT = `test-transcripts/channels/${SLUG}`;
+const SNAPSHOT_REL = `${ROOT}/snapshot.json`;
+
+// Shaped after real yt-dlp --write-auto-subs output.
+const ASR_VTT = `WEBVTT
+Kind: captions
+Language: en
+
+00:00:00.030 --> 00:00:03.919 align:start position:0%
+so<00:00:00.719> today<00:00:01.199> we're<00:00:01.439> going<00:00:01.680> to
+
+00:00:03.919 --> 00:00:03.929 align:start position:0%
+so today we're going to
+
+00:00:03.929 --> 00:00:07.070 align:start position:0%
+so today we're going to
+talk<00:00:04.320> about<00:00:04.639> the<00:00:04.879> whole<00:00:05.199> thing
+`;
+
+// Shaped after a human-uploaded caption track: no cue settings, no word timings.
+const MANUAL_VTT = `WEBVTT
+Kind: captions
+Language: en
+
+00:00:01.000 --> 00:00:04.000
+So today we're going to talk about the whole thing.
+
+00:00:04.000 --> 00:00:08.500
+It's a long story, but bear with me.
+
+00:00:08.500 --> 00:00:12.000
+Here we go.
+`;
+
+const WHISPER_JSON = JSON.stringify({
+ transcription: [{ text: "a real transcript" }],
+});
+
+type SeedVideo = {
+ id: string;
+ // "asr" writes an auto-caption-shaped VTT and metadata listing the track under
+ // automatic_captions; "manual" writes a human-shaped VTT under subtitles.
+ captions: "asr" | "manual";
+ whisper?: boolean;
+ audio?: boolean;
+ doNotClean?: boolean;
+};
+
+function dataRel(videoId: string, file: string): string {
+ return `${ROOT}/data/${videoId}/${file}`;
+}
+
+async function seedChannel(videos: SeedVideo[]): Promise<void> {
+ await mkdir(resolvePath(`${ROOT}/data`), { recursive: true });
+ // A youtube-handling channel — the case the lane exists for. audioFormat is
+ // pinned so the forced transcribe-handling download lands on audio.mp3.
+ await writeFile(
+ resolvePath(`${ROOT}/config.json`),
+ JSON.stringify({
+ handling: "youtube",
+ name: "Auto-subs test channel",
+ url: "https://www.youtube.com/@autosubs/videos",
+ audioFormat: "mp3",
+ }),
+ );
+ // retry-bucket resolves each id's source URL from the stored playlist.
+ await writeFile(
+ resolvePath(`${ROOT}/playlist`),
+ videos.map((v) => `https://www.youtube.com/watch?v=${v.id}\n`).join(""),
+ );
+ for (const v of videos) {
+ const dir = resolvePath(`${ROOT}/data/${v.id}`);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ `${dir}/transcript.en.vtt`,
+ v.captions === "asr" ? ASR_VTT : MANUAL_VTT,
+ );
+ const track = { en: [{ ext: "vtt", url: "fake://subs" }] };
+ await writeFile(
+ `${dir}/metadata.info.json`,
+ JSON.stringify({
+ id: v.id,
+ title: `Synthetic ${v.id}`,
+ upload_date: "20240101",
+ duration: 60,
+ extractor_key: "Youtube",
+ webpage_url: `https://www.youtube.com/watch?v=${v.id}`,
+ subtitles: v.captions === "manual" ? track : {},
+ automatic_captions: v.captions === "asr" ? track : {},
+ }),
+ );
+ if (v.whisper) await writeFile(`${dir}/transcript.json`, WHISPER_JSON);
+ if (v.audio) await writeFile(`${dir}/audio.mp3`, `fake audio ${v.id}\n`);
+ if (v.doNotClean) {
+ await writeFile(
+ `${dir}/do-not-clean.json`,
+ JSON.stringify({ setAt: new Date(0).toISOString() }),
+ );
+ }
+ }
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+}
+
+type Buckets = {
+ autoSubsOnly?: string[];
+ downloadedAutoSubsOnly?: string[];
+ supersededAutoSubs?: string[];
+};
+
+// Snapshot regeneration is debounced (~1s) after a page visit / job finish, so
+// every assertion on it polls rather than reading once.
+async function expectBuckets(expected: Buckets): Promise<void> {
+ await expect
+ .poll(
+ async () => {
+ const snap = await readJson<{ buckets: Buckets }>(SNAPSHOT_REL).catch(
+ () => null,
+ );
+ if (!snap) return null;
+ return {
+ autoSubsOnly: snap.buckets.autoSubsOnly ?? [],
+ downloadedAutoSubsOnly: snap.buckets.downloadedAutoSubsOnly ?? [],
+ supersededAutoSubs: snap.buckets.supersededAutoSubs ?? [],
+ };
+ },
+ { timeout: 20_000 },
+ )
+ .toEqual({
+ autoSubsOnly: expected.autoSubsOnly ?? [],
+ downloadedAutoSubsOnly: expected.downloadedAutoSubsOnly ?? [],
+ supersededAutoSubs: expected.supersededAutoSubs ?? [],
+ });
+}
+
+test("walks an auto-caption video through fetch → transcribe → purge", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData();
+ await seedChannel([
+ { id: "asrvid0001", captions: "asr" },
+ // Negative control: a human-captioned video must never enter the lane.
+ { id: "manvid0001", captions: "manual" },
+ ]);
+
+ // --- Step 0: classification -----------------------------------------------
+ await page.goto(`/channels/${SLUG}`);
+ await expectBuckets({ autoSubsOnly: ["asrvid0001"] });
+
+ await expect(
+ page.getByRole("heading", { name: "YouTube auto-captions only (1)" }),
+ ).toBeVisible();
+ await expect(
+ page
+ .getByLabel("auto captions needing audio list")
+ .getByLabel("auto captions needing audio asrvid0001"),
+ ).toBeVisible();
+
+ // --- Step 1: fetch the audio our engine transcribes from ------------------
+ await page
+ .getByLabel("retry auto-captions audio bucket")
+ .getByRole("button", { name: /^Fetch audio \(1\)$/ })
+ .click();
+ const downloadLog = page.getByLabel("Retry auto-captions audio output");
+ // The VTT must NOT prefilter the video away as "already downloaded".
+ await expect(downloadLog).toContainText("Prefilter: 1 missing destination", {
+ timeout: 60_000,
+ });
+ await expect(downloadLog).toContainText("Managed download complete", {
+ timeout: 60_000,
+ });
+ expect(await pathExists(dataRel("asrvid0001", "audio.mp3"))).toBe(true);
+
+ // --- Step 2: transcribe over the auto-captions ----------------------------
+ await page.goto(`/channels/${SLUG}`);
+ await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0001"] });
+
+ await page.reload();
+ await page
+ .getByRole("button", { name: /^Replace auto-captions \(1\)$/ })
+ .click();
+ await expect(page.getByLabel("Replace auto-captions output")).toContainText(
+ "1 succeeded",
+ { timeout: 60_000 },
+ );
+ expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true);
+ // normalizeTranscript regenerates the derived cues from the new transcript.
+ expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe(
+ true,
+ );
+ // The superseded VTT is KEPT as a backup — nothing deletes it automatically.
+ expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe(
+ true,
+ );
+
+ // --- Step 3: the backup shows up as purgeable inventory -------------------
+ await page.goto(`/channels/${SLUG}`);
+ await expectBuckets({ supersededAutoSubs: ["asrvid0001"] });
+
+ await page.reload();
+ await page.getByRole("button", { name: "Cleanup stage summary" }).click();
+ const section = page.getByLabel("superseded auto captions section");
+ await expect(
+ section.getByRole("heading", { name: "Superseded auto-captions (1)" }),
+ ).toBeVisible();
+
+ // --- Step 4: purge, and only then does the VTT go ------------------------
+ await section
+ .getByLabel("confirm purge superseded auto captions")
+ .fill("purge");
+ await section
+ .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ })
+ .click();
+ await expect(
+ page.getByLabel("Purge superseded auto-captions output"),
+ ).toContainText("Purged 1 superseded auto-caption file", {
+ timeout: 60_000,
+ });
+
+ expect(await pathExists(dataRel("asrvid0001", "transcript.en.vtt"))).toBe(
+ false,
+ );
+ expect(await pathExists(dataRel("asrvid0001", "transcript.json"))).toBe(true);
+ expect(await pathExists(dataRel("asrvid0001", "transcript.cues.json"))).toBe(
+ true,
+ );
+ // The human-captioned video was never touched at any step.
+ expect(await pathExists(dataRel("manvid0001", "transcript.en.vtt"))).toBe(
+ true,
+ );
+
+ await page.goto(`/channels/${SLUG}`);
+ await expectBuckets({});
+});
+
+test("never buckets or purges captions it can't prove are auto-generated", async ({
+ page,
+}) => {
+ test.setTimeout(90_000);
+ await resetData();
+ await seedChannel([
+ // Superseded ASR backup: the one thing the purge may remove.
+ { id: "asrdone0001", captions: "asr", whisper: true },
+ // Manual captions alongside our transcript: not a backup, never purged.
+ { id: "mandone0001", captions: "manual", whisper: true },
+ // Manual captions and no transcript of ours: not lane work either.
+ { id: "manonly0001", captions: "manual" },
+ // Archived media: shielded from the purge exactly like the Clean-audio sweep.
+ { id: "asrkeep0001", captions: "asr", whisper: true, doNotClean: true },
+ ]);
+
+ await page.goto(`/channels/${SLUG}`);
+ // Only the unprotected ASR-plus-whisper video is listed as a backup, and no
+ // manual-caption video appears in the work lane at all.
+ await expectBuckets({ supersededAutoSubs: ["asrdone0001"] });
+
+ await page.reload();
+ await page.getByRole("button", { name: "Cleanup stage summary" }).click();
+ const section = page.getByLabel("superseded auto captions section");
+ await section
+ .getByLabel("confirm purge superseded auto captions")
+ .fill("purge");
+ await section
+ .getByRole("button", { name: /^Purge superseded auto-captions \(1\)$/ })
+ .click();
+ const log = page.getByLabel("Purge superseded auto-captions output");
+ await expect(log).toContainText("Purged 1 superseded auto-caption file", {
+ timeout: 60_000,
+ });
+ await expect(log).toContainText("Skipped asrkeep0001 (marked do not clean)");
+ await expect(log).toContainText(
+ "Kept mandone0001/transcript.en.vtt (manual captions",
+ );
+
+ expect(await pathExists(dataRel("asrdone0001", "transcript.en.vtt"))).toBe(
+ false,
+ );
+ expect(await pathExists(dataRel("mandone0001", "transcript.en.vtt"))).toBe(
+ true,
+ );
+ expect(await pathExists(dataRel("manonly0001", "transcript.en.vtt"))).toBe(
+ true,
+ );
+ expect(await pathExists(dataRel("asrkeep0001", "transcript.en.vtt"))).toBe(
+ true,
+ );
+});
+
+// One enabled local worker using the fake whisper engine (mirrors auto-queue.spec).
+const ONE_WORKER = [
+ {
+ id: "w1",
+ name: "W1",
+ kind: "local",
+ enabled: true,
+ priority: 0,
+ appId: "whisper-cpp",
+ config: {},
+ },
+];
+
+test("the auto-transcribe runner transcribes over auto-captions when opted in", async ({
+ page,
+ request,
+}) => {
+ test.setTimeout(120_000);
+ await resetData();
+ await seedChannel([
+ // Already has its audio, so the transcribe runner can take it directly.
+ { id: "asrvid0003", captions: "asr", audio: true },
+ // Manual captions + audio: the runner must never pick this one up.
+ { id: "manvid0003", captions: "manual", audio: true },
+ ]);
+ await page.goto(`/channels/${SLUG}`);
+ await expectBuckets({ downloadedAutoSubsOnly: ["asrvid0003"] });
+
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ workers: ONE_WORKER,
+ autoQueue: {
+ transcription: {
+ enabled: true,
+ maxWorkers: 1,
+ // The switch under test: without it the runner's default union never
+ // reaches the auto-caption bucket.
+ replaceAutoSubs: true,
+ root: { id: "root", mode: "strict", children: [{ id: "leaf-all", match: { type: "all" } }] },
+ },
+ download: {},
+ },
+ });
+
+ const started = await request.post(`${baseUrl}/api/auto-queue/control`, {
+ data: { kind: "transcription", action: "start" },
+ });
+ expect(started.ok()).toBeTruthy();
+ try {
+ await expect
+ .poll(
+ () => pathExists(dataRel("asrvid0003", "transcript.json")),
+ { timeout: 60_000 },
+ )
+ .toBe(true);
+ // The manual-caption video stays untouched no matter how long the runner idles.
+ expect(await pathExists(dataRel("manvid0003", "transcript.json"))).toBe(
+ false,
+ );
+ } finally {
+ await request.post(`${baseUrl}/api/auto-queue/control`, {
+ data: { kind: "transcription", action: "stop" },
+ });
+ }
+});
+
+test("the auto-queue opt-in is off by default and persists when enabled", async ({
+ page,
+}) => {
+ await resetData();
+ await seedChannel([{ id: "asrvid0002", captions: "asr", audio: true }]);
+
+ await page.goto("/auto-queue");
+ const optIn = page.getByLabel(
+ "replace YouTube auto-captions for auto-transcription",
+ );
+ await expect(optIn).not.toBeChecked();
+
+ // The opt-in bucket is also offered per-leaf, so a single channel can join the
+ // lane without flipping the runner-wide switch.
+ await page.getByRole("button", { name: "+ Channel rule" }).first().click();
+ await expect(
+ page.getByRole("option", { name: "downloadedAutoSubsOnly" }),
+ ).toHaveCount(1);
+
+ await optIn.check();
+ await page.getByRole("button", { name: "Save policy" }).first().click();
+ await expect(page.getByRole("status").first()).toHaveText("Saved.");
+
+ await expect
+ .poll(
+ async () => {
+ const settings = await readJson<{
+ autoQueue?: { transcription?: { replaceAutoSubs?: boolean } };
+ }>("test-settings.json").catch(() => null);
+ return settings?.autoQueue?.transcription?.replaceAutoSubs ?? null;
+ },
+ { timeout: 10_000 },
+ )
+ .toBe(true);
+
+ await page.reload();
+ await expect(
+ page.getByLabel("replace YouTube auto-captions for auto-transcription"),
+ ).toBeChecked();
+});