commit dffe8bb8cf2f045e78d39196ab4f7bdc9b143ba1
parent d1c783cde9dc61ae5af5a588b517a4d669f612b2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 29 Jun 2026 22:07:34 -0400
Bulk-fix incomplete transcripts: shared helpers, actions, job plumbing
Add a shared per-video module (fixIncompleteTranscript.ts) as the single
source of truth for re-fixing/clearing a truncated transcript, and refactor
the existing per-video redownloadIncompleteTranscriptAction onto it.
New channel-level actions (incompleteTranscriptActions.ts):
- redownloadIncompleteBucketAction: one managed batch job that removes the
truncated audio, re-downloads, and re-transcribes each flagged video in
place (new bookmarkable kind redownload-incomplete-bucket).
- clearIncompleteTranscriptsAction: synchronously deletes the truncated audio
+ transcript so videos fall back into the undownloaded -> needs-transcript
pipeline, then enables + starts the auto-download/auto-transcribe runners.
Bulk-bar wrappers (bulkVideoActions.ts) and global all-channels actions
(actionable/actions.ts) reuse the same code. Register the new job kind and
the incompleteTranscript replay bucket.
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Diffstat:
8 files changed, 384 insertions(+), 31 deletions(-)
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -77,6 +77,13 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
bookmarkable: true,
queueKeyStrategy: "custom",
},
+ "redownload-incomplete-bucket": {
+ kind: "redownload-incomplete-bucket",
+ label: "Re-download truncated transcripts",
+ drainable: true,
+ bookmarkable: true,
+ queueKeyStrategy: "custom",
+ },
"download-from-playlist": {
kind: "download-from-playlist",
label: "Download from playlist",
diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts
@@ -18,7 +18,8 @@
export type ReplayBucket =
| "partialDownloads"
| "noTranscript"
- | "downloadedNoTranscript";
+ | "downloadedNoTranscript"
+ | "incompleteTranscript";
export type JobSpec = {
kind: string;
@@ -36,6 +37,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([
"partialDownloads",
"noTranscript",
"downloadedNoTranscript",
+ "incompleteTranscript",
]);
// Defensive parse for a spec read back from JSON (a sidecar or the bookmarks
diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts
@@ -11,6 +11,12 @@ import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand";
import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream";
import { detectDuplicateShorts } from "yt-dlp-transcript-common/controller/duplicateShorts";
+import {
+ clearIncompleteTranscriptsAction,
+ enableAutoRunners,
+ redownloadIncompleteBucketAction,
+} from "../channels/[slug]/incompleteTranscriptActions";
+import { loadActionableSummary } from "./lib/loadActionable";
export type RefreshAllResult = {
queued: string[];
@@ -123,3 +129,53 @@ export async function runDuplicateDetectionAction(
revalidatePath("/actionable");
return { ok: true, clusters, videosInClusters };
}
+
+export type GlobalIncompleteResult =
+ | { ok: true; channels: number; affected: number }
+ | { ok: false; error: string };
+
+// All channels with at least one truncated transcript right now.
+async function affectedIncompleteSlugs(): Promise<string[]> {
+ const summary = await loadActionableSummary(getPaths());
+ return summary.incompleteTranscripts.map((r) => r.channel.slug);
+}
+
+// Clear & re-queue every truncated transcript across all channels, then enable
+// the auto-runners once. Destructive — the caller confirms first.
+export async function clearAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> {
+ const slugs = await affectedIncompleteSlugs();
+ if (slugs.length === 0) {
+ return { ok: false, error: "No incomplete transcripts to clear." };
+ }
+ let cleared = 0;
+ for (const slug of slugs) {
+ // Defer enabling the runners until the end so settings flip only once.
+ const r = await clearIncompleteTranscriptsAction(slug, undefined, {
+ enableRunners: false,
+ });
+ cleared += r.succeeded;
+ }
+ await enableAutoRunners();
+ revalidatePath("/actionable");
+ return { ok: true, channels: slugs.length, affected: cleared };
+}
+
+// Queue one batch re-fix job per affected channel (fire-and-forget — the jobs
+// keep running and surface in /jobs; cancel each stream so we don't hold them
+// open).
+export async function redownloadAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> {
+ const slugs = await affectedIncompleteSlugs();
+ if (slugs.length === 0) {
+ return { ok: false, error: "No incomplete transcripts to fix." };
+ }
+ let queued = 0;
+ for (const slug of slugs) {
+ const result = await redownloadIncompleteBucketAction(slug);
+ if (result.ok) {
+ void result.stream.cancel();
+ queued++;
+ }
+ }
+ revalidatePath("/actionable");
+ return { ok: true, channels: slugs.length, affected: queued };
+}
diff --git a/editor/app/channels/[slug]/bulkVideoActions.ts b/editor/app/channels/[slug]/bulkVideoActions.ts
@@ -16,6 +16,10 @@ import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"
import { transcribeBucketAction } from "./whisperActions";
import { retryBucketAction } from "./pipelineActions";
import {
+ clearIncompleteTranscriptsAction,
+ redownloadIncompleteBucketAction,
+} from "./incompleteTranscriptActions";
+import {
deleteOneVideoDir,
markVideoUntranscribableAction,
removeAudioFilesForVideo,
@@ -50,6 +54,27 @@ export async function bulkRetryDownloadAction(
return retryBucketAction(slug, videoIds, queueKey, abortOnError);
}
+// Bulk re-fix truncated transcripts: queues ONE batch job that re-downloads +
+// re-transcribes each selected video in place (same one-job-per-bulk convention
+// as bulkTranscribeAction).
+export async function bulkRedownloadIncompleteAction(
+ slug: string,
+ videoIds: string[],
+ queueKey?: string,
+): Promise<StreamActionResult> {
+ return redownloadIncompleteBucketAction(slug, videoIds, queueKey);
+}
+
+// Bulk clear truncated transcripts: synchronous fs op (delete audio + transcript)
+// that enables the auto-runners, so it reports a per-id summary like the other
+// clear/remove bulk actions. Destructive — the caller confirms first.
+export async function bulkClearIncompleteAction(
+ slug: string,
+ videoIds: string[],
+): Promise<BulkActionSummary> {
+ return clearIncompleteTranscriptsAction(slug, videoIds);
+}
+
// Marking videos untranscribable is an instant metadata write, not a queued
// batch, so it stays synchronous and reports a per-id summary.
export async function bulkMarkUntranscribableAction(
diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts
@@ -0,0 +1,152 @@
+"use server";
+
+import { revalidatePath } from "next/cache";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import {
+ TRANSCRIPTION_QUEUE,
+ resolveQueueKey,
+} from "yt-dlp-transcript-common/lib/queueKeys";
+import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
+import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings";
+import { startAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "yt-dlp-transcript-common/jobs/streamCommand";
+import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
+import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks";
+import {
+ clearIncompleteTranscriptOne,
+ fixIncompleteTranscriptOne,
+ incompleteIdsForChannel,
+} from "./lib/fixIncompleteTranscript";
+import type { BulkActionSummary } from "./bulkVideoActions";
+
+function dedupeIds(ids: string[]): string[] {
+ return Array.from(new Set(ids.map((id) => id.trim()).filter(Boolean)));
+}
+
+// Turn on + start both auto-queue runners so cleared videos get reprocessed
+// automatically. Enabling persists across restarts (settings.json); starting
+// brings the runner up now without a server restart (same path the auto-queue
+// admin page and /api/auto-queue/control use). NOTE: the runner only picks up a
+// channel its policy tree actually matches — cleared videos also surface in the
+// manual "Download missing" / "Transcribe pending" actionable sections as a
+// fallback.
+export async function enableAutoRunners(): Promise<void> {
+ const current = getSettings();
+ if (
+ !current.autoQueue.transcription.enabled ||
+ !current.autoQueue.download.enabled
+ ) {
+ await writeSettings({
+ ...current,
+ autoQueue: {
+ ...current.autoQueue,
+ transcription: { ...current.autoQueue.transcription, enabled: true },
+ download: { ...current.autoQueue.download, enabled: true },
+ },
+ });
+ }
+ await startAutoRunner("transcription");
+ await startAutoRunner("download");
+}
+
+// One-shot batch re-fix: queue a single managed job that removes the truncated
+// audio, re-downloads, and re-transcribes each flagged video in place (the bulk
+// version of the per-video redownloadIncompleteTranscriptAction). ids default to
+// the channel's current incompleteTranscript bucket; bookmarkable so a re-run
+// re-derives the live bucket.
+export async function redownloadIncompleteBucketAction(
+ slug: string,
+ ids?: string[],
+ queueKey?: string,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `Channel "${slug}" not found` };
+ const source = ids && ids.length ? ids : await incompleteIdsForChannel(slug, paths);
+ const cleaned = dedupeIds(source);
+ if (cleaned.length === 0) {
+ return { ok: false, error: "No incomplete transcripts to fix.", info: true };
+ }
+ return runManagedFunction({
+ kind: "redownload-incomplete-bucket",
+ queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey),
+ paths,
+ channelSlug: slug,
+ spec: {
+ kind: "redownload-incomplete-bucket",
+ slug,
+ bucket: "incompleteTranscript",
+ params: { queueKey },
+ },
+ fn: async (onLog, signal, _setProgress, ctx) => {
+ const tracker = makeTaskTracker(ctx, onLog);
+ let succeeded = 0;
+ let failed = 0;
+ for (const id of cleaned) {
+ if (ctx.drainSignal?.aborted) {
+ onLog(`Drain requested; stopping before ${id}.`);
+ break;
+ }
+ try {
+ onLog(`Re-downloading & re-transcribing ${id}…`);
+ await fixIncompleteTranscriptOne({
+ slug,
+ videoId: id,
+ config,
+ paths,
+ onLog,
+ signal,
+ tracker,
+ });
+ succeeded++;
+ } catch (e) {
+ failed++;
+ onLog(`Failed ${id}: ${(e as Error).message}`);
+ }
+ }
+ onLog(
+ `Re-download incomplete: ${succeeded} fixed, ${failed} failed of ${cleaned.length}.`,
+ );
+ revalidatePath(`/channels/${slug}`);
+ },
+ });
+}
+
+// Clear & release: synchronously delete the truncated audio + transcript for the
+// flagged videos so they fall back into the normal pending pipeline, then enable
+// the auto-runners so they reprocess automatically. Destructive — gate every
+// entry point with a confirm. ids default to the channel's incompleteTranscript
+// bucket.
+export async function clearIncompleteTranscriptsAction(
+ slug: string,
+ ids?: string[],
+ opts?: { enableRunners?: boolean },
+): Promise<BulkActionSummary> {
+ const paths = getPaths();
+ const source = ids && ids.length ? ids : await incompleteIdsForChannel(slug, paths);
+ const cleaned = dedupeIds(source);
+ const failures: BulkActionSummary["failures"] = [];
+ let succeeded = 0;
+ for (const id of cleaned) {
+ try {
+ await clearIncompleteTranscriptOne({ slug, videoId: id, paths });
+ succeeded++;
+ } catch (e) {
+ failures.push({ videoId: id, error: (e as Error).message });
+ }
+ }
+ if (opts?.enableRunners !== false && cleaned.length > 0) {
+ await enableAutoRunners();
+ }
+ revalidatePath(`/channels/${slug}`);
+ requestChannelSnapshot(paths, slug);
+ return {
+ ok: failures.length === 0,
+ attempted: cleaned.length,
+ succeeded,
+ failures,
+ };
+}
diff --git a/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts b/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts
@@ -0,0 +1,126 @@
+// Shared per-video building blocks for fixing truncated ("incomplete")
+// transcripts, reused by the per-video action, the channel-level batch job, the
+// bulk-bar wrappers, and the global all-channels actions. Server-only (it does
+// fs + spawns yt-dlp/whisper) but NOT a "use server" module — it exports plain
+// helpers that take onLog/signal/tracker, which aren't serializable across the
+// server-action boundary.
+//
+// A truncated transcript means the audio download silently stopped early: the
+// audio on disk is itself short, so re-running whisper on it just reproduces the
+// short transcript. The fix MUST re-fetch the audio first.
+
+import path from "node:path";
+import { readdir, rm } from "node:fs/promises";
+import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig";
+import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths";
+import {
+ isRealAudioFile,
+ isTranscriptVtt,
+} from "yt-dlp-transcript-common/lib/videoStatus";
+import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transcribeOne";
+import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos";
+import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot";
+import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
+import type { TaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks";
+
+function videoDirOf(paths: Paths, slug: string, videoId: string): string {
+ return path.join(paths.channelsDir, slug, "data", videoId);
+}
+
+// The current snapshot's truncated-transcript ids for a channel. Empty when the
+// snapshot is missing or has no flagged videos (older snapshots lack the bucket).
+export async function incompleteIdsForChannel(
+ slug: string,
+ paths: Paths = getPaths(),
+): Promise<string[]> {
+ const snap = await readChannelSnapshot(paths, slug);
+ const ids = snap?.buckets?.incompleteTranscript;
+ return Array.isArray(ids) ? ids : [];
+}
+
+// Re-download the full audio and re-transcribe one video in place. The new
+// transcript overwrites transcript.json (transcribeWithWorker) and normalize
+// regenerates transcript.cues.json, so there is never a window with no
+// transcript. Throws on failure so the batch loop can record it per-id.
+export async function fixIncompleteTranscriptOne(opts: {
+ slug: string;
+ videoId: string;
+ config: ChannelConfig;
+ paths: Paths;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+ tracker?: TaskTracker;
+}): Promise<void> {
+ const { slug, videoId, config, paths, onLog, signal, tracker } = opts;
+ const videoDir = videoDirOf(paths, slug, videoId);
+ const audioFormat = config.audioFormat ?? "mp3";
+ const url = await findVideoSourceUrl(paths, slug, videoId, config);
+ if (!url) {
+ throw new Error(
+ `Could not determine the video URL for ${videoId}: no metadata.info.json and the playlist does not contain a matching entry.`,
+ );
+ }
+ // Remove the truncated audio so the download re-fetches the full file rather
+ // than seeing it as already present.
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ for (const name of entries.filter(isRealAudioFile)) {
+ await rm(path.join(videoDir, name), { force: true });
+ onLog(`Removed truncated audio ${name}.`);
+ }
+ onLog(`Re-downloading audio for ${videoId}…`);
+ await runYtdlp({
+ channelSlug: slug,
+ mode: "download-one-audio",
+ channelConfig: config,
+ paths,
+ onLog,
+ signal,
+ singleVideoUrl: url,
+ audioFormatOverride: audioFormat,
+ });
+ await transcribeWithWorker({
+ paths,
+ videoDir,
+ videoId,
+ audioFilename: `audio.${audioFormat}`,
+ tracker,
+ onLog,
+ signal,
+ });
+}
+
+// Clear a truncated transcript so the video drops back into the normal pending
+// pipeline: delete every real audio file, the whisper transcript.json + derived
+// transcript.cues.json, and any transcript VTT track. With no remaining artifact
+// the snapshot re-buckets it as undownloaded → (after re-download)
+// downloadedNoTranscript, where auto-download/auto-transcribe (or the manual
+// "Download missing" / "Transcribe pending" actions) reprocess it. Keeps
+// metadata.info.json (needed to resolve the URL on re-download).
+export async function clearIncompleteTranscriptOne(opts: {
+ slug: string;
+ videoId: string;
+ paths: Paths;
+}): Promise<{ removed: number }> {
+ const { slug, videoId, paths } = opts;
+ const videoDir = videoDirOf(paths, slug, videoId);
+ const dataDir = path.resolve(paths.channelsDir, slug, "data");
+ const resolved = path.resolve(videoDir);
+ // Belt-and-suspenders: never delete outside the channel's data dir.
+ if (path.dirname(resolved) !== dataDir) {
+ throw new Error(
+ `Refusing to clear: video path resolved outside the data dir (${videoId})`,
+ );
+ }
+ const entries = await readdir(resolved).catch(() => [] as string[]);
+ const toRemove = entries.filter(
+ (name) =>
+ isRealAudioFile(name) ||
+ isTranscriptVtt(name) ||
+ name === "transcript.json" ||
+ name === "transcript.cues.json",
+ );
+ for (const name of toRemove) {
+ await rm(path.join(resolved, name), { force: true });
+ }
+ return { removed: toRemove.length };
+}
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -38,6 +38,7 @@ import {
} from "yt-dlp-transcript-common/jobs/streamCommand";
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks";
+import { fixIncompleteTranscriptOne } from "../../lib/fixIncompleteTranscript";
function videoQueueKey(config: ChannelConfig, override: string | undefined): string {
return resolveQueueKey(downloadQueueKey(config), override);
@@ -318,7 +319,6 @@ export async function redownloadIncompleteTranscriptAction(
const r = await loadConfigOrError(slug);
if (!r.ok) return r;
const paths = getPaths();
- const videoDir = videoDirOf(slug, videoId);
return runManagedFunction({
kind: "whisper-video",
queueKey: videoQueueKey(r.config, queueKey),
@@ -326,39 +326,14 @@ export async function redownloadIncompleteTranscriptAction(
channelSlug: slug,
videoId,
fn: async (onLog, signal, _setProgress, ctx) => {
- const audioFormat = r.config.audioFormat ?? "mp3";
- const url = await findVideoSourceUrl(paths, slug, videoId, r.config);
- if (!url) {
- throw new Error(
- "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.",
- );
- }
- // Remove the truncated audio so the download below re-fetches the full
- // file rather than seeing it as already present.
- const entries = await readdir(videoDir).catch(() => [] as string[]);
- for (const name of entries.filter(isRealAudioFile)) {
- await rm(path.join(videoDir, name), { force: true });
- onLog(`Removed truncated audio ${name}.`);
- }
- onLog(`Re-downloading audio for ${videoId}…`);
- await runYtdlp({
- channelSlug: slug,
- mode: "download-one-audio",
- channelConfig: r.config,
+ await fixIncompleteTranscriptOne({
+ slug,
+ videoId,
+ config: r.config,
paths,
onLog,
signal,
- singleVideoUrl: url,
- audioFormatOverride: audioFormat,
- });
- await transcribeWithWorker({
- paths,
- videoDir,
- videoId,
- audioFilename: `audio.${audioFormat}`,
tracker: makeTaskTracker(ctx, onLog),
- onLog,
- signal,
});
revalidatePath(`/channels/${slug}/videos/${videoId}`);
revalidatePath(`/channels/${slug}`);
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -34,6 +34,7 @@ import {
transcribeMissingAction,
} from "../channels/[slug]/whisperActions";
import { persistKeptAction } from "../channels/[slug]/persistActions";
+import { redownloadIncompleteBucketAction } from "../channels/[slug]/incompleteTranscriptActions";
export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>;
@@ -94,6 +95,15 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
spec.bucket,
);
},
+ "redownload-incomplete-bucket": async (spec) => {
+ const { queueKey } = params(spec);
+ if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." };
+ const ids = await idsForBucket(spec.slug, spec.bucket);
+ if (ids.length === 0) {
+ return { ok: false, error: "No incomplete transcripts right now.", info: true };
+ }
+ return redownloadIncompleteBucketAction(spec.slug, ids, queueKey);
+ },
"retry-bucket": async (spec) => {
const { p, queueKey } = params(spec);
if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." };