Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit dffe8bb8cf2f045e78d39196ab4f7bdc9b143ba1
parent d1c783cde9dc61ae5af5a588b517a4d669f612b2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 29 Jun 2026 22:07:34 -0400

Bulk-fix incomplete transcripts: shared helpers, actions, job plumbing

Add a shared per-video module (fixIncompleteTranscript.ts) as the single
source of truth for re-fixing/clearing a truncated transcript, and refactor
the existing per-video redownloadIncompleteTranscriptAction onto it.

New channel-level actions (incompleteTranscriptActions.ts):
- redownloadIncompleteBucketAction: one managed batch job that removes the
  truncated audio, re-downloads, and re-transcribes each flagged video in
  place (new bookmarkable kind redownload-incomplete-bucket).
- clearIncompleteTranscriptsAction: synchronously deletes the truncated audio
  + transcript so videos fall back into the undownloaded -> needs-transcript
  pipeline, then enables + starts the auto-download/auto-transcribe runners.

Bulk-bar wrappers (bulkVideoActions.ts) and global all-channels actions
(actionable/actions.ts) reuse the same code. Register the new job kind and
the incompleteTranscript replay bucket.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

Diffstat:
Mcommon/jobs/jobKinds.ts | 7+++++++
Mcommon/jobs/jobSpec.ts | 4+++-
Meditor/app/actionable/actions.ts | 56++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/bulkVideoActions.ts | 25+++++++++++++++++++++++++
Aeditor/app/channels/[slug]/incompleteTranscriptActions.ts | 152+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Aeditor/app/channels/[slug]/lib/fixIncompleteTranscript.ts | 126+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 35+++++------------------------------
Meditor/app/jobs/jobReplayRegistry.ts | 10++++++++++
8 files changed, 384 insertions(+), 31 deletions(-)

diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts @@ -77,6 +77,13 @@ const JOB_KINDS: Record<string, JobKindMeta> = { bookmarkable: true, queueKeyStrategy: "custom", }, + "redownload-incomplete-bucket": { + kind: "redownload-incomplete-bucket", + label: "Re-download truncated transcripts", + drainable: true, + bookmarkable: true, + queueKeyStrategy: "custom", + }, "download-from-playlist": { kind: "download-from-playlist", label: "Download from playlist", diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts @@ -18,7 +18,8 @@ export type ReplayBucket = | "partialDownloads" | "noTranscript" - | "downloadedNoTranscript"; + | "downloadedNoTranscript" + | "incompleteTranscript"; export type JobSpec = { kind: string; @@ -36,6 +37,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([ "partialDownloads", "noTranscript", "downloadedNoTranscript", + "incompleteTranscript", ]); // Defensive parse for a spec read back from JSON (a sidecar or the bookmarks diff --git a/editor/app/actionable/actions.ts b/editor/app/actionable/actions.ts @@ -11,6 +11,12 @@ import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { runManagedFunction } from "yt-dlp-transcript-common/jobs/streamCommand"; import { drainStream } from "yt-dlp-transcript-common/jobs/drainStream"; import { detectDuplicateShorts } from "yt-dlp-transcript-common/controller/duplicateShorts"; +import { + clearIncompleteTranscriptsAction, + enableAutoRunners, + redownloadIncompleteBucketAction, +} from "../channels/[slug]/incompleteTranscriptActions"; +import { loadActionableSummary } from "./lib/loadActionable"; export type RefreshAllResult = { queued: string[]; @@ -123,3 +129,53 @@ export async function runDuplicateDetectionAction( revalidatePath("/actionable"); return { ok: true, clusters, videosInClusters }; } + +export type GlobalIncompleteResult = + | { ok: true; channels: number; affected: number } + | { ok: false; error: string }; + +// All channels with at least one truncated transcript right now. +async function affectedIncompleteSlugs(): Promise<string[]> { + const summary = await loadActionableSummary(getPaths()); + return summary.incompleteTranscripts.map((r) => r.channel.slug); +} + +// Clear & re-queue every truncated transcript across all channels, then enable +// the auto-runners once. Destructive — the caller confirms first. +export async function clearAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { + const slugs = await affectedIncompleteSlugs(); + if (slugs.length === 0) { + return { ok: false, error: "No incomplete transcripts to clear." }; + } + let cleared = 0; + for (const slug of slugs) { + // Defer enabling the runners until the end so settings flip only once. + const r = await clearIncompleteTranscriptsAction(slug, undefined, { + enableRunners: false, + }); + cleared += r.succeeded; + } + await enableAutoRunners(); + revalidatePath("/actionable"); + return { ok: true, channels: slugs.length, affected: cleared }; +} + +// Queue one batch re-fix job per affected channel (fire-and-forget — the jobs +// keep running and surface in /jobs; cancel each stream so we don't hold them +// open). +export async function redownloadAllIncompleteTranscriptsAction(): Promise<GlobalIncompleteResult> { + const slugs = await affectedIncompleteSlugs(); + if (slugs.length === 0) { + return { ok: false, error: "No incomplete transcripts to fix." }; + } + let queued = 0; + for (const slug of slugs) { + const result = await redownloadIncompleteBucketAction(slug); + if (result.ok) { + void result.stream.cancel(); + queued++; + } + } + revalidatePath("/actionable"); + return { ok: true, channels: slugs.length, affected: queued }; +} diff --git a/editor/app/channels/[slug]/bulkVideoActions.ts b/editor/app/channels/[slug]/bulkVideoActions.ts @@ -16,6 +16,10 @@ import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels" import { transcribeBucketAction } from "./whisperActions"; import { retryBucketAction } from "./pipelineActions"; import { + clearIncompleteTranscriptsAction, + redownloadIncompleteBucketAction, +} from "./incompleteTranscriptActions"; +import { deleteOneVideoDir, markVideoUntranscribableAction, removeAudioFilesForVideo, @@ -50,6 +54,27 @@ export async function bulkRetryDownloadAction( return retryBucketAction(slug, videoIds, queueKey, abortOnError); } +// Bulk re-fix truncated transcripts: queues ONE batch job that re-downloads + +// re-transcribes each selected video in place (same one-job-per-bulk convention +// as bulkTranscribeAction). +export async function bulkRedownloadIncompleteAction( + slug: string, + videoIds: string[], + queueKey?: string, +): Promise<StreamActionResult> { + return redownloadIncompleteBucketAction(slug, videoIds, queueKey); +} + +// Bulk clear truncated transcripts: synchronous fs op (delete audio + transcript) +// that enables the auto-runners, so it reports a per-id summary like the other +// clear/remove bulk actions. Destructive — the caller confirms first. +export async function bulkClearIncompleteAction( + slug: string, + videoIds: string[], +): Promise<BulkActionSummary> { + return clearIncompleteTranscriptsAction(slug, videoIds); +} + // Marking videos untranscribable is an instant metadata write, not a queued // batch, so it stays synchronous and reports a per-id summary. export async function bulkMarkUntranscribableAction( diff --git a/editor/app/channels/[slug]/incompleteTranscriptActions.ts b/editor/app/channels/[slug]/incompleteTranscriptActions.ts @@ -0,0 +1,152 @@ +"use server"; + +import { revalidatePath } from "next/cache"; +import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { + TRANSCRIPTION_QUEUE, + resolveQueueKey, +} from "yt-dlp-transcript-common/lib/queueKeys"; +import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings"; +import { startAutoRunner } from "yt-dlp-transcript-common/controller/autoRunner"; +import { + runManagedFunction, + type StreamActionResult, +} from "yt-dlp-transcript-common/jobs/streamCommand"; +import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; +import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; +import { + clearIncompleteTranscriptOne, + fixIncompleteTranscriptOne, + incompleteIdsForChannel, +} from "./lib/fixIncompleteTranscript"; +import type { BulkActionSummary } from "./bulkVideoActions"; + +function dedupeIds(ids: string[]): string[] { + return Array.from(new Set(ids.map((id) => id.trim()).filter(Boolean))); +} + +// Turn on + start both auto-queue runners so cleared videos get reprocessed +// automatically. Enabling persists across restarts (settings.json); starting +// brings the runner up now without a server restart (same path the auto-queue +// admin page and /api/auto-queue/control use). NOTE: the runner only picks up a +// channel its policy tree actually matches — cleared videos also surface in the +// manual "Download missing" / "Transcribe pending" actionable sections as a +// fallback. +export async function enableAutoRunners(): Promise<void> { + const current = getSettings(); + if ( + !current.autoQueue.transcription.enabled || + !current.autoQueue.download.enabled + ) { + await writeSettings({ + ...current, + autoQueue: { + ...current.autoQueue, + transcription: { ...current.autoQueue.transcription, enabled: true }, + download: { ...current.autoQueue.download, enabled: true }, + }, + }); + } + await startAutoRunner("transcription"); + await startAutoRunner("download"); +} + +// One-shot batch re-fix: queue a single managed job that removes the truncated +// audio, re-downloads, and re-transcribes each flagged video in place (the bulk +// version of the per-video redownloadIncompleteTranscriptAction). ids default to +// the channel's current incompleteTranscript bucket; bookmarkable so a re-run +// re-derives the live bucket. +export async function redownloadIncompleteBucketAction( + slug: string, + ids?: string[], + queueKey?: string, +): Promise<StreamActionResult> { + const paths = getPaths(); + const config = await readChannelConfig(paths, slug); + if (!config) return { ok: false, error: `Channel "${slug}" not found` }; + const source = ids && ids.length ? ids : await incompleteIdsForChannel(slug, paths); + const cleaned = dedupeIds(source); + if (cleaned.length === 0) { + return { ok: false, error: "No incomplete transcripts to fix.", info: true }; + } + return runManagedFunction({ + kind: "redownload-incomplete-bucket", + queueKey: resolveQueueKey(TRANSCRIPTION_QUEUE, queueKey), + paths, + channelSlug: slug, + spec: { + kind: "redownload-incomplete-bucket", + slug, + bucket: "incompleteTranscript", + params: { queueKey }, + }, + fn: async (onLog, signal, _setProgress, ctx) => { + const tracker = makeTaskTracker(ctx, onLog); + let succeeded = 0; + let failed = 0; + for (const id of cleaned) { + if (ctx.drainSignal?.aborted) { + onLog(`Drain requested; stopping before ${id}.`); + break; + } + try { + onLog(`Re-downloading & re-transcribing ${id}…`); + await fixIncompleteTranscriptOne({ + slug, + videoId: id, + config, + paths, + onLog, + signal, + tracker, + }); + succeeded++; + } catch (e) { + failed++; + onLog(`Failed ${id}: ${(e as Error).message}`); + } + } + onLog( + `Re-download incomplete: ${succeeded} fixed, ${failed} failed of ${cleaned.length}.`, + ); + revalidatePath(`/channels/${slug}`); + }, + }); +} + +// Clear & release: synchronously delete the truncated audio + transcript for the +// flagged videos so they fall back into the normal pending pipeline, then enable +// the auto-runners so they reprocess automatically. Destructive — gate every +// entry point with a confirm. ids default to the channel's incompleteTranscript +// bucket. +export async function clearIncompleteTranscriptsAction( + slug: string, + ids?: string[], + opts?: { enableRunners?: boolean }, +): Promise<BulkActionSummary> { + const paths = getPaths(); + const source = ids && ids.length ? ids : await incompleteIdsForChannel(slug, paths); + const cleaned = dedupeIds(source); + const failures: BulkActionSummary["failures"] = []; + let succeeded = 0; + for (const id of cleaned) { + try { + await clearIncompleteTranscriptOne({ slug, videoId: id, paths }); + succeeded++; + } catch (e) { + failures.push({ videoId: id, error: (e as Error).message }); + } + } + if (opts?.enableRunners !== false && cleaned.length > 0) { + await enableAutoRunners(); + } + revalidatePath(`/channels/${slug}`); + requestChannelSnapshot(paths, slug); + return { + ok: failures.length === 0, + attempted: cleaned.length, + succeeded, + failures, + }; +} diff --git a/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts b/editor/app/channels/[slug]/lib/fixIncompleteTranscript.ts @@ -0,0 +1,126 @@ +// Shared per-video building blocks for fixing truncated ("incomplete") +// transcripts, reused by the per-video action, the channel-level batch job, the +// bulk-bar wrappers, and the global all-channels actions. Server-only (it does +// fs + spawns yt-dlp/whisper) but NOT a "use server" module — it exports plain +// helpers that take onLog/signal/tracker, which aren't serializable across the +// server-action boundary. +// +// A truncated transcript means the audio download silently stopped early: the +// audio on disk is itself short, so re-running whisper on it just reproduces the +// short transcript. The fix MUST re-fetch the audio first. + +import path from "node:path"; +import { readdir, rm } from "node:fs/promises"; +import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; +import { getPaths, type Paths } from "yt-dlp-transcript-common/lib/paths"; +import { + isRealAudioFile, + isTranscriptVtt, +} from "yt-dlp-transcript-common/lib/videoStatus"; +import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transcribeOne"; +import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos"; +import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; +import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; +import type { TaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; + +function videoDirOf(paths: Paths, slug: string, videoId: string): string { + return path.join(paths.channelsDir, slug, "data", videoId); +} + +// The current snapshot's truncated-transcript ids for a channel. Empty when the +// snapshot is missing or has no flagged videos (older snapshots lack the bucket). +export async function incompleteIdsForChannel( + slug: string, + paths: Paths = getPaths(), +): Promise<string[]> { + const snap = await readChannelSnapshot(paths, slug); + const ids = snap?.buckets?.incompleteTranscript; + return Array.isArray(ids) ? ids : []; +} + +// Re-download the full audio and re-transcribe one video in place. The new +// transcript overwrites transcript.json (transcribeWithWorker) and normalize +// regenerates transcript.cues.json, so there is never a window with no +// transcript. Throws on failure so the batch loop can record it per-id. +export async function fixIncompleteTranscriptOne(opts: { + slug: string; + videoId: string; + config: ChannelConfig; + paths: Paths; + onLog: (line: string) => void; + signal: AbortSignal; + tracker?: TaskTracker; +}): Promise<void> { + const { slug, videoId, config, paths, onLog, signal, tracker } = opts; + const videoDir = videoDirOf(paths, slug, videoId); + const audioFormat = config.audioFormat ?? "mp3"; + const url = await findVideoSourceUrl(paths, slug, videoId, config); + if (!url) { + throw new Error( + `Could not determine the video URL for ${videoId}: no metadata.info.json and the playlist does not contain a matching entry.`, + ); + } + // Remove the truncated audio so the download re-fetches the full file rather + // than seeing it as already present. + const entries = await readdir(videoDir).catch(() => [] as string[]); + for (const name of entries.filter(isRealAudioFile)) { + await rm(path.join(videoDir, name), { force: true }); + onLog(`Removed truncated audio ${name}.`); + } + onLog(`Re-downloading audio for ${videoId}…`); + await runYtdlp({ + channelSlug: slug, + mode: "download-one-audio", + channelConfig: config, + paths, + onLog, + signal, + singleVideoUrl: url, + audioFormatOverride: audioFormat, + }); + await transcribeWithWorker({ + paths, + videoDir, + videoId, + audioFilename: `audio.${audioFormat}`, + tracker, + onLog, + signal, + }); +} + +// Clear a truncated transcript so the video drops back into the normal pending +// pipeline: delete every real audio file, the whisper transcript.json + derived +// transcript.cues.json, and any transcript VTT track. With no remaining artifact +// the snapshot re-buckets it as undownloaded → (after re-download) +// downloadedNoTranscript, where auto-download/auto-transcribe (or the manual +// "Download missing" / "Transcribe pending" actions) reprocess it. Keeps +// metadata.info.json (needed to resolve the URL on re-download). +export async function clearIncompleteTranscriptOne(opts: { + slug: string; + videoId: string; + paths: Paths; +}): Promise<{ removed: number }> { + const { slug, videoId, paths } = opts; + const videoDir = videoDirOf(paths, slug, videoId); + const dataDir = path.resolve(paths.channelsDir, slug, "data"); + const resolved = path.resolve(videoDir); + // Belt-and-suspenders: never delete outside the channel's data dir. + if (path.dirname(resolved) !== dataDir) { + throw new Error( + `Refusing to clear: video path resolved outside the data dir (${videoId})`, + ); + } + const entries = await readdir(resolved).catch(() => [] as string[]); + const toRemove = entries.filter( + (name) => + isRealAudioFile(name) || + isTranscriptVtt(name) || + name === "transcript.json" || + name === "transcript.cues.json", + ); + for (const name of toRemove) { + await rm(path.join(resolved, name), { force: true }); + } + return { removed: toRemove.length }; +} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -38,6 +38,7 @@ import { } from "yt-dlp-transcript-common/jobs/streamCommand"; import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler"; import { makeTaskTracker } from "yt-dlp-transcript-common/jobs/taskHooks"; +import { fixIncompleteTranscriptOne } from "../../lib/fixIncompleteTranscript"; function videoQueueKey(config: ChannelConfig, override: string | undefined): string { return resolveQueueKey(downloadQueueKey(config), override); @@ -318,7 +319,6 @@ export async function redownloadIncompleteTranscriptAction( const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); - const videoDir = videoDirOf(slug, videoId); return runManagedFunction({ kind: "whisper-video", queueKey: videoQueueKey(r.config, queueKey), @@ -326,39 +326,14 @@ export async function redownloadIncompleteTranscriptAction( channelSlug: slug, videoId, fn: async (onLog, signal, _setProgress, ctx) => { - const audioFormat = r.config.audioFormat ?? "mp3"; - const url = await findVideoSourceUrl(paths, slug, videoId, r.config); - if (!url) { - throw new Error( - "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", - ); - } - // Remove the truncated audio so the download below re-fetches the full - // file rather than seeing it as already present. - const entries = await readdir(videoDir).catch(() => [] as string[]); - for (const name of entries.filter(isRealAudioFile)) { - await rm(path.join(videoDir, name), { force: true }); - onLog(`Removed truncated audio ${name}.`); - } - onLog(`Re-downloading audio for ${videoId}…`); - await runYtdlp({ - channelSlug: slug, - mode: "download-one-audio", - channelConfig: r.config, + await fixIncompleteTranscriptOne({ + slug, + videoId, + config: r.config, paths, onLog, signal, - singleVideoUrl: url, - audioFormatOverride: audioFormat, - }); - await transcribeWithWorker({ - paths, - videoDir, - videoId, - audioFilename: `audio.${audioFormat}`, tracker: makeTaskTracker(ctx, onLog), - onLog, - signal, }); revalidatePath(`/channels/${slug}/videos/${videoId}`); revalidatePath(`/channels/${slug}`); diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts @@ -34,6 +34,7 @@ import { transcribeMissingAction, } from "../channels/[slug]/whisperActions"; import { persistKeptAction } from "../channels/[slug]/persistActions"; +import { redownloadIncompleteBucketAction } from "../channels/[slug]/incompleteTranscriptActions"; export type ReplayHandler = (spec: JobSpec) => Promise<StreamActionResult>; @@ -94,6 +95,15 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = { spec.bucket, ); }, + "redownload-incomplete-bucket": async (spec) => { + const { queueKey } = params(spec); + if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." }; + const ids = await idsForBucket(spec.slug, spec.bucket); + if (ids.length === 0) { + return { ok: false, error: "No incomplete transcripts right now.", info: true }; + } + return redownloadIncompleteBucketAction(spec.slug, ids, queueKey); + }, "retry-bucket": async (spec) => { const { p, queueKey } = params(spec); if (!spec.bucket) return { ok: false, error: "Bookmark is missing its bucket." };