Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 711021a8a43aac758234ef52c878889979ae8c89
parent 77c5beda232d35a48e31f46e2196fb367ba0f8c4
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  5 May 2026 18:59:53 -0400

hybrid channel improvements

Diffstat:
Mcommon/controller/buildIndex.ts | 35+++++++++++++++++++++++++----------
Acommon/controller/channelHealth.ts | 92+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/channels.ts | 24+++++++++---------------
Mcommon/controller/failedTranscriptions.ts | 18++++++++++++++++++
Mcommon/controller/transcribeOne.ts | 59+++++++++++++++++++++++++++++++++++++++--------------------
Mcommon/controller/verifyTranscripts.ts | 66++++++++++++++++++++++--------------------------------------------
Mcommon/lib/settings.ts | 82+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------
Acommon/lib/videoStatus.ts | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/runYtdlp.ts | 27++++++++++++---------------
Meditor/app/channels/[slug]/_components/WhisperPanel.tsx | 2+-
Meditor/app/channels/[slug]/page.tsx | 149++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-------------
Meditor/app/channels/[slug]/videos/[id]/_components/VideoPanel.tsx | 2+-
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 6------
Meditor/app/settings/_components/SettingsForm.tsx | 40++++++++++++++++++++++++++++++++++++++++
Meditor/app/settings/actions.ts | 17+++++++++++++++++
Meditor/app/settings/page.tsx | 4+++-
Meditor/e2e/dashboard.spec.ts | 7++++---
Meditor/e2e/video-page.spec.ts | 4+++-
Meditor/e2e/whisper.spec.ts | 8+++++---
19 files changed, 548 insertions(+), 152 deletions(-)

diff --git a/common/controller/buildIndex.ts b/common/controller/buildIndex.ts @@ -53,6 +53,11 @@ import { type ChannelHandling, } from "../lib/channelConfig"; import type { Paths } from "../lib/paths"; +import { + pickIndexTranscript, + readVideoFiles, + type IndexTranscript, +} from "../lib/videoStatus"; const SCHEMA_VERSION = 5; @@ -86,6 +91,7 @@ type LiveEntry = { metaMs: number; transcriptPath: string; transcriptMs: number | null; + transcriptKind: IndexTranscript["kind"] | null; }; async function exists(p: string): Promise<boolean> { @@ -140,24 +146,32 @@ async function scanSource( } catch { continue; } - const transcriptName = - cfg.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; for (const v of videoEntries) { if (!v.isDirectory()) continue; const videoDir = v.name; - const metaPath = path.join(dataDir, videoDir, "metadata.info.json"); - const transcriptPath = path.join(dataDir, videoDir, transcriptName); + const fullVideoDir = path.join(dataDir, videoDir); + const metaPath = path.join(fullVideoDir, "metadata.info.json"); let metaMs: number; try { metaMs = (await stat(metaPath)).mtimeMs; } catch { continue; } + // Hybrid: a single channel may contain both YouTube auto-subs (.vtt) + // and whisper-generated transcript.json. Prefer whisper when both + // exist; fall back to .vtt otherwise. + const files = await readVideoFiles(fullVideoDir); + const picked = pickIndexTranscript(files); + const transcriptPath = picked + ? path.join(fullVideoDir, picked.filename) + : path.join(fullVideoDir, "transcript.json"); let transcriptMs: number | null = null; - try { - transcriptMs = (await stat(transcriptPath)).mtimeMs; - } catch { - transcriptMs = null; + if (picked) { + try { + transcriptMs = (await stat(transcriptPath)).mtimeMs; + } catch { + transcriptMs = null; + } } live.push({ channelSlug: ch.name, @@ -168,6 +182,7 @@ async function scanSource( metaMs, transcriptPath, transcriptMs, + transcriptKind: picked?.kind ?? null, }); } } @@ -412,11 +427,11 @@ export async function buildIndex({ summary.id, ]; let cueList: Cue[] | undefined; - if (s.transcriptMs !== null) { + if (s.transcriptMs !== null && s.transcriptKind) { try { const raw = await readFile(s.transcriptPath, "utf8"); cueList = - s.handling === "youtube" ? parseVtt(raw) : parseWhisper(raw); + s.transcriptKind === "vtt" ? parseVtt(raw) : parseWhisper(raw); } catch { cueList = undefined; } diff --git a/common/controller/channelHealth.ts b/common/controller/channelHealth.ts @@ -0,0 +1,92 @@ +import path from "node:path"; +import { readdir } from "node:fs/promises"; +import type { Paths } from "../lib/paths"; +import { readArchive } from "../lib/archive"; +import { + isVideoDownloaded, + isVideoTranscribed, + readVideoFiles, +} from "../lib/videoStatus"; +import { loadFailedTranscriptions } from "./failedTranscriptions"; +import { findDuplicateDirs } from "./verifyTranscripts"; + +export type ChannelHealth = { + totals: { + videos: number; + transcribed: number; + downloaded: number; + }; + noTranscript: string[]; + downloadedNoTranscript: string[]; + untranscribable: string[]; + noMetadata: string[]; + failedListed: string[]; + missingFromArchive: string[]; + duplicateDirs: string[]; +}; + +export async function inspectChannelHealth( + paths: Paths, + slug: string, +): Promise<ChannelHealth> { + const channelDir = path.join(paths.channelsDir, slug); + const dataDir = path.join(channelDir, "data"); + const archivePath = path.join(channelDir, "archive"); + + const dirs = await readdir(dataDir, { withFileTypes: true }).catch(() => []); + const videoDirNames = dirs.filter((d) => d.isDirectory()).map((d) => d.name); + + const perVideo = await Promise.all( + videoDirNames.map(async (name) => ({ + id: name, + files: await readVideoFiles(path.join(dataDir, name)), + })), + ); + + const noTranscript: string[] = []; + const downloadedNoTranscript: string[] = []; + const untranscribable: string[] = []; + const noMetadata: string[] = []; + let transcribed = 0; + let downloaded = 0; + for (const v of perVideo) { + if (isVideoTranscribed(v.files)) transcribed++; + if (isVideoDownloaded(v.files)) downloaded++; + if (!v.files.hasMeta) noMetadata.push(v.id); + if (v.files.isUntranscribable) { + untranscribable.push(v.id); + continue; + } + if (!isVideoTranscribed(v.files)) { + if (v.files.audioFiles.length > 0) { + downloadedNoTranscript.push(v.id); + } else { + noTranscript.push(v.id); + } + } + } + + const failedListed = await loadFailedTranscriptions(paths, slug); + + const archive = await readArchive(archivePath); + const { dirMap, duplicates } = await findDuplicateDirs(dataDir); + const missingFromArchive: string[] = []; + for (const id of archive.ids) { + if (!dirMap.has(id)) missingFromArchive.push(id); + } + + return { + totals: { + videos: videoDirNames.length, + transcribed, + downloaded, + }, + noTranscript: noTranscript.sort(), + downloadedNoTranscript: downloadedNoTranscript.sort(), + untranscribable: untranscribable.sort(), + noMetadata: noMetadata.sort(), + failedListed, + missingFromArchive: missingFromArchive.sort(), + duplicateDirs: Array.from(duplicates).sort(), + }; +} diff --git a/common/controller/channels.ts b/common/controller/channels.ts @@ -4,9 +4,13 @@ import type { Dirent } from "node:fs"; import { parseChannelConfig, type ChannelConfig, - type ChannelHandling, } from "../lib/channelConfig"; import type { Paths } from "../lib/paths"; +import { + isVideoDownloaded, + isVideoTranscribed, + readVideoFiles, +} from "../lib/videoStatus"; export type ChannelStat = { slug: string; @@ -28,7 +32,6 @@ async function exists(p: string): Promise<boolean> { async function countDataFiles( dataDir: string, - handling: ChannelHandling, ): Promise<{ videos: number; transcripts: number; downloads: number }> { let dirs: Dirent[]; try { @@ -39,19 +42,10 @@ async function countDataFiles( const videoDirs = dirs.filter((d) => d.isDirectory()); const flags = await Promise.all( videoDirs.map(async (d) => { - const videoDir = path.join(dataDir, d.name); - if (handling === "youtube") { - // For YouTube channels the destination yt-dlp produces is the auto-sub - // .vtt file — same file counts as both "downloaded" and "transcript". - const has = await exists(path.join(videoDir, "transcript.en.vtt")); - return { transcript: has, download: has }; - } - // Transcribe channels: audio.* is the yt-dlp download, transcript.json - // is the whisper output. They diverge until whisper has run. - const entries = await readdir(videoDir).catch(() => [] as string[]); + const files = await readVideoFiles(path.join(dataDir, d.name)); return { - transcript: entries.includes("transcript.json"), - download: entries.some((e) => e.startsWith("audio.")), + transcript: isVideoTranscribed(files), + download: isVideoDownloaded(files), }; }), ); @@ -115,7 +109,7 @@ export async function listChannels(paths: Paths): Promise<ChannelStat[]> { if (!config) continue; const channelDir = path.join(paths.channelsDir, slug); const dataDir = path.join(channelDir, "data"); - const counts = await countDataFiles(dataDir, config.handling); + const counts = await countDataFiles(dataDir); out.push({ slug, config, diff --git a/common/controller/failedTranscriptions.ts b/common/controller/failedTranscriptions.ts @@ -1,4 +1,22 @@ +import path from "node:path"; import { readFile, rename, writeFile } from "node:fs/promises"; +import type { Paths } from "../lib/paths"; + +export function failedTranscriptionsFile(paths: Paths, slug: string): string { + return path.join(paths.channelsDir, slug, "failed-transcriptions"); +} + +export async function loadFailedTranscriptions( + paths: Paths, + slug: string, +): Promise<string[]> { + try { + const raw = await readFile(failedTranscriptionsFile(paths, slug), "utf8"); + return raw.split("\n").filter(Boolean); + } catch { + return []; + } +} export async function pruneFailedTranscriptions( failureListFile: string, diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts @@ -2,6 +2,13 @@ import path from "node:path"; import fs from "fs-extra"; import { execa } from "execa"; import type { Paths } from "../lib/paths"; +import { + TRANSCRIBE_PLACEHOLDER_AUDIO, + TRANSCRIBE_PLACEHOLDER_MODEL, + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_KNOWN_PLACEHOLDERS, + getSettings, +} from "../lib/settings"; const { pathExists, readdir, rename } = fs; @@ -21,6 +28,25 @@ async function resolveAudioFile( return [...candidates].sort()[0]; } +function substitutePlaceholders( + args: string[], + values: { audioFile: string; outputBase: string; model: string }, +): string[] { + return args.map((arg) => { + return arg.replace(/\{[^}]+\}/g, (token) => { + if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { + throw new Error( + `Unknown placeholder ${token} in transcribeArgs; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`, + ); + } + if (token === TRANSCRIBE_PLACEHOLDER_AUDIO) return values.audioFile; + if (token === TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE) return values.outputBase; + if (token === TRANSCRIBE_PLACEHOLDER_MODEL) return values.model; + return token; + }); + }); +} + export type TranscribeOneOptions = { paths: Paths; videoDir: string; @@ -51,30 +77,23 @@ export async function transcribeOneVideo( log(`Using ${resolvedAudio} instead of ${opts.audioFilename}`); } + const settings = getSettings(); log(`Transcribe ${opts.videoId} start (${resolvedAudio})`); const start = Date.now(); - // Whisper writes "transcript.json" relative to cwd. Use a tmp name and + // Whisper writes "<outputBase>.json" relative to cwd. Use a tmp name and // rename on success so a SIGTERM mid-write can't leave a half-baked file. const tmpBase = `transcript.tmp-${process.pid}`; - const child = execa( - opts.paths.whisperBin, - [ - "-ojf", - "-l", - "en", - "-m", - opts.paths.whisperModel, - "-of", - tmpBase, - resolvedAudio, - ], - { - cwd: opts.videoDir, - cancelSignal: opts.signal, - all: true, - buffer: false, - }, - ); + const args = substitutePlaceholders(settings.transcribeArgs, { + audioFile: resolvedAudio, + outputBase: tmpBase, + model: settings.transcribeModel, + }); + const child = execa(settings.transcribeBin, args, { + cwd: opts.videoDir, + cancelSignal: opts.signal, + all: true, + buffer: false, + }); child.all?.on("data", (c: Buffer) => log(c.toString("utf8"))); await child; const tmpPath = path.join(opts.videoDir, `${tmpBase}.json`); diff --git a/common/controller/verifyTranscripts.ts b/common/controller/verifyTranscripts.ts @@ -1,11 +1,8 @@ import path from "node:path"; -import { readdir, readFile, stat } from "node:fs/promises"; +import { readdir } from "node:fs/promises"; import type { Paths } from "../lib/paths"; -import { - parseChannelConfig, - type ChannelConfig, -} from "../lib/channelConfig"; import { readArchive } from "../lib/archive"; +import { isVideoTranscribed, readVideoFiles } from "../lib/videoStatus"; export type VerifyTranscriptsOptions = { channelSlug: string; @@ -17,45 +14,11 @@ export type VerifyTranscriptsResult = { missing: string[]; }; -async function readChannelConfig( - channelDir: string, -): Promise<ChannelConfig | null> { - try { - const raw = await readFile(path.join(channelDir, "config.json"), "utf8"); - return parseChannelConfig(JSON.parse(raw)); - } catch { - return null; - } -} - -async function exists(p: string): Promise<boolean> { - try { - await stat(p); - return true; - } catch { - return false; - } -} - -// For each ID in the archive, confirm there's a matching <id>/transcript.<ext> -// on disk. Reports missing transcripts and any duplicate video directories. -export async function verifyTranscripts({ - channelSlug, - paths, -}: VerifyTranscriptsOptions): Promise<VerifyTranscriptsResult> { - const channelDir = path.join(paths.channelsDir, channelSlug); - const dataDir = path.join(channelDir, "data"); - const archivePath = path.join(channelDir, "archive"); - - const config = await readChannelConfig(channelDir); - const transcriptFilename = - config?.handling === "youtube" ? "transcript.en.vtt" : "transcript.json"; - +export async function findDuplicateDirs(dataDir: string): Promise<{ + dirMap: Map<string, string>; + duplicates: Set<string>; +}> { const dirEntries = await readdir(dataDir).catch(() => [] as string[]); - - // Map ID -> directory name. The current layout is one dir per video where - // dirName === videoID; preserve detection of legacy "<date>_<id>-<title>" - // names by treating any duplicate as a duplicate. const dirMap = new Map<string, string>(); const duplicates = new Set<string>(); for (const dir of dirEntries) { @@ -66,7 +29,21 @@ export async function verifyTranscripts({ } dirMap.set(id, dir); } + return { dirMap, duplicates }; +} + +// For each ID in the archive, confirm there's a matching transcript on disk +// (either the YouTube auto-sub .vtt or a whisper transcript.json — channels +// can have a mix). Reports missing transcripts and duplicate video directories. +export async function verifyTranscripts({ + channelSlug, + paths, +}: VerifyTranscriptsOptions): Promise<VerifyTranscriptsResult> { + const channelDir = path.join(paths.channelsDir, channelSlug); + const dataDir = path.join(channelDir, "data"); + const archivePath = path.join(channelDir, "archive"); + const { dirMap, duplicates } = await findDuplicateDirs(dataDir); const archive = await readArchive(archivePath); const missing = new Set<string>(); for (const id of archive.ids) { @@ -75,7 +52,8 @@ export async function verifyTranscripts({ missing.add(id); continue; } - if (!(await exists(path.join(dataDir, dir, transcriptFilename)))) { + const files = await readVideoFiles(path.join(dataDir, dir)); + if (!isVideoTranscribed(files)) { missing.add(id); } } diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -7,19 +7,48 @@ export type SiteSettings = { headerTitle: string; homeTagline: string; maxTranscriptPageBytes: number; + transcribeBin: string; + transcribeArgs: string[]; + transcribeModel: string; }; export const TRANSCRIPT_PAGE_HARD_CAP_BYTES = 20 * 1024 * 1024; export const TRANSCRIPT_PAGE_MIN_BYTES = 256 * 1024; export const TRANSCRIPT_PAGE_DEFAULT_BYTES = 8 * 1024 * 1024; -const DEFAULTS: SiteSettings = { - siteTitle: "Transcript Browser", - siteDescription: "Browse and search video transcripts", - headerTitle: "Transcript Browser", - homeTagline: "", - maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES, -}; +export const TRANSCRIBE_PLACEHOLDER_AUDIO = "{audioFile}"; +export const TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE = "{outputBase}"; +export const TRANSCRIBE_PLACEHOLDER_MODEL = "{model}"; +export const TRANSCRIBE_KNOWN_PLACEHOLDERS: ReadonlyArray<string> = [ + TRANSCRIBE_PLACEHOLDER_AUDIO, + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_MODEL, +]; + +export const DEFAULT_TRANSCRIBE_ARGS: ReadonlyArray<string> = [ + "-ojf", + "-l", + "en", + "-m", + TRANSCRIBE_PLACEHOLDER_MODEL, + "-of", + TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE, + TRANSCRIBE_PLACEHOLDER_AUDIO, +]; + +function defaults(): SiteSettings { + const paths = getPaths(); + return { + siteTitle: "Transcript Browser", + siteDescription: "Browse and search video transcripts", + headerTitle: "Transcript Browser", + homeTagline: "", + maxTranscriptPageBytes: TRANSCRIPT_PAGE_DEFAULT_BYTES, + transcribeBin: paths.whisperBin, + transcribeArgs: [...DEFAULT_TRANSCRIBE_ARGS], + transcribeModel: paths.whisperModel, + }; +} export function getSettings(): SiteSettings { const file = getPaths().settingsFile; @@ -29,8 +58,17 @@ export function getSettings(): SiteSettings { } catch { parsed = {}; } - const merged = { ...DEFAULTS, ...parsed }; + const merged: SiteSettings = { ...defaults(), ...parsed }; merged.maxTranscriptPageBytes = clampPageBytes(merged.maxTranscriptPageBytes); + if (!Array.isArray(merged.transcribeArgs) || merged.transcribeArgs.length === 0) { + merged.transcribeArgs = [...DEFAULT_TRANSCRIBE_ARGS]; + } + if (typeof merged.transcribeBin !== "string" || !merged.transcribeBin.trim()) { + merged.transcribeBin = defaults().transcribeBin; + } + if (typeof merged.transcribeModel !== "string") { + merged.transcribeModel = defaults().transcribeModel; + } return merged; } @@ -44,11 +82,39 @@ function clampPageBytes(value: unknown): number { return Math.floor(n); } +export function validateTranscribeArgs(args: string[]): string | null { + if (!Array.isArray(args) || args.length === 0) { + return "Transcribe args must contain at least one entry"; + } + const joined = args.join(" "); + if (!joined.includes(TRANSCRIBE_PLACEHOLDER_AUDIO)) { + return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_AUDIO} placeholder`; + } + if (!joined.includes(TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE)) { + return `Transcribe args must include the ${TRANSCRIBE_PLACEHOLDER_OUTPUT_BASE} placeholder`; + } + for (const arg of args) { + const tokens = arg.match(/\{[^}]+\}/g) ?? []; + for (const token of tokens) { + if (!TRANSCRIBE_KNOWN_PLACEHOLDERS.includes(token)) { + return `Unknown placeholder ${token}; valid: ${TRANSCRIBE_KNOWN_PLACEHOLDERS.join(", ")}`; + } + } + } + return null; +} + export async function writeSettings(next: SiteSettings): Promise<void> { + const argsErr = validateTranscribeArgs(next.transcribeArgs); + if (argsErr) throw new Error(argsErr); + if (!next.transcribeBin.trim()) { + throw new Error("Transcribe binary is required"); + } const file = getPaths().settingsFile; const merged: SiteSettings = { ...next, maxTranscriptPageBytes: clampPageBytes(next.maxTranscriptPageBytes), + transcribeArgs: next.transcribeArgs.map((a) => String(a)), }; const tmp = `${file}.tmp-${process.pid}`; await fs.promises.writeFile(tmp, JSON.stringify(merged, null, 2) + "\n"); diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -0,0 +1,58 @@ +import path from "node:path"; +import { readdir, readFile } from "node:fs/promises"; + +export type VideoFiles = { + hasMeta: boolean; + hasYtVtt: boolean; + hasWhisper: boolean; + isUntranscribable: boolean; + audioFiles: string[]; +}; + +export type IndexTranscript = + | { kind: "whisper"; filename: "transcript.json" } + | { kind: "vtt"; filename: "transcript.en.vtt" }; + +export const VTT_FILENAME = "transcript.en.vtt"; +export const WHISPER_FILENAME = "transcript.json"; +export const META_FILENAME = "metadata.info.json"; + +export async function readVideoFiles(videoDir: string): Promise<VideoFiles> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + const hasMeta = entries.includes(META_FILENAME); + const hasYtVtt = entries.includes(VTT_FILENAME); + const hasWhisper = entries.includes(WHISPER_FILENAME); + const audioFiles = entries.filter( + (e) => e.startsWith("audio.") && !e.includes(".tmp-"), + ); + let isUntranscribable = false; + if (hasWhisper) { + try { + const raw = await readFile(path.join(videoDir, WHISPER_FILENAME), "utf8"); + const parsed = JSON.parse(raw) as { transcription?: unknown[] }; + isUntranscribable = + Array.isArray(parsed.transcription) && parsed.transcription.length === 0; + } catch { + isUntranscribable = false; + } + } + return { hasMeta, hasYtVtt, hasWhisper, isUntranscribable, audioFiles }; +} + +export function pickIndexTranscript(files: VideoFiles): IndexTranscript | null { + if (files.hasWhisper) return { kind: "whisper", filename: WHISPER_FILENAME }; + if (files.hasYtVtt) return { kind: "vtt", filename: VTT_FILENAME }; + return null; +} + +export function isVideoTranscribed(files: VideoFiles): boolean { + return files.hasWhisper || files.hasYtVtt; +} + +export function isVideoDownloaded(files: VideoFiles): boolean { + return ( + files.hasWhisper || + files.hasYtVtt || + files.audioFiles.length > 0 + ); +} diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -5,7 +5,6 @@ import { readFile, rename, rm, - stat, writeFile, } from "node:fs/promises"; import { execa } from "execa"; @@ -349,23 +348,21 @@ export async function destinationExists( handling: ChannelHandling, ): Promise<boolean> { const dir = path.join(dataDir, id); - if (handling === "youtube") { - return fileExists(path.join(dir, "transcript.en.vtt")); - } const entries = await readdir(dir).catch(() => [] as string[]); - return ( - entries.includes("transcript.json") || - entries.some((e) => e.startsWith("audio.")) - ); -} - -async function fileExists(p: string): Promise<boolean> { - try { - await stat(p); + // Either transcript counts as "already done" — channels can be hybrid. + if ( + entries.includes("transcript.en.vtt") || + entries.includes("transcript.json") + ) { return true; - } catch { - return false; } + // Transcribe channels treat raw audio as a download in progress so we don't + // re-fetch it before whisper runs. YouTube channels expect a .vtt; an audio + // file alone shouldn't suppress the next sync. + if (handling === "transcribe") { + return entries.some((e) => e.startsWith("audio.")); + } + return false; } async function sync(opts: RunYtdlpOpts): Promise<void> { diff --git a/editor/app/channels/[slug]/_components/WhisperPanel.tsx b/editor/app/channels/[slug]/_components/WhisperPanel.tsx @@ -172,7 +172,7 @@ export function WhisperPanel({ <div className="flex flex-col gap-2"> <Heading title="Clean audio for transcribed videos" - desc="Delete audio.* files in any video directory that has a transcript.json. Useful for reclaiming disk space once transcription is complete." + desc="Delete audio.* files in any video directory that has a transcript.json (whisper output). YouTube videos with only auto-sub .vtt files have no audio and are skipped automatically." /> <StreamActionLog trigger={() => cleanAudioAction(slug, cleanQueue)} diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -1,9 +1,9 @@ -import path from "node:path"; -import { readFile } from "node:fs/promises"; import type { Metadata } from "next"; import Link from "next/link"; import { notFound } from "next/navigation"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { inspectChannelHealth, type ChannelHealth } from "yt-dlp-transcript-common/controller/channelHealth"; +import { loadFailedTranscriptions } from "yt-dlp-transcript-common/controller/failedTranscriptions"; import { listUndownloadedVideos } from "yt-dlp-transcript-common/controller/undownloadedVideos"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { @@ -23,16 +23,6 @@ import { type ActionResult, } from "../actions"; -async function loadFailedVideoIds(slug: string): Promise<string[]> { - const file = path.join(getPaths().channelsDir, slug, "failed-transcriptions"); - try { - const raw = await readFile(file, "utf8"); - return raw.split("\n").filter(Boolean); - } catch { - return []; - } -} - export const dynamic = "force-dynamic"; export async function generateMetadata({ @@ -62,9 +52,10 @@ export default async function ChannelDetailPage({ const defaultQueueKey = platformQueueKey( config.platform ?? detectPlatform(config.url), ); - const failedVideoIds = await loadFailedVideoIds(slug); + const failedVideoIds = await loadFailedTranscriptions(paths, slug); const undownloadedVideos = await listUndownloadedVideos(paths, slug, config); const undownloadedIds = undownloadedVideos.map((v) => v.videoId); + const health = await inspectChannelHealth(paths, slug); return ( <div className="flex flex-col gap-6"> @@ -115,17 +106,17 @@ export default async function ChannelDetailPage({ /> </section> - {config.handling === "transcribe" && ( - <section className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6"> - <h2 className="text-lg font-semibold">Transcription</h2> - <WhisperPanel - slug={slug} - existingQueues={existingQueues} - failedVideoIds={failedVideoIds} - defaultConcurrency={paths.parallelTranscribeLimit} - /> - </section> - )} + <ChannelHealthSection slug={slug} health={health} /> + + <section className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6"> + <h2 className="text-lg font-semibold">Transcription</h2> + <WhisperPanel + slug={slug} + existingQueues={existingQueues} + failedVideoIds={failedVideoIds} + defaultConcurrency={paths.parallelTranscribeLimit} + /> + </section> <section className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6"> <h2 className="text-lg font-semibold">Inspect a video</h2> @@ -171,3 +162,113 @@ export default async function ChannelDetailPage({ </div> ); } + +type Bucket = { + key: keyof Omit<ChannelHealth, "totals">; + label: string; + description: string; + ariaLabel: string; +}; + +const BUCKETS: ReadonlyArray<Bucket> = [ + { + key: "downloadedNoTranscript", + label: "Downloaded but not transcribed", + description: "Audio is on disk but no .vtt or transcript.json yet.", + ariaLabel: "downloaded without transcript", + }, + { + key: "noTranscript", + label: "Missing transcript and no download", + description: "Video directory exists but has no transcript or audio.", + ariaLabel: "no transcript or download", + }, + { + key: "untranscribable", + label: "Marked untranscribable", + description: "transcript.json present but contains an empty transcription array.", + ariaLabel: "marked untranscribable", + }, + { + key: "noMetadata", + label: "Missing metadata.info.json", + description: "Video directory exists without yt-dlp metadata; index will skip it.", + ariaLabel: "missing metadata", + }, + { + key: "failedListed", + label: "In failed-transcriptions", + description: "Listed in the failure file; can be retried from the Transcription panel.", + ariaLabel: "failed list", + }, + { + key: "missingFromArchive", + label: "Archived ID with no directory", + description: "Video ID is in the archive but its data directory is missing.", + ariaLabel: "archived without dir", + }, + { + key: "duplicateDirs", + label: "Duplicate directories", + description: "Multiple directories resolve to the same video ID (e.g., legacy names).", + ariaLabel: "duplicate directories", + }, +]; + +function ChannelHealthSection({ + slug, + health, +}: { + slug: string; + health: ChannelHealth; +}) { + const populated = BUCKETS.filter((b) => health[b.key].length > 0); + const total = populated.reduce((acc, b) => acc + health[b.key].length, 0); + return ( + <section + aria-label="channel health" + className="flex flex-col gap-3 border-t border-zinc-200 dark:border-zinc-800 pt-6" + > + <div> + <h2 className="text-lg font-semibold">Channel health</h2> + <p className="text-sm text-zinc-500"> + {health.totals.videos} video dirs · {health.totals.transcribed}{" "} + transcribed · {health.totals.downloaded} downloaded + {total > 0 ? ` · ${total} abnormal` : ""} + </p> + </div> + {populated.length === 0 ? ( + <p + aria-label="channel health empty" + className="text-sm text-zinc-600 dark:text-zinc-400" + > + All clear. + </p> + ) : ( + <div className="grid grid-cols-1 lg:grid-cols-2 gap-3"> + {populated.map((b) => ( + <div + key={b.key} + className="flex flex-col gap-2 rounded border border-zinc-200 dark:border-zinc-800 p-3" + > + <div> + <h3 className="text-sm font-semibold"> + {b.label} ({health[b.key].length}) + </h3> + <p className="text-xs text-zinc-500">{b.description}</p> + </div> + <VideoIdList + slug={slug} + ids={health[b.key]} + ariaLabel={`${b.ariaLabel} list`} + emptyAriaLabel={`${b.ariaLabel} empty`} + emptyMessage="None" + itemAriaLabel={(id) => `${b.ariaLabel} ${id}`} + /> + </div> + ))} + </div> + )} + </section> + ); +} diff --git a/editor/app/channels/[slug]/videos/[id]/_components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/_components/VideoPanel.tsx @@ -95,7 +95,7 @@ export function VideoPanel({ /> )} - {handling === "transcribe" && !hasTranscript && ( + {!hasTranscript && ( <MarkUntranscribableSection slug={slug} videoId={videoId} /> )} diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -283,12 +283,6 @@ export async function markVideoUntranscribableAction( ): Promise<{ ok: true } | { ok: false; error: string }> { const r = await loadConfigOrError(slug); if (!r.ok) return r; - if (r.config.handling !== "transcribe") { - return { - ok: false, - error: "Only available on transcribe channels.", - }; - } const videoDir = videoDirOf(slug, videoId); try { const s = await stat(videoDir); diff --git a/editor/app/settings/_components/SettingsForm.tsx b/editor/app/settings/_components/SettingsForm.tsx @@ -47,6 +47,46 @@ export function SettingsForm({ initial }: Props) { hint="Bytes per generated transcript page (256 KB – 20 MB). Affects pagination next build." type="number" /> + <fieldset className="flex flex-col gap-3 border border-zinc-200 dark:border-zinc-800 rounded p-3"> + <legend className="px-1 text-sm font-medium">Transcription command</legend> + <p className="text-xs text-zinc-500"> + Used by the channel "Transcribe missing" / "Retry failures" actions. + The command must write <code>{`<outputBase>.json`}</code> in + whisper.cpp's <code>{`{ transcription: [{ offsets, text }] }`}</code>{" "} + format. Available placeholders:{" "} + <code>{"{audioFile}"}</code> (required),{" "} + <code>{"{outputBase}"}</code> (required),{" "} + <code>{"{model}"}</code> (optional). + </p> + <Field + label="Binary" + name="transcribeBin" + defaultValue={initial.transcribeBin} + required + hint="e.g., whisper-cli, /usr/local/bin/whisper, or your own wrapper." + /> + <Field + label="Model" + name="transcribeModel" + defaultValue={initial.transcribeModel} + hint="Substituted for {model}. Leave blank if your binary doesn't take a model." + /> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Args (one per line)</span> + <textarea + name="transcribeArgs" + defaultValue={initial.transcribeArgs.join("\n")} + rows={Math.max(6, initial.transcribeArgs.length + 1)} + className="rounded border border-zinc-300 dark:border-zinc-700 bg-white dark:bg-zinc-900 px-2 py-1 text-sm font-mono" + /> + <span className="text-xs text-zinc-500"> + Default whisper.cpp call:{" "} + <code> + -ojf -l en -m {"{model}"} -of {"{outputBase}"} {"{audioFile}"} + </code> + </span> + </label> + </fieldset> <div className="flex items-center gap-3"> <button type="submit" diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -4,6 +4,7 @@ import { revalidatePath } from "next/cache"; import { TRANSCRIPT_PAGE_HARD_CAP_BYTES, TRANSCRIPT_PAGE_MIN_BYTES, + validateTranscribeArgs, writeSettings, type SiteSettings, } from "yt-dlp-transcript-common/lib/settings"; @@ -19,9 +20,22 @@ export async function saveSettingsAction( const headerTitle = String(formData.get("headerTitle") ?? "").trim(); const homeTagline = String(formData.get("homeTagline") ?? "").trim(); const maxBytesRaw = String(formData.get("maxTranscriptPageBytes") ?? "").trim(); + const transcribeBin = String(formData.get("transcribeBin") ?? "").trim(); + const transcribeModel = String(formData.get("transcribeModel") ?? "").trim(); + const transcribeArgsRaw = String(formData.get("transcribeArgs") ?? ""); + const transcribeArgs = transcribeArgsRaw + .split("\n") + .map((s) => s.trim()) + .filter(Boolean); if (!siteTitle) return { ok: false, error: "Site title is required" }; if (!headerTitle) return { ok: false, error: "Header title is required" }; + if (!transcribeBin) { + return { ok: false, error: "Transcribe binary is required" }; + } + + const argsErr = validateTranscribeArgs(transcribeArgs); + if (argsErr) return { ok: false, error: argsErr }; const parsed = Number.parseInt(maxBytesRaw, 10); if (!Number.isFinite(parsed)) { @@ -43,6 +57,9 @@ export async function saveSettingsAction( headerTitle, homeTagline, maxTranscriptPageBytes: parsed, + transcribeBin, + transcribeModel, + transcribeArgs, }; await writeSettings(next); revalidatePath("/settings"); diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx @@ -48,7 +48,9 @@ export default function SettingsPage() { Read-only view of <code>getPaths()</code>. Set the matching env vars (TRANSCRIPTS_DIR, SETTINGS_FILE, EXPORT_PUBLIC_DIR, YTDLP_BIN, WHISPER_BIN, WHISPER_MODEL, PARALLEL_TRANSCRIBE_LIMIT) before - launching to override. + launching to override. The transcription command above takes + precedence over <code>whisperBin</code> / <code>whisperModel</code>{" "} + when running channel transcribe actions. </p> <table className="text-sm mt-3 w-full"> <tbody> diff --git a/editor/e2e/dashboard.spec.ts b/editor/e2e/dashboard.spec.ts @@ -37,7 +37,8 @@ test("sidebar links to Channels, Build, Jobs, Settings", async ({ page }) => { test("settings page lists getPaths() values", async ({ page }) => { await page.goto("/settings"); await page.locator("summary").filter({ hasText: "System paths" }).click(); - await expect(page.getByText("transcriptsDir")).toBeVisible(); - await expect(page.getByText("ytdlpBin")).toBeVisible(); - await expect(page.getByText("whisperBin")).toBeVisible(); + // Match the table-row headers, not stray <code> mentions in nearby blurbs. + await expect(page.getByRole("rowheader", { name: "transcriptsDir" })).toBeVisible(); + await expect(page.getByRole("rowheader", { name: "ytdlpBin" })).toBeVisible(); + await expect(page.getByRole("rowheader", { name: "whisperBin" })).toBeVisible(); }); diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts @@ -291,9 +291,11 @@ test("Mark untranscribable writes empty transcript and prunes failure list", asy expect(list.trim().split("\n")).toEqual(["vidOther"]); }); -test("Mark untranscribable button is absent on YouTube channels", async ({ +test("Mark untranscribable button is absent when transcript already exists", async ({ page, }) => { + // Hybrid channels: any video that already has a .vtt or transcript.json + // hides the marker, regardless of channel handling. await resetData("one-youtube-channel-with-data"); await page.goto("/channels/test-youtube/videos/20240101_test1234567"); await expect( diff --git a/editor/e2e/whisper.spec.ts b/editor/e2e/whisper.spec.ts @@ -122,11 +122,13 @@ test("Clean audio removes audio files only from transcribed videos", async ({ ).toBe(true); }); -test("whisper panel hidden for youtube channels", async ({ page }) => { +test("whisper panel is shown on youtube channels for hybrid use", async ({ page }) => { await resetData("one-youtube-channel"); await page.goto("/channels/test-youtube"); + // Channels can carry whisper transcripts alongside auto-subs; the + // transcription panel actions should always be reachable. await expect( page.getByRole("button", { name: "Transcribe missing" }), - ).toHaveCount(0); - await expect(page.getByRole("button", { name: "Verify" })).toHaveCount(0); + ).toBeVisible(); + await expect(page.getByRole("button", { name: "Verify" })).toBeVisible(); });