Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 90fd8452e28d057895650bf68e67df69a5f9f3f1
parent 54305634cbeb1a3b3bb50f8833df13213b299f62
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 30 May 2026 00:03:35 -0400

Fix single pipeline-skip transcribe false-detecting audio files with prefix

Diffstat:
Mcommon/controller/transcribeOne.ts | 5++---
Mcommon/lib/videoStatus.ts | 51+++++++++++++++++++++++++++++++--------------------
Meditor/CHANGELOG.md | 1+
Meditor/app/channels/[slug]/bulkVideoActions.ts | 13++-----------
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 3++-
Meditor/e2e/whisper-video.spec.ts | 79++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
6 files changed, 116 insertions(+), 36 deletions(-)

diff --git a/common/controller/transcribeOne.ts b/common/controller/transcribeOne.ts @@ -10,6 +10,7 @@ import { getSettings, } from "../lib/settings"; import { normalizeTranscript } from "./normalizeTranscript"; +import { isRealAudioFile } from "../lib/videoStatus"; const { pathExists, readdir, rename } = fs; @@ -23,9 +24,7 @@ async function resolveAudioFile( if (await pathExists(path.join(videoDir, requested))) return requested; if (strict) return null; const entries = await readdir(videoDir); - const candidates = entries.filter( - (e) => e.startsWith("audio.") && !e.endsWith(".part"), - ); + const candidates = entries.filter(isRealAudioFile); if (candidates.length === 0) return null; for (const preferred of AUDIO_PREFERENCE) { if (candidates.includes(preferred)) return preferred; diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -25,6 +25,35 @@ export const LIVE_CHAT_FILENAME = "transcript.live_chat.json"; export const LIVE_CHAT_CUES_FILENAME = "live_chat.cues.json"; export const META_FILENAME = "metadata.info.json"; +// A finalized media output under data/<id>/audio.<ext>. Excludes yt-dlp partials +// (.part) and audio-check snapshots (.part.good/.part.testing), the metadata +// sidecar (*.info.json), temp work files (.tmp-...), and the live-chat sidecar +// (audio.live_chat.json[.part]) — live_chat ignores the `subtitle:` output prefix +// and lands under the default audio.%(ext)s template, which is the source of the +// "no audio file found" whisper crash. +export function isRealAudioFile(name: string): boolean { + if (!name.startsWith("audio.")) return false; + if (name.includes(".tmp-")) return false; + if (name.endsWith(".info.json")) return false; + if (name.endsWith(".live_chat.json")) return false; + if (name.endsWith(".part")) return false; // covers .live_chat.json.part too + if (name.endsWith(".part.good")) return false; + if (name.endsWith(".part.testing")) return false; + return true; +} + +// A genuine resumable partial: an interrupted audio download, NOT a live-chat +// sidecar that merely ends in .part. +export function isPartAudioFile(name: string): boolean { + if (!name.startsWith("audio.")) return false; + if (name.includes(".tmp-")) return false; + if (name.endsWith(".live_chat.json.part")) return false; + if (!name.endsWith(".part")) return false; + if (name.endsWith(".part.good")) return false; + if (name.endsWith(".part.testing")) return false; + return true; +} + export type SubTrack = { // "live_chat" | language code like "es", "en-orig", etc. track: string; @@ -58,26 +87,8 @@ export async function readVideoFiles( const hasYtVtt = entries.includes(VTT_FILENAME); const hasWhisper = entries.includes(WHISPER_FILENAME); const hasCuesJson = entries.includes(CUES_JSON_FILENAME); - const audioFiles = entries.filter( - (e) => - e.startsWith("audio.") && - !e.includes(".tmp-") && - !e.endsWith(".info.json") && - // yt-dlp writes audio.<ext>.part while downloading; treat those as - // incomplete so prefilters re-invoke yt-dlp (which resumes via -c). - !e.endsWith(".part") && - // audio-check internals — not user-visible audio outputs. - !e.endsWith(".part.good") && - !e.endsWith(".part.testing"), - ); - const partAudioFiles = entries.filter( - (e) => - e.startsWith("audio.") && - !e.includes(".tmp-") && - e.endsWith(".part") && - !e.endsWith(".part.good") && - !e.endsWith(".part.testing"), - ); + const audioFiles = entries.filter(isRealAudioFile); + const partAudioFiles = entries.filter(isPartAudioFile); let isUntranscribable = false; if (hasWhisper && opts.checkUntranscribable) { const whisperPath = path.join(videoDir, WHISPER_FILENAME); diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **"Audio + Whisper (skip pipeline)" no longer mistakes a live-chat sidecar for audio.** yt-dlp writes the live-chat track as `audio.live_chat.json` (it ignores the subtitle output path), so an interrupted download could leave an `audio.live_chat.json.part` behind. One-click whisper saw the `audio.` prefix, decided audio was already on disk, skipped the download, and then failed with "no audio file found". Live-chat sidecars (`.part` or completed) are now excluded everywhere audio is detected, so whisper downloads the real audio and transcribes as expected. - **Cleanup tasks on `/actionable`.** The global Actionable view now surfaces two housekeeping sections alongside the download/transcribe backlog: **channels with cleanable transcribed audio** (videos that have a whisper transcript but still keep their audio on disk) and **channels with extra audio formats** (videos with leftover audio files beside the channel's configured format), each with a per-channel count and an inline button. Because these delete files, the buttons pop a confirm dialog before queuing. The dashboard "Needs attention" card is unchanged — it still tracks only download/transcribe work. - **Per-video "do not clean" archive toggle.** Each video detail page has a new **Archive media** section to mark a video "do not clean". Marked videos are skipped by both cleanup jobs (their audio is preserved for archiving) and excluded from the cleanup counts on `/actionable`. The marker is reversible from the same toggle, and an "archived" badge shows on the video page while it's set. - **Twitch.tv support.** Twitch is now a first-class platform: Twitch channel/VOD URLs are auto-detected, "Twitch" is selectable in the channel form's platform dropdown and the charts platform filter, videos play via an in-browser Twitch embed, and downloads are queued on a `platform:twitch` queue like the other platforms. diff --git a/editor/app/channels/[slug]/bulkVideoActions.ts b/editor/app/channels/[slug]/bulkVideoActions.ts @@ -10,6 +10,7 @@ import { queueKeyForUrl, } from "yt-dlp-transcript-common/lib/platform"; import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels"; +import { isRealAudioFile } from "yt-dlp-transcript-common/lib/videoStatus"; import { downloadVideoPipelineAction, markVideoUntranscribableAction, @@ -53,17 +54,7 @@ async function firstAudioFile(slug: string, videoId: string): Promise<string | n } catch { return null; } - const audio = entries - .filter( - (e) => - e.startsWith("audio.") && - !e.includes(".tmp-") && - !e.endsWith(".part") && - !e.endsWith(".part.good") && - !e.endsWith(".part.testing") && - !e.endsWith(".info.json"), - ) - .sort(); + const audio = entries.filter(isRealAudioFile).sort(); return audio[0] ?? null; } diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -10,6 +10,7 @@ import type { } from "yt-dlp-transcript-common/lib/channelConfig"; import { AUDIO_FORMAT_VALUES } from "yt-dlp-transcript-common/lib/channelConfig"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; +import { isRealAudioFile } from "yt-dlp-transcript-common/lib/videoStatus"; import { platformQueueKey, queueKeyForUrl, @@ -194,7 +195,7 @@ export async function whisperVideoAction( fn: async (onLog, signal, _setProgress, ctx) => { const audioFormat = r.config.audioFormat ?? "mp3"; const entries = await readdir(videoDir).catch(() => [] as string[]); - const hasAudio = entries.some((e: string) => e.startsWith("audio.")); + const hasAudio = entries.some(isRealAudioFile); if (!hasAudio) { const url = await findVideoSourceUrl(paths, slug, videoId); if (!url) { diff --git a/editor/e2e/whisper-video.spec.ts b/editor/e2e/whisper-video.spec.ts @@ -1,4 +1,4 @@ -import { readFile, rm } from "node:fs/promises"; +import { readFile, rm, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { pathExists, resetData, resolvePath } from "./helpers"; @@ -63,3 +63,80 @@ test("one-click whisper skips download when audio is already on disk", async ({ const invocations = await readFile(invocationsPath, "utf8").catch(() => ""); expect(invocations).not.toContain("download-one:"); }); + +// yt-dlp writes the live-chat track under the default audio.%(ext)s output +// template (live_chat ignores the `subtitle:` prefix), so an interrupted +// download leaves an `audio.live_chat.json.part` sidecar. A naive +// startsWith("audio.") prefilter mistook it for real audio, skipped the +// download, then crashed in whisper with "no audio file found". +test("one-click whisper ignores a leftover live-chat .part sidecar", async ({ + page, +}) => { + await resetData("youtube-with-playlist"); + await rm( + resolvePath( + "test-transcripts/channels/test-youtube/data/fake00000001/transcript.en.vtt", + ), + ); + await writeFile( + resolvePath( + "test-transcripts/channels/test-youtube/data/fake00000001/audio.live_chat.json.part", + ), + "[]", + ); + + await page.goto("/channels/test-youtube/videos/fake00000001"); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); + + const log = page.getByLabel("Audio + Whisper for fake00000001 output"); + await expect(log).toContainText("downloading", { timeout: 30_000 }); + await expect(log).toContainText("Transcribe fake00000001 done", { + timeout: 30_000, + }); + await expect(log).not.toContainText("skipping download"); + await expect(log).not.toContainText("no audio file found"); + + expect( + await pathExists( + "test-transcripts/channels/test-youtube/data/fake00000001/audio.mp3", + ), + ).toBe(true); + expect( + await pathExists( + "test-transcripts/channels/test-youtube/data/fake00000001/transcript.json", + ), + ).toBe(true); +}); + +// The same blind spot for a completed live-chat sidecar: a finalized +// audio.live_chat.json (no .part) must not be treated as audio either, or it +// would be handed to whisper as if it were the audio file. +test("one-click whisper ignores a completed live-chat sidecar", async ({ + page, +}) => { + await resetData("youtube-with-playlist"); + await rm( + resolvePath( + "test-transcripts/channels/test-youtube/data/fake00000001/transcript.en.vtt", + ), + ); + await writeFile( + resolvePath( + "test-transcripts/channels/test-youtube/data/fake00000001/audio.live_chat.json", + ), + "[]", + ); + + await page.goto("/channels/test-youtube/videos/fake00000001"); + await page.getByLabel("download mode").selectOption("whisper"); + await page.getByRole("button", { name: /^Audio \+ Whisper$/ }).click(); + + const log = page.getByLabel("Audio + Whisper for fake00000001 output"); + await expect(log).toContainText("downloading", { timeout: 30_000 }); + await expect(log).toContainText("Transcribe fake00000001 done", { + timeout: 30_000, + }); + await expect(log).not.toContainText("skipping download"); + await expect(log).not.toContainText("no audio file found"); +});