commit 68fc72000590eee36090f4206d0696f261a29f70
parent c199f78397fc06f359cd903394e52519d34f4d67
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 9 Jun 2026 22:23:49 -0400
fix regressions
Diffstat:
5 files changed, 71 insertions(+), 26 deletions(-)
diff --git a/common/ytdlp/audioCheckedDownload.ts b/common/ytdlp/audioCheckedDownload.ts
@@ -203,7 +203,7 @@ async function prepareDataTree(
signal: AbortSignal;
onCheckpoint: (rec: CheckpointRecord) => void;
},
-): Promise<void> {
+): Promise<string | null> {
// Drop any leftover .testing snapshots from crashed prior runs, and
// promote any orphan .good back to .part so yt-dlp resume picks up where
// we left off. When `expectedVideoIdHint` is set, scoped to that single
@@ -223,11 +223,18 @@ async function prepareDataTree(
try {
videoDirs = await readdir(dataDir);
} catch {
- return;
+ return null;
}
if (expectedVideoIdHint) {
videoDirs = videoDirs.filter((d) => d === expectedVideoIdHint);
}
+ // The subdir we treated as THIS download's resume target — restored a .good,
+ // or rolled back / discarded a malformed .part. The caller excludes it from
+ // its pre-launch snapshot so post-launch discovery still scopes to it even
+ // when expectedVideoIdHint is null. Clean pre-existing .parts (stale
+ // neighbours from other videos) are NOT reported, preserving the discovery
+ // logic's "ignore unrelated interrupted attempts" guarantee.
+ let resumeDir: string | null = null;
for (const sub of videoDirs) {
const dir = path.join(dataDir, sub);
let entries: string[] = [];
@@ -261,6 +268,7 @@ async function prepareDataTree(
onLog(`Restored prior validated snapshot: ${restored}\n`);
// Don't probe the restored snapshot — .good is clean by construction.
preExisting = null;
+ resumeDir = sub;
}
if (precheck && preExisting && !precheck.signal.aborted) {
@@ -295,6 +303,7 @@ async function prepareDataTree(
verdict: probe.verdict,
action: "rollback",
});
+ resumeDir = sub;
} else {
await rm(preExisting, { force: true });
onLog(
@@ -306,10 +315,12 @@ async function prepareDataTree(
verdict: probe.verdict,
action: "restart",
});
+ resumeDir = sub;
}
}
}
}
+ return resumeDir;
}
async function snapshotPart(
@@ -483,7 +494,7 @@ export async function runAudioCheckedYtdlp(
// Sticky across rollback/restart loops since the id stays the same.
let resolvedVideoDir: string | null = null;
- await prepareDataTree(
+ const resumeDir = await prepareDataTree(
opts.channelDir,
opts.expectedVideoIdHint,
opts.onLog,
@@ -497,11 +508,16 @@ export async function runAudioCheckedYtdlp(
// Snapshot of subdirs that existed before yt-dlp launched. discoverPartFile
// and the success-path fallback scan use this to ignore stale .part files
// belonging to other videos' interrupted attempts when the hint dir
- // doesn't yet exist (Rumble case).
+ // doesn't yet exist (Rumble case). Exclude the resume dir the pre-check just
+ // acted on: that subdir IS this download's target (e.g. a malformed .part we
+ // rolled back), so discovery must still scope to it even though it existed
+ // before launch — otherwise the finished file there is never found.
const preLaunchSubdirs: ReadonlySet<string> = new Set(
- await readdir(path.join(opts.channelDir, "data")).catch(
- () => [] as string[],
- ),
+ (
+ await readdir(path.join(opts.channelDir, "data")).catch(
+ () => [] as string[],
+ )
+ ).filter((d) => d !== resumeDir),
);
// Main loop: each iteration = one yt-dlp launch. Decides what to do based
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged.
- **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template).
- **Global default for parallel transcriptions.** A new **Settings → Parallel transcriptions** field sets how many videos a "Transcribe missing" / bucket run transcribes at once when its per-run Concurrency input is left blank (default **2**, clamped 1–16). This replaces the old `PARALLEL_TRANSCRIBE_LIMIT` env-var default of 4 as the source of the default: the value is now persisted in `settings.json`, shown as the placeholder in each channel's Concurrency input, and used as the server-side fallback. The per-run Concurrency input still overrides it for a single run.
- **Managed downloads skip live and upcoming videos by default.** Before downloading each video, the managed downloader now runs a quick metadata-only pass, then evaluates an app-level filter: videos that are currently live or scheduled/upcoming are skipped (their finished VODs still download normally — a skip isn't archived, so the next **Sync** / **Download missing** retries the video once the stream ends). A skip is recorded in the video's `download-outcome.json` (`status: "skipped-filtered"` with the filter name and reason) and logged to `download.log`; it never counts as a failure or aborts the batch, and the run's summary line reports how many were skipped. Skipped-live videos are surfaced in a new **Skipped: live or upcoming** bucket in the channel's Diagnostics (with a **Retry** to force an attempt now). On by default for every channel via a new **Settings → Skip live and upcoming videos** toggle, with a per-channel override (`config.json` `skipLiveDownloads`). To support the filter (and as a reusable building block for future per-video decisions), each managed download is now split into a metadata fetch followed by the real download, which reuses that metadata via `--load-info-json` so the second pass doesn't re-extract; the audio-integrity-checked download path re-extracts as before. Turn the toggle off for a channel that streams nothing to allow live captures.
diff --git a/editor/e2e/import-video.spec.ts b/editor/e2e/import-video.spec.ts
@@ -26,12 +26,19 @@ test("import single video by URL fetches it and appends to the archive", async (
await expect(log).toContainText("oneoff00001", { timeout: 30_000 });
// The video lands in the canonical data/<id>/ dir, with metadata + transcript
- // for a youtube-handling channel.
- expect(
- await pathExists(
- "test-transcripts/channels/test-pipeline/data/oneoff00001/metadata.info.json",
- ),
- ).toBe(true);
+ // for a youtube-handling channel. The log above matches as soon as the
+ // command echo (which contains the URL) streams, which can be before the
+ // prefetch has written the file — so poll for the artifact rather than
+ // reading it once.
+ await expect
+ .poll(
+ () =>
+ pathExists(
+ "test-transcripts/channels/test-pipeline/data/oneoff00001/metadata.info.json",
+ ),
+ { timeout: 30_000 },
+ )
+ .toBe(true);
// And it's recorded in the channel archive so a later sync won't re-fetch it.
await expect(async () => {
diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts
@@ -61,8 +61,14 @@ test("list shows configured sites and edit updates branding", async ({
page.getByRole("status").filter({ hasText: "Saved" }),
).toBeVisible();
- const site = await readJson<SiteFile>("test-transcripts/sites/alpha/site.json");
- expect(site.siteTitle).toBe("Alpha Renamed");
+ // The "Saved" status and the on-disk write can land slightly apart; poll the
+ // persisted file until the rename completes rather than reading it once.
+ await expect(async () => {
+ const site = await readJson<SiteFile>(
+ "test-transcripts/sites/alpha/site.json",
+ );
+ expect(site.siteTitle).toBe("Alpha Renamed");
+ }).toPass({ timeout: 10_000 });
});
test("delete removes the site config", async ({ page }) => {
diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts
@@ -57,11 +57,24 @@ test("failed entry stays in the list across reloads (skip semantics)", async ({
await expect(page.getByLabel("failed transcription vidA")).toBeVisible();
// Running Transcribe missing should skip the listed failure rather than
// re-attempt it (the old retry-failures flow has been replaced by Clear).
+ // Failed videos are filtered out of the batch up front, so only the other
+ // two (vidB, vidC) are attempted and vidA is never started.
await page.getByRole("button", { name: "Transcribe missing" }).click();
- await expect(page.getByLabel("Transcribe missing output")).toContainText(
- "Skipping previously failed transcription for vidA",
+ const out = page.getByLabel("Transcribe missing output");
+ await expect(out).toContainText(
+ "Whisper batch: 2 succeeded, 0 failed, 0 skipped, 2 attempted.",
{ timeout: 30_000 },
);
+ // vidA was excluded, never transcribed…
+ expect(await out.textContent()).not.toContain("Transcribe vidA");
+ // …and it stays on the failed list (skip semantics persist).
+ const failList = await readFile(
+ resolvePath(
+ "test-transcripts/channels/test-transcribe/failed-transcriptions",
+ ),
+ "utf8",
+ );
+ expect(failList).toContain("vidA");
await page.reload();
await expect(page.getByLabel("failed transcription vidA")).toBeVisible();
});
@@ -72,15 +85,17 @@ test("transcribe-one writes transcript.json for a specific audio file", async ({
await resetData("one-transcribe-channel-with-audio");
await page.goto("/channels/test-transcribe/videos/vidA");
await page.getByRole("button", { name: "Transcribe audio.m4a" }).click();
- await expect(page.getByLabel("Transcribe audio.m4a output")).toContainText(
- "wrote",
- { timeout: 30_000 },
- );
- expect(
- await pathExists(
- "test-transcripts/channels/test-transcribe/data/vidA/transcript.json",
- ),
- ).toBe(true);
+ // Gate on the durable artifact the action produces rather than the streamed
+ // "[fake-whisper] wrote …" line, which can race the assertion.
+ await expect
+ .poll(
+ () =>
+ pathExists(
+ "test-transcripts/channels/test-transcribe/data/vidA/transcript.json",
+ ),
+ { timeout: 30_000 },
+ )
+ .toBe(true);
});
test("transcode keeps source and writes audio.<target>", async ({ page }) => {