Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 68fc72000590eee36090f4206d0696f261a29f70
parent c199f78397fc06f359cd903394e52519d34f4d67
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Tue,  9 Jun 2026 22:23:49 -0400

fix regressions

Diffstat:
Mcommon/ytdlp/audioCheckedDownload.ts | 30+++++++++++++++++++++++-------
Meditor/CHANGELOG.md | 1+
Meditor/e2e/import-video.spec.ts | 19+++++++++++++------
Meditor/e2e/sites-crud.spec.ts | 10++++++++--
Meditor/e2e/video-page.spec.ts | 37++++++++++++++++++++++++++-----------
5 files changed, 71 insertions(+), 26 deletions(-)

diff --git a/common/ytdlp/audioCheckedDownload.ts b/common/ytdlp/audioCheckedDownload.ts @@ -203,7 +203,7 @@ async function prepareDataTree( signal: AbortSignal; onCheckpoint: (rec: CheckpointRecord) => void; }, -): Promise<void> { +): Promise<string | null> { // Drop any leftover .testing snapshots from crashed prior runs, and // promote any orphan .good back to .part so yt-dlp resume picks up where // we left off. When `expectedVideoIdHint` is set, scoped to that single @@ -223,11 +223,18 @@ async function prepareDataTree( try { videoDirs = await readdir(dataDir); } catch { - return; + return null; } if (expectedVideoIdHint) { videoDirs = videoDirs.filter((d) => d === expectedVideoIdHint); } + // The subdir we treated as THIS download's resume target — restored a .good, + // or rolled back / discarded a malformed .part. The caller excludes it from + // its pre-launch snapshot so post-launch discovery still scopes to it even + // when expectedVideoIdHint is null. Clean pre-existing .parts (stale + // neighbours from other videos) are NOT reported, preserving the discovery + // logic's "ignore unrelated interrupted attempts" guarantee. + let resumeDir: string | null = null; for (const sub of videoDirs) { const dir = path.join(dataDir, sub); let entries: string[] = []; @@ -261,6 +268,7 @@ async function prepareDataTree( onLog(`Restored prior validated snapshot: ${restored}\n`); // Don't probe the restored snapshot — .good is clean by construction. preExisting = null; + resumeDir = sub; } if (precheck && preExisting && !precheck.signal.aborted) { @@ -295,6 +303,7 @@ async function prepareDataTree( verdict: probe.verdict, action: "rollback", }); + resumeDir = sub; } else { await rm(preExisting, { force: true }); onLog( @@ -306,10 +315,12 @@ async function prepareDataTree( verdict: probe.verdict, action: "restart", }); + resumeDir = sub; } } } } + return resumeDir; } async function snapshotPart( @@ -483,7 +494,7 @@ export async function runAudioCheckedYtdlp( // Sticky across rollback/restart loops since the id stays the same. let resolvedVideoDir: string | null = null; - await prepareDataTree( + const resumeDir = await prepareDataTree( opts.channelDir, opts.expectedVideoIdHint, opts.onLog, @@ -497,11 +508,16 @@ export async function runAudioCheckedYtdlp( // Snapshot of subdirs that existed before yt-dlp launched. discoverPartFile // and the success-path fallback scan use this to ignore stale .part files // belonging to other videos' interrupted attempts when the hint dir - // doesn't yet exist (Rumble case). + // doesn't yet exist (Rumble case). Exclude the resume dir the pre-check just + // acted on: that subdir IS this download's target (e.g. a malformed .part we + // rolled back), so discovery must still scope to it even though it existed + // before launch — otherwise the finished file there is never found. const preLaunchSubdirs: ReadonlySet<string> = new Set( - await readdir(path.join(opts.channelDir, "data")).catch( - () => [] as string[], - ), + ( + await readdir(path.join(opts.channelDir, "data")).catch( + () => [] as string[], + ) + ).filter((d) => d !== resumeDir), ); // Main loop: each iteration = one yt-dlp launch. Decides what to do based diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged. - **First-class support for multiple transcription apps (chough + whisper.cpp).** Transcription is no longer hardcoded to whisper.cpp. **Settings → Transcription** now has an **App** dropdown (whisper.cpp / chough) with per-app fields, replacing the old flat Binary / Model / Args command. Each app owns how it builds its command line, what file it writes, how its output JSON is parsed, and how its progress output is read — so adding another tool is a small code module (`common/lib/transcriptionApps.ts`). **chough** is supported in both **local** and **remote** modes (set a **Remote URL** to transcribe via a `chough --server`, passing `CHOUGH_URL`; leave it blank for local), with optional **Chunk size** (`-c`) and **Model** (`CHOUGH_MODEL`) fields; its binary defaults to the `CHOUGH_BIN` env var. whisper.cpp keeps its Binary / Model / custom-args template. Because chough writes its output to the exact `-o` path (no `.json` appended, unlike whisper-cli's `-of`), the runner now renames the app-declared output file — fixing transcripts that previously failed to materialize under chough. Transcript parsing is **format-aware and back-compatible**: existing whisper.cpp `transcript.json` files and new chough files coexist, each parsed correctly by content sniff (chough's seconds-based `chunk_data` vs whisper's millisecond `transcription` offsets), with the detected format recorded per video in `transcript.cues.json` (`transcriptFormat`) and a fallback to whisper.cpp for anything unrecognized — so re-indexing a mixed corpus (including across shard machines running different tools) just works. An existing `settings.json` migrates automatically: legacy `transcribeBin`/`transcribeArgs`/`transcribeModel` map onto the matching app (a binary named `chough` adopts the chough app; everything else becomes whisper.cpp, preserving a customized args template). - **Global default for parallel transcriptions.** A new **Settings → Parallel transcriptions** field sets how many videos a "Transcribe missing" / bucket run transcribes at once when its per-run Concurrency input is left blank (default **2**, clamped 1–16). This replaces the old `PARALLEL_TRANSCRIBE_LIMIT` env-var default of 4 as the source of the default: the value is now persisted in `settings.json`, shown as the placeholder in each channel's Concurrency input, and used as the server-side fallback. The per-run Concurrency input still overrides it for a single run. - **Managed downloads skip live and upcoming videos by default.** Before downloading each video, the managed downloader now runs a quick metadata-only pass, then evaluates an app-level filter: videos that are currently live or scheduled/upcoming are skipped (their finished VODs still download normally — a skip isn't archived, so the next **Sync** / **Download missing** retries the video once the stream ends). A skip is recorded in the video's `download-outcome.json` (`status: "skipped-filtered"` with the filter name and reason) and logged to `download.log`; it never counts as a failure or aborts the batch, and the run's summary line reports how many were skipped. Skipped-live videos are surfaced in a new **Skipped: live or upcoming** bucket in the channel's Diagnostics (with a **Retry** to force an attempt now). On by default for every channel via a new **Settings → Skip live and upcoming videos** toggle, with a per-channel override (`config.json` `skipLiveDownloads`). To support the filter (and as a reusable building block for future per-video decisions), each managed download is now split into a metadata fetch followed by the real download, which reuses that metadata via `--load-info-json` so the second pass doesn't re-extract; the audio-integrity-checked download path re-extracts as before. Turn the toggle off for a channel that streams nothing to allow live captures. diff --git a/editor/e2e/import-video.spec.ts b/editor/e2e/import-video.spec.ts @@ -26,12 +26,19 @@ test("import single video by URL fetches it and appends to the archive", async ( await expect(log).toContainText("oneoff00001", { timeout: 30_000 }); // The video lands in the canonical data/<id>/ dir, with metadata + transcript - // for a youtube-handling channel. - expect( - await pathExists( - "test-transcripts/channels/test-pipeline/data/oneoff00001/metadata.info.json", - ), - ).toBe(true); + // for a youtube-handling channel. The log above matches as soon as the + // command echo (which contains the URL) streams, which can be before the + // prefetch has written the file — so poll for the artifact rather than + // reading it once. + await expect + .poll( + () => + pathExists( + "test-transcripts/channels/test-pipeline/data/oneoff00001/metadata.info.json", + ), + { timeout: 30_000 }, + ) + .toBe(true); // And it's recorded in the channel archive so a later sync won't re-fetch it. await expect(async () => { diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts @@ -61,8 +61,14 @@ test("list shows configured sites and edit updates branding", async ({ page.getByRole("status").filter({ hasText: "Saved" }), ).toBeVisible(); - const site = await readJson<SiteFile>("test-transcripts/sites/alpha/site.json"); - expect(site.siteTitle).toBe("Alpha Renamed"); + // The "Saved" status and the on-disk write can land slightly apart; poll the + // persisted file until the rename completes rather than reading it once. + await expect(async () => { + const site = await readJson<SiteFile>( + "test-transcripts/sites/alpha/site.json", + ); + expect(site.siteTitle).toBe("Alpha Renamed"); + }).toPass({ timeout: 10_000 }); }); test("delete removes the site config", async ({ page }) => { diff --git a/editor/e2e/video-page.spec.ts b/editor/e2e/video-page.spec.ts @@ -57,11 +57,24 @@ test("failed entry stays in the list across reloads (skip semantics)", async ({ await expect(page.getByLabel("failed transcription vidA")).toBeVisible(); // Running Transcribe missing should skip the listed failure rather than // re-attempt it (the old retry-failures flow has been replaced by Clear). + // Failed videos are filtered out of the batch up front, so only the other + // two (vidB, vidC) are attempted and vidA is never started. await page.getByRole("button", { name: "Transcribe missing" }).click(); - await expect(page.getByLabel("Transcribe missing output")).toContainText( - "Skipping previously failed transcription for vidA", + const out = page.getByLabel("Transcribe missing output"); + await expect(out).toContainText( + "Whisper batch: 2 succeeded, 0 failed, 0 skipped, 2 attempted.", { timeout: 30_000 }, ); + // vidA was excluded, never transcribed… + expect(await out.textContent()).not.toContain("Transcribe vidA"); + // …and it stays on the failed list (skip semantics persist). + const failList = await readFile( + resolvePath( + "test-transcripts/channels/test-transcribe/failed-transcriptions", + ), + "utf8", + ); + expect(failList).toContain("vidA"); await page.reload(); await expect(page.getByLabel("failed transcription vidA")).toBeVisible(); }); @@ -72,15 +85,17 @@ test("transcribe-one writes transcript.json for a specific audio file", async ({ await resetData("one-transcribe-channel-with-audio"); await page.goto("/channels/test-transcribe/videos/vidA"); await page.getByRole("button", { name: "Transcribe audio.m4a" }).click(); - await expect(page.getByLabel("Transcribe audio.m4a output")).toContainText( - "wrote", - { timeout: 30_000 }, - ); - expect( - await pathExists( - "test-transcripts/channels/test-transcribe/data/vidA/transcript.json", - ), - ).toBe(true); + // Gate on the durable artifact the action produces rather than the streamed + // "[fake-whisper] wrote …" line, which can race the assertion. + await expect + .poll( + () => + pathExists( + "test-transcripts/channels/test-transcribe/data/vidA/transcript.json", + ), + { timeout: 30_000 }, + ) + .toBe(true); }); test("transcode keeps source and writes audio.<target>", async ({ page }) => {