Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit fa48d29d2e54a8a6e14254152e4d580ed2b6f6d5
parent 1964b15ee092b93d827abb36bf41be9731a80743
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 20 May 2026 19:54:45 -0400

fix unchecked audio integrity downloads

Diffstat:
Mcommon/ytdlp/audioCheckedDownload.ts | 83+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--------------
Mcommon/ytdlp/downloadOneManaged.ts | 10++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/e2e/audio-check-scenarios.spec.ts | 58++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
4 files changed, 138 insertions(+), 14 deletions(-)

diff --git a/common/ytdlp/audioCheckedDownload.ts b/common/ytdlp/audioCheckedDownload.ts @@ -97,6 +97,15 @@ export type AudioCheckedOpts = { onLog: (s: string) => void; signal: AbortSignal; archiveMarker: string; + // On-disk subdir name under channelDir/data/ this launch is expected to + // write to. For most platforms it's extractVideoId(url). Null when the + // caller can't predict the dir name (Rumble: yt-dlp's internal id only + // shows up after metadata.info.json is written). The orchestrator scopes + // .part discovery and final-file resolution to this dir; when null it + // restricts to subdirs that didn't exist before launch. Without this, + // channels with many stale .parts from prior interrupted attempts can + // probe the wrong file and let a malformed download finalize. + expectedVideoIdHint: string | null; // Optional test-only knob: extra ms to sleep while SIGSTOPped to help // simulate cancellation-during-pause scenarios. debugPauseMs?: number; @@ -395,28 +404,47 @@ async function snapshotPart( } } +async function findPartIn(dir: string): Promise<string | null> { + const entries = await readdir(dir).catch(() => [] as string[]); + for (const e of entries) { + if ( + e.startsWith("audio.") && + e.endsWith(".part") && + !e.endsWith(".part.testing") && + !e.endsWith(".part.good") + ) { + return path.join(dir, e); + } + } + return null; +} + async function discoverPartFile( channelDir: string, + expectedVideoIdHint: string | null, + preLaunchSubdirs: ReadonlySet<string>, signal: AbortSignal, ): Promise<string | null> { const dataDir = path.join(channelDir, "data"); const deadline = Date.now() + PART_DISCOVERY_TIMEOUT_MS; while (Date.now() < deadline) { if (signal.aborted) return null; + // Preferred path: scope to the hinted subdir. extractVideoId(url) gives + // the right answer for most platforms; this avoids latching onto stale + // .part files in other video subdirs. + if (expectedVideoIdHint) { + const hit = await findPartIn(path.join(dataDir, expectedVideoIdHint)); + if (hit) return hit; + } + // Fallback (Rumble, or hint mismatch): only consider subdirs that + // appeared *after* launch. Pre-existing subdirs are skipped — their + // .part files belong to other videos' interrupted attempts. const subdirs = await readdir(dataDir).catch(() => [] as string[]); for (const sub of subdirs) { - const dir = path.join(dataDir, sub); - const entries = await readdir(dir).catch(() => [] as string[]); - for (const e of entries) { - if ( - e.startsWith("audio.") && - e.endsWith(".part") && - !e.endsWith(".part.testing") && - !e.endsWith(".part.good") - ) { - return path.join(dir, e); - } - } + if (sub === expectedVideoIdHint) continue; // already checked above + if (preLaunchSubdirs.has(sub)) continue; + const hit = await findPartIn(path.join(dataDir, sub)); + if (hit) return hit; } await new Promise<void>((resolve) => { const t = setTimeout(resolve, PART_DISCOVERY_INTERVAL_MS); @@ -453,6 +481,16 @@ export async function runAudioCheckedYtdlp( onCheckpoint: (rec) => checkpoints.push(rec), }); + // Snapshot of subdirs that existed before yt-dlp launched. discoverPartFile + // and the success-path fallback scan use this to ignore stale .part files + // belonging to other videos' interrupted attempts when the hint dir + // doesn't yet exist (Rumble case). + const preLaunchSubdirs: ReadonlySet<string> = new Set( + await readdir(path.join(opts.channelDir, "data")).catch( + () => [] as string[], + ), + ); + // Main loop: each iteration = one yt-dlp launch. Decides what to do based // on whether the launch ended in success, rollback, restart, or error. while (true) { @@ -779,7 +817,12 @@ export async function runAudioCheckedYtdlp( // Start the watcher: discover .part, then arm interval. let launchVideoDir: string | null = resolvedVideoDir; const watcherTask = (async () => { - partFile = await discoverPartFile(opts.channelDir, opts.signal); + partFile = await discoverPartFile( + opts.channelDir, + opts.expectedVideoIdHint, + preLaunchSubdirs, + opts.signal, + ); if (partFile) { launchVideoDir = path.dirname(partFile); resolvedVideoDir = launchVideoDir; @@ -870,10 +913,22 @@ export async function runAudioCheckedYtdlp( if (await pathExists(candidate)) resolvedFinal = candidate; } if (!resolvedFinal) { - // Scan the data tree to find any newly written audio.<ext>. + // Scan the data tree to find any newly written audio.<ext>. Scope + // the candidate set the same way discoverPartFile does: prefer the + // hinted subdir, then any subdir that didn't exist at launch time. + // Pre-existing subdirs from other videos are skipped — picking one + // would mis-probe an unrelated file (e.g. another video's already- + // finalized audio.mp3) and skip the real download's final probe. const dataDir = path.join(opts.channelDir, "data"); + const candidates: string[] = []; + if (opts.expectedVideoIdHint) candidates.push(opts.expectedVideoIdHint); const subdirs = await readdir(dataDir).catch(() => [] as string[]); for (const sub of subdirs) { + if (sub === opts.expectedVideoIdHint) continue; + if (preLaunchSubdirs.has(sub)) continue; + candidates.push(sub); + } + for (const sub of candidates) { const dir = path.join(dataDir, sub); const { partFile: leftover, finalFile } = await findExistingPartOrFinal(dir); diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -298,6 +298,15 @@ export async function downloadOneManaged( "--", opts.videoUrl, ]; + // Hint the orchestrator at which data/<id>/ subdir this launch will + // write to, so its .part discovery and final-file resolution don't + // latch onto stale .parts from prior interrupted attempts. For Rumble + // the on-disk dir is yt-dlp's internal id, only knowable after + // metadata.info.json appears — pass null and let the orchestrator + // fall back to the new-subdir heuristic. + const expectedVideoIdHint = isRumbleUrl(opts.videoUrl) + ? null + : extractVideoId(opts.videoUrl); const audioOutcome = await runAudioCheckedYtdlp({ paths: opts.paths, channelDir, @@ -306,6 +315,7 @@ export async function downloadOneManaged( onLog: opts.onLog, signal: opts.signal, archiveMarker: ARCHIVE_MARKER, + expectedVideoIdHint, }); primaryRes = { exitCode: audioOutcome.ytdlpExitCode, diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -22,4 +22,5 @@ - Channel page sticky/anchor offsets bumped to clear the sticky `StatusHeader`. The left video list and the pipeline stage rail previously pinned too high (`top: 1rem` and `top: 8rem` respectively) and sat partially behind the header. Stage anchors used a fixed `scroll-margin-top: 8rem` that wasn't enough on narrow widths where the header's mobile badge row wraps. The video-list and stage-rail sticky tops are now `top: 7rem` (lg+), and stage sections use a responsive `scroll-margin-top` (11rem on mobile, 7rem on lg+). - Clicking a mobile stage badge (or a stage rail item) now auto-expands the collapsible "Channel pipeline & settings" wrapper if the user had collapsed it. Previously the hash navigation succeeded but the target section was inside a closed `<details>` and not visible. The target is also re-scrolled into view after the expansion so it lands at the correct position. - Audio-check probe now flags `.part` files with heavy mid-stream codec corruption as malformed even when ffmpeg exits 0. Previously a stream could rack up hundreds of per-packet AAC decode failures (`Error submitting packet to decoder`) while ffmpeg still exited 0, and the file was promoted to a finished download. The classifier now also treats >5 decoder-error lines in stderr as malformed. +- Audio-check pipeline now scopes its `.part` discovery and final-probe to the video being downloaded, instead of scanning the whole channel `data/` tree and picking the first match. On channels with many in-progress `.part` files (cornbreadman: ~248), the orchestrator would latch onto a stale neighbour's `.part`, probe it instead of the current download, and then the success-path fallback would resolve a totally unrelated subdir's already-finalized `audio.mp3` as "the final file" — skipping transcode and letting the real `.mp4` finalize with no integrity check, only to fail noisily during the next transcode-untranscoded sweep. - Editor's `/changelog` page no longer renders unstyled. The `@source` directive in `editor/app/globals.css` pointed at a non-existent path, so Tailwind never scanned the shared `<Changelog>` component for class usage. diff --git a/editor/e2e/audio-check-scenarios.spec.ts b/editor/e2e/audio-check-scenarios.spec.ts @@ -272,6 +272,64 @@ test.describe("audio-checked download scenarios", () => { .toBe(true); }); + test("ignores stale .parts and final outputs from other video subdirs", async ({ + page, + }) => { + // Regression for the cornbreadman bug: with many in-progress .part + // files lingering under data/<channel>/, discoverPartFile used to scan + // the whole tree and latch onto the first match — often a stale .part + // from a prior interrupted download of a *different* video. The + // success-path fallback then resolved an unrelated subdir's + // audio.mp3 as "the final file", skipped transcode, and let the + // actual download finalize as audio.mp4 with no integrity probe. + // + // After the fix, the orchestrator scopes its discovery to the hinted + // subdir (extractVideoId returns null for "https://odysee.com/abc123", + // so this test exercises the new-subdir fallback) and ignores subdirs + // that already existed at launch time. + await resetData("audio-check-channel"); + const dataRoot = resolvePath(`${CHANNEL_ROOT}/data`); + // Both seeds sort before "abc123" (digits < letters), so the buggy + // first-match scan would prefer them. + await mkdir(`${dataRoot}/0aaaaaaa-stale-part`, { recursive: true }); + await writeFile( + `${dataRoot}/0aaaaaaa-stale-part/audio.mp4.part`, + "clean bytes from a prior interrupted download of a different video", + ); + await mkdir(`${dataRoot}/0zzzzzzz-already-done`, { recursive: true }); + await writeFile( + `${dataRoot}/0zzzzzzz-already-done/audio.mp3`, + "another video's already-finalized audio output", + ); + + await writeFakeConfig({ mode: "happy", totalChunks: 6, chunkDelayMs: 100 }); + await triggerDownload(page); + + const log = page.getByLabel("Download videos output"); + await expect(log).toContainText("audio-check download complete", { + timeout: 30_000, + }); + + const outcome = await waitForOutcome( + (o) => o.status === "ok-audio-checked", + "scoped-discovery outcome", + ); + expect(outcome.attempts[0].audioCheck?.rollbacks).toBe(0); + + // The real download was probed and transcoded under its own subdir. + expect(await pathExists(`${CHANNEL_ROOT}/data/${VIDEO_ID}/audio.mp3`)).toBe( + true, + ); + // The stale neighbours were left untouched — no checkpoint snapshots, + // no rollbacks, no accidental probes of the wrong final. + expect( + await pathExists(`${CHANNEL_ROOT}/data/0aaaaaaa-stale-part/audio.mp4.part`), + ).toBe(true); + expect( + await pathExists(`${CHANNEL_ROOT}/data/0zzzzzzz-already-done/audio.mp3`), + ).toBe(true); + }); + test("pre-check discards a malformed .part with no .good baseline", async ({ page, }) => {