Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit c628d37babb3c44be7a0b42f5a66a35b32082114
parent 84e00f7a2848c07b598acf659a22efbfed70a529
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 17 May 2026 11:44:38 -0400

odysee handling part 1

Diffstat:
Mcommon/ytdlp/ffmpegStreamClassify.ts | 64+++++++++++++++++++++-------------------------------------------
Mcommon/ytdlp/ffmpegStreamProbe.ts | 31++++++++++++++++++++++++-------
Mcommon/ytdlp/runYtdlp.ts | 9++++-----
Meditor/e2e/audio-check-classifier.spec.ts | 68++++++++++++++++++++++++++------------------------------------------
Meditor/e2e/channels-counts.spec.ts | 3++-
Meditor/e2e/export-search.spec.ts | 2+-
Meditor/e2e/fixtures/bin/fake-ffmpeg.mjs | 40++++++++++------------------------------
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 44++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/test-settings.default.json | 3++-
Meditor/e2e/pipeline.spec.ts | 10+++++++---
10 files changed, 141 insertions(+), 133 deletions(-)

diff --git a/common/ytdlp/ffmpegStreamClassify.ts b/common/ytdlp/ffmpegStreamClassify.ts @@ -1,53 +1,31 @@ -// Classifies the stderr output of an ffmpeg "decode-only" probe of a partial +// Classifies the result of an ffmpeg transcode-to-null probe of a partial // (or finalized) media file. Used by the audio-checked download flow to // detect mid-stream corruption that would render the final transcode useless. // +// The probe runs a real transcode (e.g. libmp3lame) to /dev/null. ffmpeg's +// exit code is the authoritative signal: it tolerates truncated containers +// (returns 0 with a "partial file" stderr line) but exits non-zero when a +// codec hits genuinely malformed data — e.g. AAC's "Sample rate index in +// program config element does not match". +// // Pure-function module: no side effects, no IO. Keeps test specs from having // to pull execa/etc. into the test process. export type StreamVerdict = "clean" | "partial" | "malformed"; -// Lines we treat as expected when probing an in-progress download (the -// trailing bytes are mid-frame, so ffmpeg complains as it hits EOF). These -// should not cause a rollback by themselves. -export const PARTIAL_PATTERNS: readonly RegExp[] = [ - /Invalid data found when processing input/i, - /Truncating packet/i, - /End of file/i, - /unexpected end of file/i, - /could not find codec parameters/i, - /moov atom not found/i, - /Partial frame/i, -]; - -// Lines we treat as definitive corruption signals. Drawn from real-world -// libavcodec output observed against malformed Odysee streams. -export const MALFORMED_PATTERNS: readonly RegExp[] = [ - /Sample rate index in program config element does not match/i, - /Error while decoding stream/i, - /corrupt/i, - /invalid NAL/i, - /non[- ]?existing PPS/i, - /decode_slice_header error/i, - /channel element .* is not allocated/i, - /Number of bands \(\d+\) exceeds limit/i, - /reference picture missing/i, -]; +// Stderr line that ffmpeg emits when the input container is cut off +// mid-frame but the encoder still got enough decoded audio to produce a +// valid output. Presence of this line means the file is partial-but-clean. +export const PARTIAL_STDERR_PATTERN = /partial file/i; -export function classifyFfmpegStderr(stderr: string): StreamVerdict { - const lines = stderr - .split(/\r?\n/) - .map((l) => l.trim()) - .filter((l) => l.length > 0); - if (lines.length === 0) return "clean"; - let sawPartial = false; - for (const line of lines) { - if (MALFORMED_PATTERNS.some((p) => p.test(line))) return "malformed"; - if (PARTIAL_PATTERNS.some((p) => p.test(line))) { - sawPartial = true; - continue; - } - return "malformed"; - } - return sawPartial ? "partial" : "clean"; +// Exit-code-driven classification. The transcode probe is the authoritative +// signal; stderr is consulted only to distinguish a clean tail (no errors) +// from a partial-but-OK tail (ffmpeg reported "partial file" but encoded +// the available audio fine). +export function classifyFfmpegProbe( + exitCode: number | null, + stderr: string, +): StreamVerdict { + if (exitCode !== 0) return "malformed"; + return PARTIAL_STDERR_PATTERN.test(stderr) ? "partial" : "clean"; } diff --git a/common/ytdlp/ffmpegStreamProbe.ts b/common/ytdlp/ffmpegStreamProbe.ts @@ -1,6 +1,6 @@ import { execa } from "execa"; import { - classifyFfmpegStderr, + classifyFfmpegProbe, type StreamVerdict, } from "./ffmpegStreamClassify"; @@ -9,6 +9,10 @@ export type ProbeAudioStreamOptions = { file: string; signal: AbortSignal; onLog?: (s: string) => void; + // Target audio codec for the probe transcode. We mirror the channel's + // output codec so the probe exercises the same encode path the final + // transcode will use. Defaults to libmp3lame. + codecArg?: string; }; export type ProbeAudioStreamResult = { @@ -17,19 +21,32 @@ export type ProbeAudioStreamResult = { exitCode: number | null; }; +// Probe by performing a real audio transcode to /dev/null. The exit code is +// the authoritative signal: +// - 0 → clean (or partial-but-decodable; either is fine for our flow) +// - != 0 → malformed (a real codec error tripped ffmpeg) +// ffmpeg treats a truncated container as a recoverable warning ("partial +// file" stderr line, exit 0). Only true mid-stream corruption — e.g. AAC's +// "Sample rate index in program config element does not match" — produces a +// non-zero exit. -xerror with -f null underreports because the null muxer +// swallows decoder errors during write. export async function probeAudioStream( opts: ProbeAudioStreamOptions, ): Promise<ProbeAudioStreamResult> { + const codecArg = opts.codecArg ?? "libmp3lame"; const args = [ + "-y", "-v", "error", "-nostdin", - "-xerror", "-i", opts.file, + "-vn", + "-c:a", + codecArg, "-f", - "null", - "-", + "mp3", + "/dev/null", ]; opts.onLog?.(`$ ${opts.ffmpegBin} ${args.join(" ")}\n`); const child = execa(opts.ffmpegBin, args, { @@ -44,12 +61,12 @@ export async function probeAudioStream( stderr += chunk; opts.onLog?.(chunk); }); - // Discard stdout (-f null -); we only care about stderr. child.stdout?.on("data", () => {}); const result = await child; + const exitCode = result.exitCode ?? null; return { - verdict: classifyFfmpegStderr(stderr), + verdict: classifyFfmpegProbe(exitCode, stderr), stderr, - exitCode: result.exitCode ?? null, + exitCode, }; } diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -364,11 +364,10 @@ async function downloadPlaylistManaged( if (firstFailure && abortOnError && !opts.signal.aborted) { throw firstFailure; } - if (failedCount > 0) { - opts.onLog( - `Managed download complete with ${failedCount} failed video(s) (abort-on-error=${abortOnError}).\n`, - ); - } + const okCount = processedCount - failedCount; + opts.onLog( + `Managed download complete: ${okCount} succeeded, ${failedCount} failed (${processedCount}/${items.length} processed).\n`, + ); await touchLastFullDownload(opts); await safeBackfillAvailability(opts); diff --git a/editor/e2e/audio-check-classifier.spec.ts b/editor/e2e/audio-check-classifier.spec.ts @@ -1,61 +1,45 @@ -// Pure-function tests for the ffmpeg stderr classifier. These don't need the -// dev server, fixtures, or a browser — but the project uses Playwright for -// everything, so they live here too. +// Pure-function tests for the ffmpeg probe-result classifier. These don't +// need the dev server, fixtures, or a browser — but the project uses +// Playwright for everything, so they live here too. import { test, expect } from "@playwright/test"; -import { classifyFfmpegStderr } from "../../common/ytdlp/ffmpegStreamClassify"; +import { classifyFfmpegProbe } from "../../common/ytdlp/ffmpegStreamClassify"; -test.describe("classifyFfmpegStderr", () => { - test("empty stderr is clean", () => { - expect(classifyFfmpegStderr("")).toBe("clean"); - expect(classifyFfmpegStderr(" \n\n\t\n")).toBe("clean"); +test.describe("classifyFfmpegProbe", () => { + test("exit 0 with empty stderr → clean", () => { + expect(classifyFfmpegProbe(0, "")).toBe("clean"); + expect(classifyFfmpegProbe(0, "\n \t\n")).toBe("clean"); }); - test("only partial-stream patterns → partial", () => { + test("exit 0 with 'partial file' stderr → partial", () => { expect( - classifyFfmpegStderr( - "[mov,mp4 @ 0x123] Invalid data found when processing input at EOF", - ), - ).toBe("partial"); - expect(classifyFfmpegStderr("[aac @ 0x1] Truncating packet of size 137")).toBe( - "partial", - ); - expect( - classifyFfmpegStderr( - "Invalid data found when processing input at EOF\nTruncating packet of size 200", + classifyFfmpegProbe( + 0, + "[in#0/mov,mp4,m4a,3gp,3g2,mj2 @ 0x1234] stream 1, offset 0x10483924: partial file\n", ), ).toBe("partial"); }); - test("known malformed patterns → malformed", () => { + test("exit 0 even with noisy decoder warnings → clean (partial detected only via 'partial file')", () => { + // ffmpeg often prints decoder warnings for a partial container without + // any 'partial file' line — exit code is still 0 and the encoder still + // produces output. We trust the exit code. expect( - classifyFfmpegStderr( - "[aac @ 0x1] Sample rate index in program config element does not match the sample rate index configured by the container.", + classifyFfmpegProbe( + 0, + "[h264 @ 0x1] Invalid NAL unit size (17306 > 1576).\n[h264 @ 0x1] missing picture in access unit with size 1586\n", ), - ).toBe("malformed"); - expect(classifyFfmpegStderr("Error while decoding stream #0:0")).toBe( - "malformed", - ); - expect(classifyFfmpegStderr("[mp4 @ 0x1] corrupt frame")).toBe("malformed"); - expect(classifyFfmpegStderr("[h264 @ 0x1] invalid NAL unit size")).toBe( - "malformed", - ); + ).toBe("clean"); }); - test("mixed partial + malformed → malformed wins", () => { + test("non-zero exit → malformed (regardless of stderr)", () => { expect( - classifyFfmpegStderr( - "Truncating packet of size 137\n[aac @ 0x1] Error while decoding stream #0:0\nInvalid data found when processing input at EOF", + classifyFfmpegProbe( + 1, + "[aac @ 0x1] Sample rate index in program config element does not match the sample rate index configured by the container.\n", ), ).toBe("malformed"); - }); - - test("unrecognised error → malformed (conservative)", () => { - expect(classifyFfmpegStderr("Something completely unexpected")).toBe( - "malformed", - ); - expect( - classifyFfmpegStderr("FormatIO failed reading from socket"), - ).toBe("malformed"); + expect(classifyFfmpegProbe(2, "")).toBe("malformed"); + expect(classifyFfmpegProbe(null, "killed by signal")).toBe("malformed"); }); }); diff --git a/editor/e2e/channels-counts.spec.ts b/editor/e2e/channels-counts.spec.ts @@ -28,6 +28,7 @@ test("shows playlist length when playlist exists", async ({ page }) => { }); test("counts update after store-playlist and download", async ({ page }) => { + test.setTimeout(120_000); await resetData("test-pipeline"); await page.goto("/channels"); await expect( @@ -59,7 +60,7 @@ test("counts update after store-playlist and download", async ({ page }) => { await page.getByRole("button", { name: "Download from playlist" }).click(); await expect( page.getByLabel("Download from playlist output"), - ).toContainText("download complete", { timeout: 30_000 }); + ).toContainText("Managed download complete", { timeout: 60_000 }); await page.goto("/channels"); await expect( diff --git a/editor/e2e/export-search.spec.ts b/editor/e2e/export-search.spec.ts @@ -1,6 +1,6 @@ import { test, expect, type Page } from "@playwright/test"; -const EXPORT_BASE = `http://localhost:${process.env.EXPORT_PORT ?? 3000}`; +const EXPORT_BASE = `http://localhost:${process.env.EXPORT_PORT ?? 3010}`; type Summary = { slug: string; diff --git a/editor/e2e/fixtures/bin/fake-ffmpeg.mjs b/editor/e2e/fixtures/bin/fake-ffmpeg.mjs @@ -5,11 +5,9 @@ // -> writes a small placeholder to <out>. If <src> contains the // CORRUPT_MARKER sentinel, exits 1 with a libavcodec-style stderr // line so the audio-check classifier flags it as malformed. -// -// 2. Probe: ffmpeg -v error -nostdin -xerror -i <file> -f null - -// -> reads <file>; if it contains CORRUPT_MARKER, writes a -// libavcodec-style error to stderr and exits 1. Else exits 0 -// silently. Used by the audio-checked download orchestrator. +// -> If <out> is /dev/null (the audio-checked probe), behaves the +// same w.r.t. corruption detection but doesn't actually write +// output. import { readFile, writeFile } from "node:fs/promises"; const CORRUPT_MARKER = "__CORRUPT__"; @@ -32,31 +30,6 @@ async function fileHasCorruptMarker(file) { } const src = arg("-i") ?? "<unknown>"; - -// Probe shape: ends with `-f null -`. Stdout suppressed; only stderr matters. -const last = argv[argv.length - 1]; -const secondLast = argv[argv.length - 2]; -if (secondLast === "-f" && (last === "null" || last === "null,")) { - if (await fileHasCorruptMarker(src)) { - process.stderr.write( - `[aac @ 0x1234] Sample rate index in program config element does not match the sample rate index configured by the container.\n`, - ); - process.exit(1); - } - process.exit(0); -} -// Also accept the older form `-f null -` with `-` as last positional. -if (last === "-" && argv[argv.length - 3] === "-f" && secondLast === "null") { - if (await fileHasCorruptMarker(src)) { - process.stderr.write( - `[aac @ 0x1234] Sample rate index in program config element does not match the sample rate index configured by the container.\n`, - ); - process.exit(1); - } - process.exit(0); -} - -// Transcode shape: out is the last positional (not a flag). const out = argv[argv.length - 1]; if (!out || out.startsWith("-")) { process.stderr.write(`[fake-ffmpeg] missing output path\n`); @@ -70,5 +43,12 @@ if (await fileHasCorruptMarker(src)) { process.exit(1); } +if (out === "/dev/null") { + // Probe invocation — no output written. Real ffmpeg would also print a + // "partial file" warning here if the input was truncated; we don't need + // to simulate that for the exit-code-driven classifier. + process.exit(0); +} + await writeFile(out, `fake-ffmpeg transcoded from ${src}\n`); process.stdout.write(`[fake-ffmpeg] wrote ${out}\n`); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -283,6 +283,32 @@ async function modeSync(url) { await downloadOne("https://www.youtube.com/watch?v=fakeSync0001"); } +// Managed YouTube-handling single-URL download (one per video, as +// downloadOneManaged invokes per URL): write metadata + transcript for the +// URL's id and print the DLOM_ARCHIVE line. The archive itself is appended +// by downloadOneManaged based on the printed line, so the fake must NOT +// write to ./archive. +async function modeYoutubeSingleUrlManaged(url) { + const id = urlIdYouTube(url); + if (!id) { + process.stderr.write(`[fake-ytdlp] could not derive id from ${url}\n`); + process.exit(2); + } + const videoDir = path.join("data", id); + await ensureDir(videoDir); + await appendFile( + "fake-ytdlp.invocations", + `youtube-single:${url}\n`, + ); + process.stdout.write(`[fake-ytdlp] managed single-URL ${id}\n`); + if (!existsSync(path.join(videoDir, "metadata.info.json"))) { + await writeMetadata(videoDir, id); + } + await writeTranscript(videoDir); + process.stdout.write(`DLOM_ARCHIVE youtube ${id}\n`); + process.stdout.write(`[fake-ytdlp] single-url fetched ${id}\n`); +} + function sleep(ms) { return new Promise((resolve) => setTimeout(resolve, ms)); } @@ -394,6 +420,24 @@ async function main() { return; } + // Managed YouTube-handling per-URL invocation: --skip-download with + // --write-auto-subs, no -a, last positional is the URL. + if ( + has("--skip-download") && + has("--write-auto-subs") && + !has("--flat-playlist") + ) { + const url = lastNonFlag(); + if (!url) { + process.stderr.write( + `[fake-ytdlp] youtube single-url mode missing URL\n`, + ); + process.exit(2); + } + await modeYoutubeSingleUrlManaged(url); + return; + } + process.stderr.write( `[fake-ytdlp] unknown invocation: ${argv.join(" ")}\n`, ); diff --git a/editor/e2e/fixtures/test-settings.default.json b/editor/e2e/fixtures/test-settings.default.json @@ -3,5 +3,6 @@ "siteDescription": "Test description", "headerTitle": "Test Browser", "homeTagline": "", - "maxTranscriptPageBytes": 8388608 + "maxTranscriptPageBytes": 8388608, + "sleepBetweenDownloadsSeconds": 0 } diff --git a/editor/e2e/pipeline.spec.ts b/editor/e2e/pipeline.spec.ts @@ -18,6 +18,7 @@ test("store playlist writes a playlist file with 5 URLs", async ({ page }) => { test("download from playlist fetches all 5 URLs and writes the archive", async ({ page, }) => { + test.setTimeout(120_000); await resetData("test-pipeline"); await page.goto("/channels/test-pipeline"); await page.getByRole("button", { name: "Store playlist" }).click(); @@ -31,7 +32,9 @@ test("download from playlist fetches all 5 URLs and writes the archive", async ( await expect(downloadLog).toContainText("5 new, 0 already archived", { timeout: 30_000, }); - await expect(downloadLog).toContainText("download complete"); + await expect(downloadLog).toContainText("Managed download complete", { + timeout: 60_000, + }); for (let i = 1; i <= 5; i++) { expect( @@ -47,6 +50,7 @@ test("download from playlist fetches all 5 URLs and writes the archive", async ( }); test("re-running download skips already-archived entries", async ({ page }) => { + test.setTimeout(120_000); await resetData("test-pipeline"); await page.goto("/channels/test-pipeline"); await page.getByRole("button", { name: "Store playlist" }).click(); @@ -57,8 +61,8 @@ test("re-running download skips already-archived entries", async ({ page }) => { await page.getByRole("button", { name: "Download from playlist" }).click(); const downloadLog = page.getByLabel("Download from playlist output"); - await expect(downloadLog).toContainText("download complete", { - timeout: 30_000, + await expect(downloadLog).toContainText("Managed download complete", { + timeout: 60_000, }); await page.getByRole("button", { name: "Download from playlist" }).click();