Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 485ce9904d31b7f85ba0373e0bdd628b3b3f7a3b
parent add79a2fc87ea8949bc3178c6b1d3f66807458c0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat, 13 Jun 2026 17:04:23 -0400

reconstruct single video URL from filename when no metadata

Diffstat:
Mcommon/controller/undownloadedVideos.ts | 12++++++++++--
Mcommon/lib/platform.ts | 12++++++++++++
Mcommon/lib/transcripts-server.ts | 8+-------
Meditor/CHANGELOG.md | 1+
Meditor/app/channels/[slug]/videos/[id]/videoActions.ts | 4++--
Aeditor/e2e/reconstruct-download-url.spec.ts | 46++++++++++++++++++++++++++++++++++++++++++++++
6 files changed, 72 insertions(+), 11 deletions(-)

diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts @@ -1,6 +1,8 @@ import path from "node:path"; import { readFile } from "node:fs/promises"; import type { Paths } from "../lib/paths"; +import type { ChannelConfig } from "../lib/channelConfig"; +import { defaultWebpageUrl, detectPlatform } from "../lib/platform"; import { extractVideoId } from "../ytdlp/runYtdlp"; async function readPlaylistUrls(playlistPath: string): Promise<string[]> { @@ -20,6 +22,7 @@ export async function findVideoSourceUrl( paths: Paths, slug: string, videoId: string, + config?: ChannelConfig, ): Promise<string | null> { const channelRoot = path.join(paths.channelsDir, slug); const metaPath = path.join(channelRoot, "data", videoId, "metadata.info.json"); @@ -32,12 +35,17 @@ export async function findVideoSourceUrl( } catch { // fall through to playlist scan } - const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); - if (urls.length === 0) return null; // Post-reconcile a video's dir name is its canonical id, so match the // requested videoId against each URL's canonical id directly. + const urls = await readPlaylistUrls(path.join(channelRoot, "playlist")); for (const url of urls) { if (extractVideoId(url) === videoId) return url; } + // Last resort: re-create the URL from the canonical id (the dir name) and the + // channel's platform. Only fires when both metadata.info.json and the + // playlist came up empty, so the exact original URL always wins when present. + // Requires a known platform — we don't blindly guess one. + const platform = config?.platform ?? detectPlatform(config?.url ?? null); + if (platform) return defaultWebpageUrl(platform, videoId); return null; } diff --git a/common/lib/platform.ts b/common/lib/platform.ts @@ -23,6 +23,18 @@ export function detectPlatform( return null; } +// Best-effort canonical webpage URL for a video given its platform and +// canonical id. Used both as a metadata fallback in summarize() and to +// re-create a download URL when a video has no metadata.info.json. Note: for +// Odysee the canonical id is only the claim hash (the channel-name prefix is +// dropped by extractVideoId), so a reconstructed Odysee URL may not resolve. +export function defaultWebpageUrl(platform: Platform, id: string): string { + if (platform === "rumble") return `https://rumble.com/${id}`; + if (platform === "odysee") return `https://odysee.com/${id}`; + if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`; + return `https://www.youtube.com/watch?v=${id}`; +} + export function platformQueueKey( platform: Platform | null | undefined, ): string { diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts @@ -1,6 +1,7 @@ import { readFile } from "node:fs/promises"; import path from "node:path"; import { formatDate, formatDuration } from "./format"; +import { defaultWebpageUrl } from "./platform"; import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts"; import type { VideoStat } from "./stats"; @@ -62,13 +63,6 @@ function detectPlatform(meta: RawMetadata): Platform { return "youtube"; } -function defaultWebpageUrl(platform: Platform, id: string): string { - if (platform === "rumble") return `https://rumble.com/${id}`; - if (platform === "odysee") return `https://odysee.com/${id}`; - if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`; - return `https://www.youtube.com/watch?v=${id}`; -} - export function summarize( channelSlug: string, videoDir: string, diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **The single-video download pipeline reconstructs the source URL when a video has no `metadata.info.json`.** Running the per-video **Download** / **Audio + Whisper** action resolves the video's URL by reading its `metadata.info.json` (`webpage_url`), then by matching the canonical id against the channel's stored `playlist`. When a video directory exists but has neither — e.g. a manually-placed video, or a channel whose playlist was never stored — the pipeline previously gave up with "Could not determine the video URL". It now re-creates the URL from the video's canonical id (its directory name) and the channel's platform (from `config.platform`, falling back to detecting it from `config.url`), so the download proceeds. Reconstruction is a last resort — the exact URL from metadata or the playlist always wins when present — and only fires when the platform is known (no blind guess); note a reconstructed Odysee URL drops the channel-name prefix, so it may not resolve. - **New transcription app: parakeet.cpp with overlapping-segment stitching.** A third **App** option (alongside whisper.cpp and chough) in **Settings → Transcription** transcribes via `parakeet-cli`, working around its two long-audio limitations: it only accepts a single WAV, and a >4GB PCM stream silently yields an empty transcript (dr_wav's data-chunk size is 32-bit). A standalone wrapper (`scripts/parakeet-stitch.mjs`) — usable both as the app's binary and directly on the CLI — slices the source into **overlapping** 16kHz-mono windows (the format parakeet-cli resamples to internally anyway), runs `parakeet-cli --json --timestamps` on each, then stitches the per-window word timestamps back into one transcript. The overlap guarantees a word clipped at one window's boundary is captured whole by the neighbour; a cut point in the middle of each overlap region decides which window owns each word, so nothing is dropped or duplicated across seams. Output is chough-native JSON (seconds-based `chunk_data`), so it flows through the existing format-aware normalize/index path with no downstream changes and is tagged `chough-json` in `transcript.cues.json`. The settings fields are **Model** (the `.gguf` path) and **Chunk size** (per-window length, default 480s); the underlying `parakeet-cli` binary and default model resolve from `PARAKEET_CLI` / `PARAKEET_MODEL` (overlap is tunable via `PARAKEET_OVERLAP_SEC`). Run on the CLI with `scripts/parakeet-stitch.mjs --model <gguf> <audio> [out.json]` (prints JSON to stdout when no output path is given). Per-window progress drives the job progress bars. - **Batch jobs estimate the time remaining.** A running batch's overall progress bar on `/jobs/active` now shows an estimate of how long is left (e.g. `~4:30 left`), alongside the existing `Transcripts: 12 / 50` count. The estimate is **remaining tasks × the measured average time per task**: each download/transcription's real wall-clock duration is folded into per-job running totals as it finishes, and that average is converted to a wall-clock figure using the parallelism observed so far (so a 4-way parallel transcribe batch isn't estimated as if it ran one at a time). It reads as **estimating…** until the first sub-operation completes (no average yet), and disappears once no work remains. Jobs without per-task tracking (e.g. storing a playlist) show no estimate. - **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged. diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts @@ -134,7 +134,7 @@ export async function downloadVideoPipelineAction( const r = await loadConfigOrError(slug); if (!r.ok) return r; const paths = getPaths(); - const url = await findVideoSourceUrl(paths, slug, videoId); + const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { return { ok: false, @@ -197,7 +197,7 @@ export async function whisperVideoAction( const entries = await readdir(videoDir).catch(() => [] as string[]); const hasAudio = entries.some(isRealAudioFile); if (!hasAudio) { - const url = await findVideoSourceUrl(paths, slug, videoId); + const url = await findVideoSourceUrl(paths, slug, videoId, r.config); if (!url) { throw new Error( "Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.", diff --git a/editor/e2e/reconstruct-download-url.spec.ts b/editor/e2e/reconstruct-download-url.spec.ts @@ -0,0 +1,46 @@ +// When a video has no metadata.info.json and the channel has no playlist entry +// for it, the single-video download pipeline reconstructs the source URL from +// the video's canonical id (its dir name) and the channel's platform — derived +// here from the channel config's odysee `url`. Without this, the pipeline used +// to fail with "Could not determine the video URL". + +import { readFile, rm } from "node:fs/promises"; +import { test, expect } from "@playwright/test"; +import { pathExists, resetData, resolvePath } from "./helpers"; + +const CHANNEL = "test-transcribe"; +const CHANNEL_ROOT = `test-transcripts/channels/${CHANNEL}`; + +test("download pipeline reconstructs the URL when metadata + playlist are missing", async ({ + page, +}) => { + test.setTimeout(60_000); + await resetData("one-transcribe-channel-with-audio"); + // Strip vidA down to an empty dir: no audio, no metadata.info.json. The + // fixture also ships no `playlist` file, so neither URL source exists. The + // channel config's `url` (https://odysee.com/@example) is the only platform + // signal, so reconstruction must produce an odysee URL for canonical id vidA. + await rm(resolvePath(`${CHANNEL_ROOT}/data/vidA/audio.m4a`), { force: true }); + + await page.goto(`/channels/${CHANNEL}/videos/vidA`); + await page.getByRole("button", { name: /^Run download pipeline$/ }).click(); + + const log = page.getByLabel("Run download pipeline for vidA output"); + // The reconstructed odysee URL flows straight into yt-dlp's invocation. + await expect(log).toContainText("https://odysee.com/vidA", { + timeout: 30_000, + }); + await expect(log).not.toContainText("Could not determine the video URL"); + // Wait for the download to actually finish before inspecting the filesystem. + await expect(log).toContainText("download complete", { timeout: 30_000 }); + + // End-to-end: the audio is actually (re)downloaded under the canonical id. + expect(await pathExists(`${CHANNEL_ROOT}/data/vidA/audio.m4a`)).toBe(true); + + // Belt-and-suspenders: the fake yt-dlp recorded the reconstructed URL. + const invocations = await readFile( + resolvePath(`${CHANNEL_ROOT}/fake-ytdlp.invocations`), + "utf8", + ); + expect(invocations).toContain("https://odysee.com/vidA"); +});