commit 485ce9904d31b7f85ba0373e0bdd628b3b3f7a3b
parent add79a2fc87ea8949bc3178c6b1d3f66807458c0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 13 Jun 2026 17:04:23 -0400
reconstruct single video URL from filename when no metadata
Diffstat:
6 files changed, 72 insertions(+), 11 deletions(-)
diff --git a/common/controller/undownloadedVideos.ts b/common/controller/undownloadedVideos.ts
@@ -1,6 +1,8 @@
import path from "node:path";
import { readFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import { defaultWebpageUrl, detectPlatform } from "../lib/platform";
import { extractVideoId } from "../ytdlp/runYtdlp";
async function readPlaylistUrls(playlistPath: string): Promise<string[]> {
@@ -20,6 +22,7 @@ export async function findVideoSourceUrl(
paths: Paths,
slug: string,
videoId: string,
+ config?: ChannelConfig,
): Promise<string | null> {
const channelRoot = path.join(paths.channelsDir, slug);
const metaPath = path.join(channelRoot, "data", videoId, "metadata.info.json");
@@ -32,12 +35,17 @@ export async function findVideoSourceUrl(
} catch {
// fall through to playlist scan
}
- const urls = await readPlaylistUrls(path.join(channelRoot, "playlist"));
- if (urls.length === 0) return null;
// Post-reconcile a video's dir name is its canonical id, so match the
// requested videoId against each URL's canonical id directly.
+ const urls = await readPlaylistUrls(path.join(channelRoot, "playlist"));
for (const url of urls) {
if (extractVideoId(url) === videoId) return url;
}
+ // Last resort: re-create the URL from the canonical id (the dir name) and the
+ // channel's platform. Only fires when both metadata.info.json and the
+ // playlist came up empty, so the exact original URL always wins when present.
+ // Requires a known platform — we don't blindly guess one.
+ const platform = config?.platform ?? detectPlatform(config?.url ?? null);
+ if (platform) return defaultWebpageUrl(platform, videoId);
return null;
}
diff --git a/common/lib/platform.ts b/common/lib/platform.ts
@@ -23,6 +23,18 @@ export function detectPlatform(
return null;
}
+// Best-effort canonical webpage URL for a video given its platform and
+// canonical id. Used both as a metadata fallback in summarize() and to
+// re-create a download URL when a video has no metadata.info.json. Note: for
+// Odysee the canonical id is only the claim hash (the channel-name prefix is
+// dropped by extractVideoId), so a reconstructed Odysee URL may not resolve.
+export function defaultWebpageUrl(platform: Platform, id: string): string {
+ if (platform === "rumble") return `https://rumble.com/${id}`;
+ if (platform === "odysee") return `https://odysee.com/${id}`;
+ if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`;
+ return `https://www.youtube.com/watch?v=${id}`;
+}
+
export function platformQueueKey(
platform: Platform | null | undefined,
): string {
diff --git a/common/lib/transcripts-server.ts b/common/lib/transcripts-server.ts
@@ -1,6 +1,7 @@
import { readFile } from "node:fs/promises";
import path from "node:path";
import { formatDate, formatDuration } from "./format";
+import { defaultWebpageUrl } from "./platform";
import type { DisplaySummary, Platform, TranscriptSummary } from "./transcripts";
import type { VideoStat } from "./stats";
@@ -62,13 +63,6 @@ function detectPlatform(meta: RawMetadata): Platform {
return "youtube";
}
-function defaultWebpageUrl(platform: Platform, id: string): string {
- if (platform === "rumble") return `https://rumble.com/${id}`;
- if (platform === "odysee") return `https://odysee.com/${id}`;
- if (platform === "twitch") return `https://www.twitch.tv/videos/${id}`;
- return `https://www.youtube.com/watch?v=${id}`;
-}
-
export function summarize(
channelSlug: string,
videoDir: string,
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **The single-video download pipeline reconstructs the source URL when a video has no `metadata.info.json`.** Running the per-video **Download** / **Audio + Whisper** action resolves the video's URL by reading its `metadata.info.json` (`webpage_url`), then by matching the canonical id against the channel's stored `playlist`. When a video directory exists but has neither — e.g. a manually-placed video, or a channel whose playlist was never stored — the pipeline previously gave up with "Could not determine the video URL". It now re-creates the URL from the video's canonical id (its directory name) and the channel's platform (from `config.platform`, falling back to detecting it from `config.url`), so the download proceeds. Reconstruction is a last resort — the exact URL from metadata or the playlist always wins when present — and only fires when the platform is known (no blind guess); note a reconstructed Odysee URL drops the channel-name prefix, so it may not resolve.
- **New transcription app: parakeet.cpp with overlapping-segment stitching.** A third **App** option (alongside whisper.cpp and chough) in **Settings → Transcription** transcribes via `parakeet-cli`, working around its two long-audio limitations: it only accepts a single WAV, and a >4GB PCM stream silently yields an empty transcript (dr_wav's data-chunk size is 32-bit). A standalone wrapper (`scripts/parakeet-stitch.mjs`) — usable both as the app's binary and directly on the CLI — slices the source into **overlapping** 16kHz-mono windows (the format parakeet-cli resamples to internally anyway), runs `parakeet-cli --json --timestamps` on each, then stitches the per-window word timestamps back into one transcript. The overlap guarantees a word clipped at one window's boundary is captured whole by the neighbour; a cut point in the middle of each overlap region decides which window owns each word, so nothing is dropped or duplicated across seams. Output is chough-native JSON (seconds-based `chunk_data`), so it flows through the existing format-aware normalize/index path with no downstream changes and is tagged `chough-json` in `transcript.cues.json`. The settings fields are **Model** (the `.gguf` path) and **Chunk size** (per-window length, default 480s); the underlying `parakeet-cli` binary and default model resolve from `PARAKEET_CLI` / `PARAKEET_MODEL` (overlap is tunable via `PARAKEET_OVERLAP_SEC`). Run on the CLI with `scripts/parakeet-stitch.mjs --model <gguf> <audio> [out.json]` (prints JSON to stdout when no output path is given). Per-window progress drives the job progress bars.
- **Batch jobs estimate the time remaining.** A running batch's overall progress bar on `/jobs/active` now shows an estimate of how long is left (e.g. `~4:30 left`), alongside the existing `Transcripts: 12 / 50` count. The estimate is **remaining tasks × the measured average time per task**: each download/transcription's real wall-clock duration is folded into per-job running totals as it finishes, and that average is converted to a wall-clock figure using the parallelism observed so far (so a 4-way parallel transcribe batch isn't estimated as if it ran one at a time). It reads as **estimating…** until the first sub-operation completes (no average yet), and disappears once no work remains. Jobs without per-task tracking (e.g. storing a playlist) show no estimate.
- **Audio-checked downloads recover correctly when an interrupted attempt left a malformed `.part` and the video id isn't derivable from its URL.** For sources where the canonical id only appears after metadata (e.g. Odysee), the integrity-checked downloader's pre-check would correctly roll back a corrupt leftover `.part` to its `.good` snapshot (or discard it), but the post-launch file discovery then ignored that same directory as a "pre-existing" one — so the freshly re-downloaded audio was never found and the download was recorded as failed. The pre-check now reports the directory it acted on, and discovery scopes to it, so the resumed download finalizes as `ok-audio-checked`. Unrelated stale `.part`s from other videos' interrupted attempts are still ignored (clean pre-existing parts aren't reported), so the cross-video protection is unchanged.
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -134,7 +134,7 @@ export async function downloadVideoPipelineAction(
const r = await loadConfigOrError(slug);
if (!r.ok) return r;
const paths = getPaths();
- const url = await findVideoSourceUrl(paths, slug, videoId);
+ const url = await findVideoSourceUrl(paths, slug, videoId, r.config);
if (!url) {
return {
ok: false,
@@ -197,7 +197,7 @@ export async function whisperVideoAction(
const entries = await readdir(videoDir).catch(() => [] as string[]);
const hasAudio = entries.some(isRealAudioFile);
if (!hasAudio) {
- const url = await findVideoSourceUrl(paths, slug, videoId);
+ const url = await findVideoSourceUrl(paths, slug, videoId, r.config);
if (!url) {
throw new Error(
"Could not determine the video URL: no metadata.info.json and the playlist does not contain a matching entry.",
diff --git a/editor/e2e/reconstruct-download-url.spec.ts b/editor/e2e/reconstruct-download-url.spec.ts
@@ -0,0 +1,46 @@
+// When a video has no metadata.info.json and the channel has no playlist entry
+// for it, the single-video download pipeline reconstructs the source URL from
+// the video's canonical id (its dir name) and the channel's platform — derived
+// here from the channel config's odysee `url`. Without this, the pipeline used
+// to fail with "Could not determine the video URL".
+
+import { readFile, rm } from "node:fs/promises";
+import { test, expect } from "@playwright/test";
+import { pathExists, resetData, resolvePath } from "./helpers";
+
+const CHANNEL = "test-transcribe";
+const CHANNEL_ROOT = `test-transcripts/channels/${CHANNEL}`;
+
+test("download pipeline reconstructs the URL when metadata + playlist are missing", async ({
+ page,
+}) => {
+ test.setTimeout(60_000);
+ await resetData("one-transcribe-channel-with-audio");
+ // Strip vidA down to an empty dir: no audio, no metadata.info.json. The
+ // fixture also ships no `playlist` file, so neither URL source exists. The
+ // channel config's `url` (https://odysee.com/@example) is the only platform
+ // signal, so reconstruction must produce an odysee URL for canonical id vidA.
+ await rm(resolvePath(`${CHANNEL_ROOT}/data/vidA/audio.m4a`), { force: true });
+
+ await page.goto(`/channels/${CHANNEL}/videos/vidA`);
+ await page.getByRole("button", { name: /^Run download pipeline$/ }).click();
+
+ const log = page.getByLabel("Run download pipeline for vidA output");
+ // The reconstructed odysee URL flows straight into yt-dlp's invocation.
+ await expect(log).toContainText("https://odysee.com/vidA", {
+ timeout: 30_000,
+ });
+ await expect(log).not.toContainText("Could not determine the video URL");
+ // Wait for the download to actually finish before inspecting the filesystem.
+ await expect(log).toContainText("download complete", { timeout: 30_000 });
+
+ // End-to-end: the audio is actually (re)downloaded under the canonical id.
+ expect(await pathExists(`${CHANNEL_ROOT}/data/vidA/audio.m4a`)).toBe(true);
+
+ // Belt-and-suspenders: the fake yt-dlp recorded the reconstructed URL.
+ const invocations = await readFile(
+ resolvePath(`${CHANNEL_ROOT}/fake-ytdlp.invocations`),
+ "utf8",
+ );
+ expect(invocations).toContain("https://odysee.com/vidA");
+});