commit 400adc37096fffe053b191ec8a679f19bd887577
parent 5221cddf6656fd323c88f4a84e94ebe224799a98
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 20 Sep 2026 19:14:47 -0400
clip windows: a fetched window lives beside the video, with provenance
umtool's clip bench runs yt-dlp itself, into its own project-local
out/clips-raw. Every tool that wants the same seconds pays for them again, and
none of those fetches go through the editor's cookie policy, its per-platform
sleeps or its 429 cooldown. This is the common half of moving that fetch into
the editor: a window lands in the corpus at
channels/<slug>/data/<id>/clips/<from>-<to>.mp4
with a sidecar saying who asked and why.
The load-bearing rule is that the fetch never writes metadata.info.json — the
index keys a video's presence on that file — so fetchWindowManaged assembles
its own argv rather than going through outputArgsForUrl or downloadOneManaged's
prefetch. Its format selector is umtool's, byte for byte, because a window
either side fetches has to be the same file or the two caches diverge; the
comment says why it is not downloadFormatPreset's.
runOneYtdlp and the per-channel argv lift out of downloadOneManaged (which
delegates, unchanged) so the window fetch shares them instead of growing a
second copy that drifts.
reconcileVideoDirs gets a clips/ case: the collision rule is written for files
and would stash the directory as clips.dup-<src>, hiding every window in it
from the only reader. Two clip stores for one video are two halves of one
cache, so they merge.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
12 files changed, 891 insertions(+), 64 deletions(-)
diff --git a/common/controller/reconcileVideoDirs.ts b/common/controller/reconcileVideoDirs.ts
@@ -1,6 +1,7 @@
import path from "node:path";
-import { readdir, readFile, rename, rmdir, stat } from "node:fs/promises";
+import { readdir, readFile, rename, rm, rmdir, stat } from "node:fs/promises";
import { extractVideoId } from "../ytdlp/runYtdlp";
+import { CLIPS_DIR_NAME } from "../lib/clipWindow";
// The app keys every video by its canonical, URL-derived id
// (`extractVideoId(webpage_url)`) and expects its data under
@@ -80,6 +81,17 @@ async function mergeDir(
const dstSet = new Set(dstEntries);
for (const f of srcEntries) {
const src = path.join(srcDir, f);
+ // `clips/` IS A DIRECTORY, and the collision rule below is written for
+ // files: stashing it as `clips.dup-<srcName>` would hide every window it
+ // holds from listClipWindows, which only reads `clips/`. Two clip stores
+ // for one video are two halves of one cache, so they MERGE — and because a
+ // window's name IS its span, a file present in both is the same bytes.
+ if (f === CLIPS_DIR_NAME && dstSet.has(f)) {
+ const inner = await mergeDir(src, path.join(dstDir, f), srcName, dryRun);
+ moved.push(...inner.map((n) => path.join(f, n)));
+ if (!dryRun) await rm(src, { recursive: false, force: true }).catch(() => {});
+ continue;
+ }
if (!dstSet.has(f)) {
if (!dryRun) await rename(src, path.join(dstDir, f));
moved.push(f);
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -247,6 +247,19 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "custom",
needsMedia: true,
},
+ // ONE WINDOW of a video's source media, fetched into data/<id>/clips/ for a
+ // tool that asked for it by name (umtool's clip bench). Platform-queued like
+ // every other fetch so it takes its turn behind the channel's own downloads;
+ // replayable because the request is a few numbers and a reason, which is
+ // exactly what a JobSpec holds.
+ "fetch-window": {
+ kind: "fetch-window",
+ label: "Fetch window",
+ drainable: false,
+ replayable: true,
+ queueKeyStrategy: "platform",
+ needsMedia: true,
+ },
"redownload-archive": {
kind: "redownload-archive",
label: "Archive source video",
diff --git a/common/lib/clipWindow-server.ts b/common/lib/clipWindow-server.ts
@@ -0,0 +1,107 @@
+// The filesystem half of the clip-window store. See clipWindow.ts for the
+// layout and for why a `clips/` subdirectory is invisible to every video-dir
+// enumerator in the repo.
+
+import path from "node:path";
+import { mkdir, readFile, readdir, rename, stat, writeFile } from "node:fs/promises";
+import {
+ clipsDirFor,
+ clipWindowFile,
+ clipWindowSidecar,
+ parseClipProvenance,
+ parseClipWindowName,
+ tightestClipWindow,
+ type ClipProvenance,
+} from "./clipWindow";
+
+export type ClipWindow = {
+ // `<from>-<to>.mp4`, the file's own name.
+ file: string;
+ // Absolute path to it.
+ path: string;
+ from: number;
+ to: number;
+ bytes: number;
+ // The sidecar's contents, when there is one. A window fetched before the
+ // sidecar existed (or one whose sidecar was hand-removed) is still a window.
+ provenance: ClipProvenance | null;
+};
+
+// Every fetched window in a video dir, oldest-window-first by start second.
+//
+// A missing `clips/` is an empty list, not an error: most video dirs have none,
+// and the video page asks on every render.
+export async function listClipWindows(videoDir: string): Promise<ClipWindow[]> {
+ const dir = clipsDirFor(videoDir);
+ const entries = await readdir(dir).catch(() => [] as string[]);
+ const out: ClipWindow[] = [];
+ for (const name of entries) {
+ if (!name.endsWith(".mp4")) continue;
+ const span = parseClipWindowName(name);
+ if (!span) continue;
+ const abs = path.join(dir, name);
+ const st = await stat(abs).catch(() => null);
+ if (!st?.isFile()) continue;
+ const sidecar = await readFile(
+ path.join(dir, clipWindowSidecar(span.from, span.to)),
+ "utf8",
+ ).catch(() => null);
+ let provenance: ClipProvenance | null = null;
+ if (sidecar !== null) {
+ try {
+ provenance = parseClipProvenance(JSON.parse(sidecar));
+ } catch {
+ provenance = null;
+ }
+ }
+ out.push({
+ file: name,
+ path: abs,
+ from: span.from,
+ to: span.to,
+ bytes: st.size,
+ provenance,
+ });
+ }
+ out.sort((a, b) => a.from - b.from || a.to - b.to);
+ return out;
+}
+
+// The tightest already-fetched window covering [from, to], or null. This is
+// what makes a second request for the same seconds free — and what makes a
+// deliberately generous fetch BE the next caller's cache rather than a second
+// download of the same bytes.
+export async function findContainingClipWindow(
+ videoDir: string,
+ from: number,
+ to: number,
+): Promise<ClipWindow | null> {
+ return tightestClipWindow(await listClipWindows(videoDir), from, to);
+}
+
+// tmp + rename, the idiom the rest of the repo writes sidecars with, so a
+// crash mid-write never leaves a half-parsed provenance under the real name.
+export async function writeClipProvenance(
+ videoDir: string,
+ from: number,
+ to: number,
+ provenance: ClipProvenance,
+): Promise<string> {
+ const dir = clipsDirFor(videoDir);
+ await mkdir(dir, { recursive: true });
+ const file = path.join(dir, clipWindowSidecar(from, to));
+ const tmp = `${file}.tmp-${process.pid}`;
+ await writeFile(tmp, JSON.stringify(provenance, null, 2) + "\n");
+ await rename(tmp, file);
+ return file;
+}
+
+// Where a window for [from, to] would live. Exported so a caller can name the
+// destination before deciding to fetch it.
+export function clipWindowPath(
+ videoDir: string,
+ from: number,
+ to: number,
+): string {
+ return path.join(clipsDirFor(videoDir), clipWindowFile(from, to));
+}
diff --git a/common/lib/clipWindow.test.ts b/common/lib/clipWindow.test.ts
@@ -0,0 +1,136 @@
+import test from "node:test";
+import assert from "node:assert/strict";
+import {
+ clipWindowContains,
+ clipWindowFile,
+ clipWindowName,
+ clipWindowSidecar,
+ parseClipProvenance,
+ parseClipWindowName,
+ tightestClipWindow,
+ WIN_EPS,
+} from "./clipWindow";
+import { parseSavedVideoPointer } from "./savedVideo";
+
+test("a window name round-trips through two decimals", () => {
+ assert.equal(clipWindowName(12, 42), "12.00-42.00");
+ assert.equal(clipWindowFile(12, 42), "12.00-42.00.mp4");
+ assert.equal(clipWindowSidecar(12, 42), "12.00-42.00.json");
+ // The rounding is the name's, so a request carrying more precision than the
+ // name can hold still addresses ONE file.
+ assert.equal(clipWindowName(12.004, 41.999), "12.00-42.00");
+ assert.deepEqual(parseClipWindowName("12.00-42.00.mp4"), { from: 12, to: 42 });
+ assert.deepEqual(parseClipWindowName("12.00-42.00"), { from: 12, to: 42 });
+ assert.deepEqual(parseClipWindowName("0.00-1234.50.mp4"), {
+ from: 0,
+ to: 1234.5,
+ });
+});
+
+test("only <from>-<to>.mp4 parses as a window", () => {
+ for (const name of [
+ "12.00-42.00.json",
+ ".12.00-42.00.mp4.part",
+ "audio.mp3",
+ "source-media.mp4",
+ "12.00.mp4",
+ "-42.00.mp4",
+ "42.00-12.00.mp4", // to <= from is not a window
+ "12.00-12.00.mp4",
+ "abc-def.mp4",
+ ]) {
+ assert.equal(parseClipWindowName(name), null, name);
+ }
+});
+
+test("containment is to the naming tolerance, not exact", () => {
+ const w = { from: 10, to: 20 };
+ assert.ok(clipWindowContains(w, 10, 20));
+ assert.ok(clipWindowContains(w, 12, 18));
+ // A 2 dp manifest can ask for a hair outside the file that produced it.
+ assert.ok(clipWindowContains(w, 10 - WIN_EPS / 2, 20 + WIN_EPS / 2));
+ assert.ok(!clipWindowContains(w, 9, 20));
+ assert.ok(!clipWindowContains(w, 10, 21));
+});
+
+test("the tightest container wins, and overlap alone does not count", () => {
+ const windows = [
+ { from: 0, to: 100, name: "wide" },
+ { from: 9, to: 21, name: "tight" },
+ { from: 5, to: 50, name: "middle" },
+ { from: 15, to: 30, name: "overlap-only" },
+ ];
+ assert.equal(tightestClipWindow(windows, 10, 20)?.name, "tight");
+ assert.equal(tightestClipWindow(windows, 60, 70)?.name, "wide");
+ assert.equal(tightestClipWindow(windows, 99, 101), null);
+ assert.equal(tightestClipWindow([], 1, 2), null);
+});
+
+test("provenance parses tolerantly and refuses an anonymous sidecar", () => {
+ const full = parseClipProvenance({
+ requestedBy: "umtool",
+ manifest: "elfpire-eva",
+ clipId: "c03",
+ reason: "the clip ends mid-sentence",
+ requestedAt: "2026-09-20T00:00:00.000Z",
+ pad: 20,
+ bytes: 1234,
+ fetchedAt: "2026-09-20T00:00:05.000Z",
+ ytdlp: { args: ["--ignore-config", "-f", "best"] },
+ });
+ assert.equal(full?.requestedBy, "umtool");
+ assert.equal(full?.clipId, "c03");
+ assert.deepEqual(full?.ytdlp?.args, ["--ignore-config", "-f", "best"]);
+
+ // The requester is the one required field: a sidecar that cannot say who
+ // asked is not provenance.
+ assert.equal(parseClipProvenance({ manifest: "x" }), null);
+ assert.equal(parseClipProvenance(null), null);
+ assert.equal(parseClipProvenance("umtool"), null);
+ assert.equal(parseClipProvenance([]), null);
+
+ // Junk in the optional fields degrades to absent, never to a throw.
+ const partial = parseClipProvenance({
+ requestedBy: "umtool",
+ manifest: 7,
+ pad: "twenty",
+ ytdlp: { args: ["ok", 3, null] },
+ });
+ assert.equal(partial?.manifest, undefined);
+ assert.equal(partial?.pad, undefined);
+ assert.deepEqual(partial?.ytdlp?.args, ["ok"]);
+});
+
+test("a saved-video pointer written before `origin` parses unchanged", () => {
+ const legacy = parseSavedVideoPointer({
+ storedAt: "2026-01-01T00:00:00.000Z",
+ dir: "/store/chan/vid",
+ file: "source-media.mp4",
+ bytes: 42,
+ keepReason: "override",
+ });
+ assert.equal(legacy?.origin, undefined);
+ assert.equal(legacy?.keepReason, "override");
+
+ const sourced = parseSavedVideoPointer({
+ storedAt: "2026-01-01T00:00:00.000Z",
+ dir: "/store/chan/vid",
+ file: "source-media.mp4",
+ bytes: 42,
+ keepReason: "override",
+ origin: { requestedBy: "umtool", manifest: "m", clipId: "c01", reason: "" },
+ });
+ assert.equal(sourced?.origin?.requestedBy, "umtool");
+ assert.equal(sourced?.origin?.clipId, "c01");
+ // An empty reason is absent, not "".
+ assert.equal(sourced?.origin?.reason, undefined);
+ // A requester-less origin is dropped without taking the pointer with it.
+ const bad = parseSavedVideoPointer({
+ dir: "/d",
+ file: "f.mp4",
+ bytes: 1,
+ origin: { manifest: "m" },
+ });
+ assert.ok(bad);
+ assert.equal(bad?.origin, undefined);
+});
diff --git a/common/lib/clipWindow.ts b/common/lib/clipWindow.ts
@@ -0,0 +1,167 @@
+// A CLIP WINDOW: a few seconds of a video's source media, fetched on purpose
+// and kept beside the video it came from.
+//
+// umtool's clip bench used to run yt-dlp itself into its own project-local
+// `out/clips-raw`. Every tool that wanted the same seconds paid for them again,
+// and none of those fetches went through the editor's cookie policy, its
+// per-platform sleeps or its 429 cooldown. A window now lands in the corpus —
+//
+// channels/<slug>/data/<id>/clips/<from>-<to>.mp4
+// channels/<slug>/data/<id>/clips/<from>-<to>.json (who asked, and why)
+//
+// — where any tool can reuse it and the video page can show it.
+//
+// THE LOAD-BEARING RULE: a window fetch never writes metadata.info.json.
+// buildIndex.ts keys a video's presence in the LMDB index on that file's mtime
+// (controller/buildIndex.ts, the `metaMs` stat that `continue`s when it
+// throws), so writing one would put an undownloaded video into the index on the
+// strength of thirty seconds of audio. That is why the fetch assembles its own
+// argv rather than going through outputArgsForUrl / downloadOneManaged.
+//
+// Everything a `clips/` subdirectory has to survive is a consequence of two
+// facts, both re-verified when this landed:
+// * the video-dir predicates in lib/mediaFiles.ts are ANCHORED to one segment
+// after a fixed base name (`audio.<ext>`, `source-media.<ext>`), so a
+// directory named `clips` matches none of them, and readVideoFiles
+// (lib/videoStatus.ts) filters a plain readdir with exactly those;
+// * controller/scanCorruptMedia.ts filters the same listing by EXTENSION, and
+// `clips` has none.
+// The saved-video walks (pruneSavedVideos, savedVideoInventory) are
+// pointer-driven and iterate the VIDEO dirs under data/, one level above this.
+//
+// This module is pure (string + number math) so a client component can import
+// it; the filesystem half lives in clipWindow-server.ts.
+
+import path from "node:path";
+
+// The subdirectory inside a video dir that holds fetched windows.
+export const CLIPS_DIR_NAME = "clips";
+
+// A window read back from a 2 dp name can sit a hair outside the request that
+// produced it. The same tolerance umtool's build uses for the same reason
+// (report-to-video/build-video.mjs WIN_EPS) — the two have to agree or a file
+// one of them fetched is invisible to the other.
+export const WIN_EPS = 0.02;
+
+export function clipsDirFor(videoDir: string): string {
+ return path.join(videoDir, CLIPS_DIR_NAME);
+}
+
+// TWO DECIMALS, ALWAYS. The name IS the window: a reader parses it back rather
+// than opening a sidecar, so the formatting has to be a function of the numbers
+// and nothing else.
+export function clipWindowName(from: number, to: number): string {
+ return `${from.toFixed(2)}-${to.toFixed(2)}`;
+}
+
+export function clipWindowFile(from: number, to: number): string {
+ return `${clipWindowName(from, to)}.mp4`;
+}
+
+export function clipWindowSidecar(from: number, to: number): string {
+ return `${clipWindowName(from, to)}.json`;
+}
+
+const WINDOW_RE = /^(\d+(?:\.\d+)?)-(\d+(?:\.\d+)?)$/;
+
+// Parse `<from>-<to>.mp4` (or the bare `<from>-<to>`) back into its numbers, or
+// null for anything else in the directory — a sidecar, a `.part`, a stray.
+export function parseClipWindowName(
+ name: string,
+): { from: number; to: number } | null {
+ const stem = name.endsWith(".mp4") ? name.slice(0, -4) : name;
+ const m = WINDOW_RE.exec(stem);
+ if (!m) return null;
+ const from = Number(m[1]);
+ const to = Number(m[2]);
+ if (!Number.isFinite(from) || !Number.isFinite(to) || to <= from) return null;
+ return { from, to };
+}
+
+// Who asked for this window, and why. Written beside the file so the video page
+// can say "downloaded by umtool for <manifest>/<clipId>, because <reason>"
+// without a job log — a job log is pruned, and the window is not.
+export type ClipProvenance = {
+ // The tool that asked. "umtool" today; the field exists so a second one is
+ // distinguishable from it without a schema change.
+ requestedBy: string;
+ // The report manifest and the clip within it this window was fetched for.
+ manifest?: string;
+ clipId?: string;
+ // The operator-facing sentence: why this clip needs these seconds.
+ reason?: string;
+ requestedAt?: string;
+ // The pad, in seconds, the requester added around the clip's own window.
+ pad?: number;
+ bytes?: number;
+ fetchedAt?: string;
+ // The argv the fetch actually ran, minus the binary. For the case where a
+ // window looks wrong and the question is which format selector produced it.
+ ytdlp?: { args: string[] };
+};
+
+const str = (v: unknown): string | undefined =>
+ typeof v === "string" && v !== "" ? v : undefined;
+const num = (v: unknown): number | undefined =>
+ typeof v === "number" && Number.isFinite(v) ? v : undefined;
+
+// TOLERANT, like every other sidecar parse in this repo: a hand-edited or
+// half-written file degrades to "no provenance", never to a crashed page.
+export function parseClipProvenance(raw: unknown): ClipProvenance | null {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
+ const r = raw as Record<string, unknown>;
+ const requestedBy = str(r.requestedBy);
+ if (!requestedBy) return null;
+ const out: ClipProvenance = { requestedBy };
+ const manifest = str(r.manifest);
+ if (manifest) out.manifest = manifest;
+ const clipId = str(r.clipId);
+ if (clipId) out.clipId = clipId;
+ const reason = str(r.reason);
+ if (reason) out.reason = reason;
+ const requestedAt = str(r.requestedAt);
+ if (requestedAt) out.requestedAt = requestedAt;
+ const pad = num(r.pad);
+ if (pad !== undefined) out.pad = pad;
+ const bytes = num(r.bytes);
+ if (bytes !== undefined) out.bytes = bytes;
+ const fetchedAt = str(r.fetchedAt);
+ if (fetchedAt) out.fetchedAt = fetchedAt;
+ const y = r.ytdlp;
+ if (y && typeof y === "object" && Array.isArray((y as { args?: unknown }).args)) {
+ const args = (y as { args: unknown[] }).args.filter(
+ (a): a is string => typeof a === "string",
+ );
+ out.ytdlp = { args };
+ }
+ return out;
+}
+
+export type ClipWindowSpan = { from: number; to: number };
+
+// Does this window hold [from, to] whole, to the naming tolerance?
+export function clipWindowContains(
+ w: ClipWindowSpan,
+ from: number,
+ to: number,
+): boolean {
+ return !(w.from > from + WIN_EPS || w.to < to - WIN_EPS);
+}
+
+// The TIGHTEST window containing [from, to], or null.
+//
+// Tightest rather than widest for the same reason the build picks it: a
+// consumer decodes the whole file, so a 40-second container costs more than the
+// 14-second one that would also have done.
+export function tightestClipWindow<T extends ClipWindowSpan>(
+ windows: readonly T[],
+ from: number,
+ to: number,
+): T | null {
+ let best: T | null = null;
+ for (const w of windows) {
+ if (!clipWindowContains(w, from, to)) continue;
+ if (!best || w.to - w.from < best.to - best.from) best = w;
+ }
+ return best;
+}
diff --git a/common/lib/savedVideo-server.ts b/common/lib/savedVideo-server.ts
@@ -13,6 +13,7 @@ import {
parseSavedVideoPointer,
savedVideoPath,
type SavedVideoKeepReason,
+ type SavedVideoOrigin,
type SavedVideoPointer,
} from "./savedVideo";
@@ -105,6 +106,10 @@ export async function persistSourceVideo(opts: {
sourceFilename: string;
storeDir: string;
keepReason?: SavedVideoKeepReason;
+ // Who asked for this container, for a manually-requested persist. See
+ // SavedVideoOrigin — it records the requester WITHOUT changing keepReason,
+ // which is what keeps the container out of the retention prune.
+ origin?: SavedVideoOrigin;
}): Promise<SavedVideoPointer> {
const src = path.join(opts.videoDir, opts.sourceFilename);
const dest = path.join(opts.storeDir, opts.sourceFilename);
@@ -116,6 +121,7 @@ export async function persistSourceVideo(opts: {
file: opts.sourceFilename,
bytes: st.size,
...(opts.keepReason ? { keepReason: opts.keepReason } : {}),
+ ...(opts.origin ? { origin: opts.origin } : {}),
};
await writePointer(opts.videoDir, pointer);
return pointer;
diff --git a/common/lib/savedVideo.ts b/common/lib/savedVideo.ts
@@ -41,8 +41,40 @@ export type SavedVideoPointer = {
keepReason?: SavedVideoKeepReason;
// Optional content hash, populated by the backup/verify step (Phase 4).
sha256?: string;
+ // WHO asked for this container, when it was not the keep-latest rule.
+ //
+ // The clip-window feature lets umtool ask the editor for a whole source
+ // ("full: true"), and the operator then needs to know months later why a
+ // 4 GB file is on the platter. `keepReason` cannot say it: it stays
+ // "override"/"pin" precisely so pruneSavedVideos never evicts a container
+ // somebody asked for. Absent on every pointer written before this and on
+ // every automatic persist.
+ origin?: SavedVideoOrigin;
};
+// The requester of a manually-sourced container: the tool, and the report clip
+// it was wanted for. Deliberately the same vocabulary as ClipProvenance
+// (lib/clipWindow.ts) so the video page renders both the same way.
+export type SavedVideoOrigin = {
+ requestedBy: string;
+ manifest?: string;
+ clipId?: string;
+ reason?: string;
+ requestedAt?: string;
+};
+
+function parseOrigin(raw: unknown): SavedVideoOrigin | null {
+ if (!raw || typeof raw !== "object" || Array.isArray(raw)) return null;
+ const r = raw as Record<string, unknown>;
+ if (typeof r.requestedBy !== "string" || r.requestedBy === "") return null;
+ const out: SavedVideoOrigin = { requestedBy: r.requestedBy };
+ for (const k of ["manifest", "clipId", "reason", "requestedAt"] as const) {
+ const v = r[k];
+ if (typeof v === "string" && v !== "") out[k] = v;
+ }
+ return out;
+}
+
function isKeepReason(v: unknown): v is SavedVideoKeepReason {
return v === "keep-latest" || v === "pin" || v === "override";
}
@@ -88,5 +120,8 @@ export function parseSavedVideoPointer(raw: unknown): SavedVideoPointer | null {
if (typeof r.sha256 === "string" && r.sha256 !== "") {
pointer.sha256 = r.sha256;
}
+ // TOLERANT: a legacy pointer has no `origin` and must parse exactly as it did.
+ const origin = parseOrigin(r.origin);
+ if (origin) pointer.origin = origin;
return pointer;
}
diff --git a/common/ytdlp/channelArgs.ts b/common/ytdlp/channelArgs.ts
@@ -0,0 +1,21 @@
+// The per-channel argv every yt-dlp invocation in this repo appends: the cookie
+// source when the resolved policy calls for one, then the channel's own
+// `ytdlpExtraArgs` verbatim.
+//
+// Lifted out of ytdlp/downloadOneManaged.ts (where it was `channelConfigArgs`,
+// which still delegates here) because the clip-window fetch has to honour the
+// same two things and must not grow its own idea of them: an operator who set
+// `--limit-rate` on a channel meant it for every byte that channel costs, not
+// only for the bytes a full download costs.
+
+import type { ChannelConfig } from "../lib/channelConfig";
+
+export function channelExtraArgs(
+ config: ChannelConfig,
+ cookies?: string,
+): string[] {
+ const args: string[] = [];
+ if (cookies) args.push("--cookies-from-browser", cookies);
+ if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
+ return args;
+}
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -29,7 +29,7 @@ import {
import { isDoNotClean } from "../lib/doNotClean-server";
import { transcodeAudio } from "../controller/transcode";
import { findSourceMedia } from "../lib/videoStatus";
-import { savedVideoDir } from "../lib/savedVideo";
+import { savedVideoDir, type SavedVideoOrigin } from "../lib/savedVideo";
import { persistSourceVideo } from "../lib/savedVideo-server";
import {
type AudioCheckAttemptStats,
@@ -59,6 +59,12 @@ import {
import type { Paths } from "../lib/paths";
import { transcribeWithWorker } from "../controller/transcribeOne";
import { extractVideoId, outputArgsForUrl } from "./runYtdlp";
+import { channelExtraArgs } from "./channelArgs";
+import {
+ ARCHIVE_MARKER,
+ runOneYtdlp as runOneYtdlpRaw,
+ type AttemptOutcome,
+} from "./runOneYtdlp";
import { runAudioCheckedYtdlp } from "./audioCheckedDownload";
import {
resolveDownloadFormatSelector,
@@ -66,8 +72,6 @@ import {
} from "./downloadFormat";
import { DOWNLOAD_PROGRESS_TEMPLATE } from "../jobs/progressParsers";
-const STDERR_TAIL_BYTES = 64 * 1024;
-const ARCHIVE_MARKER = "DLOM_ARCHIVE";
// The `--print` archive marker below implies `--quiet`, which otherwise
// suppresses every extraction log and the [download] progress lines. Re-enable
@@ -75,7 +79,11 @@ const ARCHIVE_MARKER = "DLOM_ARCHIVE";
// directly into the per-video progress bars (human log readability is secondary
// to reliable progress parsing). --progress-delta keeps it to ~1 line/sec,
// matching the /jobs poll.
-const FULL_LOG_PROGRESS_ARGS = [
+// EXPORTED because the clip-window fetch (ytdlp/fetchWindowManaged.ts) reuses
+// it verbatim: the progress lines the /jobs bars are parsed from come from this
+// template, and a second fetch path with its own idea of it would show no
+// progress at all.
+export const FULL_LOG_PROGRESS_ARGS = [
"--no-quiet",
"--progress",
"--newline",
@@ -123,6 +131,11 @@ export type ManagedDownloadOpts = {
keepWindow?: KeepWindow;
// Per-run override: force-keep (true) / force-discard (false) the source video.
keepSourceVideoOverride?: boolean;
+ // Who asked for this container, recorded on the saved-video pointer. Set by
+ // the full-source variant of the clip-window fetch, so "why is this 4 GB
+ // file here" has an answer months later. Never changes `keepReason` — that
+ // stays override/pin so the retention prune leaves it alone.
+ persistOrigin?: SavedVideoOrigin;
// Per-run override: extract audio now and discard the container even for a
// video the keep-latest rule would otherwise persist (the save-disk backfill).
extractImmediately?: boolean;
@@ -218,6 +231,8 @@ async function finalizeAppExtraction(opts: {
// The persistence cause, recorded on the saved-video pointer (governs the
// retention prune). Only meaningful when persist is true.
category: PersistenceDecision["category"];
+ // Threaded straight onto the pointer; see ManagedDownloadOpts.persistOrigin.
+ origin?: SavedVideoOrigin;
onLog: (s: string) => void;
signal: AbortSignal;
}): Promise<void> {
@@ -262,6 +277,7 @@ async function finalizeAppExtraction(opts: {
sourceFilename: source,
storeDir,
keepReason: opts.category === "none" ? undefined : opts.category,
+ origin: opts.origin,
});
opts.onLog(
`Persisted source video to ${path.join(pointer.dir, pointer.file)} (${pointer.bytes} bytes).\n`,
@@ -295,71 +311,23 @@ function transcribeHandlingArgsForAudioCheck(
];
}
-function channelConfigArgs(
- config: ChannelConfig,
- cookies?: string,
-): string[] {
- const args: string[] = [];
- if (cookies) args.push("--cookies-from-browser", cookies);
- if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
- return args;
-}
+// Kept as a local name because every call site below reads `channelConfigArgs`;
+// the body lives in ytdlp/channelArgs.ts so the clip-window fetch shares it.
+const channelConfigArgs = channelExtraArgs;
-type AttemptOutcome = {
- exitCode: number | null;
- stderrTail: string;
- archiveLine: string | null;
-};
-
-async function runOneYtdlp(
+// The one-invocation runner moved to ytdlp/runOneYtdlp.ts (the clip-window
+// fetch needs the same log tee, stderr tail and archive scrape). This wrapper
+// keeps the ManagedDownloadOpts-shaped call sites below unchanged.
+function runOneYtdlp(
opts: ManagedDownloadOpts,
cwd: string,
args: string[],
): Promise<AttemptOutcome> {
- opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`);
- const child = execa(opts.paths.ytdlpBin, args, {
+ return runOneYtdlpRaw(
+ { ytdlpBin: opts.paths.ytdlpBin, onLog: opts.onLog, signal: opts.signal },
cwd,
- cancelSignal: opts.signal,
- all: false,
- buffer: false,
- reject: false,
- });
-
- let stderrTail = "";
- child.stderr?.on("data", (c: Buffer) => {
- const chunk = c.toString("utf8");
- opts.onLog(chunk);
- stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_BYTES);
- });
-
- let stdoutBuf = "";
- let archiveLine: string | null = null;
- child.stdout?.on("data", (c: Buffer) => {
- const chunk = c.toString("utf8");
- opts.onLog(chunk);
- stdoutBuf += chunk;
- // Pull whole lines out of the buffer; keep the trailing partial line.
- let nl: number;
- while ((nl = stdoutBuf.indexOf("\n")) !== -1) {
- const line = stdoutBuf.slice(0, nl).trim();
- stdoutBuf = stdoutBuf.slice(nl + 1);
- if (line.startsWith(`${ARCHIVE_MARKER} `)) {
- archiveLine = line.slice(ARCHIVE_MARKER.length + 1).trim();
- }
- }
- });
-
- const result = await child;
- // Flush any final partial line.
- const trailing = stdoutBuf.trim();
- if (trailing.startsWith(`${ARCHIVE_MARKER} `)) {
- archiveLine = trailing.slice(ARCHIVE_MARKER.length + 1).trim();
- }
- return {
- exitCode: result.exitCode ?? null,
- stderrTail,
- archiveLine,
- };
+ args,
+ );
}
function attemptSucceeded(exitCode: number | null): boolean {
@@ -974,6 +942,7 @@ async function runManagedDownload(
fmt,
persist: plan.persist,
category: plan.category,
+ origin: opts.persistOrigin,
onLog: opts.onLog,
signal: opts.signal,
});
@@ -1059,6 +1028,7 @@ async function runManagedDownload(
fmt,
persist: plan.persist,
category: plan.category,
+ origin: opts.persistOrigin,
onLog: opts.onLog,
signal: opts.signal,
});
diff --git a/common/ytdlp/fetchWindowManaged.ts b/common/ytdlp/fetchWindowManaged.ts
@@ -0,0 +1,274 @@
+// Fetch ONE window of a video's source media into the corpus, politely.
+//
+// This is the managed replacement for umtool running yt-dlp itself. The
+// operator's rule is that no fetch happens by hand: a window pulled from here
+// inherits the channel's cookie policy and its `ytdlpExtraArgs`, the
+// per-platform 429 cooldown, the auth retry, and the job log every other
+// download writes.
+//
+// WHAT IT DELIBERATELY DOES NOT DO: write metadata.info.json. See
+// lib/clipWindow.ts — the index keys a video's presence on that file, so a
+// thirty-second window must not create one. That is why the argv is assembled
+// HERE, in full, instead of going through outputArgsForUrl or
+// downloadOneManaged's prefetch.
+
+import path from "node:path";
+import { mkdir, readdir, rename, rm, stat } from "node:fs/promises";
+import {
+ AUTH_RETRY_CLASSES,
+ classifyDownloadFailure,
+ parseUnavailableFromStderr,
+} from "../lib/availability";
+import {
+ DEFAULT_COOKIE_MODE,
+ alwaysCookies,
+ authRetryCookies,
+ type ResolvedCookiePolicy,
+} from "../lib/cookiePolicy";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import {
+ clipsDirFor,
+ clipWindowFile,
+ type ClipProvenance,
+} from "../lib/clipWindow";
+import {
+ clipWindowPath,
+ findContainingClipWindow,
+ writeClipProvenance,
+ type ClipWindow,
+} from "../lib/clipWindow-server";
+import { channelExtraArgs } from "./channelArgs";
+import { runOneYtdlp } from "./runOneYtdlp";
+import { FULL_LOG_PROGRESS_ARGS } from "./downloadOneManaged";
+
+// The source height a clip is worth fetching at. 720 is umtool's number and the
+// reason is the render: the deliverable is 1080p with a clip inset, so pixels
+// above 720 are thrown away after paying for them.
+export const DEFAULT_CLIP_MAX_HEIGHT = 720;
+
+// THE FORMAT SELECTOR, and why it is not downloadFormatPreset's.
+//
+// resolveDownloadFormatSelector answers "what should we ARCHIVE" — best
+// available, container-agnostic. This answers "what can ffmpeg cut and
+// re-encode cheaply, right now". Those differ on one specific trap: left alone
+// yt-dlp picks VP9+Opus at these heights, and since --force-keyframes-at-cuts
+// re-encodes, that means libvpx-vp9 — 27 s to cut a 5 s clip, measured — and it
+// writes .webm, which it then appends to the -o name. So H.264/AAC in mp4 is
+// pinned, with three progressively looser fallbacks. Byte-identical to the
+// selector umtool's build-video.mjs uses, on purpose: a window this fetches and
+// a window that fetched must be the same file, or the two caches diverge.
+export function clipFormatSelector(maxHeight: number): string {
+ return [
+ `bv*[vcodec^=avc1][height<=${maxHeight}]+ba[acodec^=mp4a]`,
+ `bv*[ext=mp4][height<=${maxHeight}]+ba[ext=m4a]`,
+ `b[ext=mp4][height<=${maxHeight}]`,
+ `b[height<=${maxHeight}]`,
+ ].join("/");
+}
+
+export type FetchWindowProvenance = {
+ requestedBy: string;
+ manifest?: string;
+ clipId?: string;
+ reason?: string;
+ pad?: number;
+ requestedAt?: string;
+};
+
+export type FetchWindowOpts = {
+ channelSlug: string;
+ channelConfig: ChannelConfig;
+ paths: Paths;
+ // channels/<slug>/data/<id> — the window lands in its `clips/` subdirectory.
+ videoDir: string;
+ videoId: string;
+ videoUrl: string;
+ from: number;
+ to: number;
+ provenance: FetchWindowProvenance;
+ cookiePolicy?: ResolvedCookiePolicy;
+ maxHeight?: number;
+ onLog: (s: string) => void;
+ signal: AbortSignal;
+ // Called when the failure classifies as a platform-level signal (a 429 or a
+ // bot check). The caller owns the platform key, so it owns the cooldown.
+ onPlatformBackoff?: (
+ failureClass: "rate_limit",
+ ) => Promise<void> | void;
+};
+
+export type FetchWindowResult = {
+ // Absolute path to the mp4 holding the window.
+ file: string;
+ // The window the returned FILE holds, which on a cache hit is WIDER than the
+ // one asked for. Every consumer expresses cuts relative to it, so returning
+ // the request instead would seek a caller into the wrong seconds.
+ from: number;
+ to: number;
+ bytes: number;
+ cached: boolean;
+ provenance: ClipProvenance | null;
+};
+
+function cachedResult(hit: ClipWindow): FetchWindowResult {
+ return {
+ file: hit.path,
+ from: hit.from,
+ to: hit.to,
+ bytes: hit.bytes,
+ cached: true,
+ provenance: hit.provenance,
+ };
+}
+
+export async function fetchWindowManaged(
+ opts: FetchWindowOpts,
+): Promise<FetchWindowResult> {
+ const { from, to, videoDir } = opts;
+ const policy: ResolvedCookiePolicy = opts.cookiePolicy ?? {
+ cookies: opts.channelConfig.cookiesFromBrowser,
+ mode: DEFAULT_COOKIE_MODE,
+ };
+
+ // ASK THE CACHE FIRST, and accept a WIDER file. A report cites the same
+ // stream more than once; without containing-window reuse a generous fetch for
+ // one clip is worthless to the neighbour it already covers.
+ const hit = await findContainingClipWindow(videoDir, from, to);
+ if (hit) {
+ opts.onLog(
+ `Window ${from.toFixed(2)}–${to.toFixed(2)} is already covered by ` +
+ `clips/${hit.file}; nothing to fetch.\n`,
+ );
+ return cachedResult(hit);
+ }
+
+ const clipsDir = clipsDirFor(videoDir);
+ await mkdir(clipsDir, { recursive: true });
+ const dest = clipWindowPath(videoDir, from, to);
+ // A dotfile, so a half-written window can never be parsed as one: listing
+ // only admits `<from>-<to>.mp4`.
+ const part = path.join(clipsDir, `.${clipWindowFile(from, to)}.part`);
+ await rm(part, { force: true });
+
+ const maxHeight = opts.maxHeight ?? DEFAULT_CLIP_MAX_HEIGHT;
+ const argsWith = (cookies: string | undefined): string[] => [
+ // The operator's own yt-dlp config redirects output and attaches thumbnail
+ // and metadata post-processors; without this the window lands elsewhere —
+ // and a metadata post-processor is exactly what must not run here.
+ "--ignore-config",
+ "--no-playlist",
+ // One request per second, the politeness this repo applies everywhere it
+ // touches a source.
+ "--sleep-requests",
+ "1",
+ "--download-sections",
+ `*${from.toFixed(2)}-${to.toFixed(2)}`,
+ // Without this the cut snaps to the nearest preceding keyframe, which can
+ // be seconds early — fine for scrubbing, not fine when the clip IS the
+ // citation.
+ "--force-keyframes-at-cuts",
+ "-f",
+ clipFormatSelector(maxHeight),
+ "--merge-output-format",
+ "mp4",
+ "-o",
+ part,
+ ...FULL_LOG_PROGRESS_ARGS,
+ ...channelExtraArgs(opts.channelConfig, cookies),
+ "--",
+ opts.videoUrl,
+ ];
+
+ const run = async (cookies: string | undefined) =>
+ runOneYtdlp(
+ {
+ ytdlpBin: opts.paths.ytdlpBin,
+ onLog: opts.onLog,
+ signal: opts.signal,
+ },
+ videoDir,
+ argsWith(cookies),
+ );
+
+ let args = argsWith(alwaysCookies(policy));
+ let outcome = await run(alwaysCookies(policy));
+
+ if (outcome.exitCode !== 0) {
+ const availability = parseUnavailableFromStderr(outcome.stderrTail);
+ const failure = classifyDownloadFailure(outcome.stderrTail, availability);
+ if (failure === "rate_limit") {
+ // The 429 / bot-check cooldown the auto-download runner and a clicked
+ // Sync both honour. Recorded before the throw so the NEXT request is
+ // refused at the door rather than re-storming the source.
+ await opts.onPlatformBackoff?.("rate_limit");
+ } else if (AUTH_RETRY_CLASSES.has(availability)) {
+ const retryCookies = authRetryCookies(policy);
+ if (retryCookies) {
+ opts.onLog(
+ `Window fetch failed with ${availability}; retrying once with cookies.\n`,
+ );
+ args = argsWith(retryCookies);
+ outcome = await run(retryCookies);
+ }
+ }
+ }
+
+ if (outcome.exitCode !== 0) {
+ await rm(part, { force: true });
+ const tail = outcome.stderrTail.trim().split("\n").slice(-4).join(" / ");
+ throw new Error(
+ `yt-dlp failed fetching ${from.toFixed(2)}–${to.toFixed(2)} of ` +
+ `${opts.videoId} (exit ${outcome.exitCode ?? "null"}): ${tail}`,
+ );
+ }
+
+ // A fallback branch of the selector can still force another container, in
+ // which case yt-dlp writes "<part>.<realext>". Adopt it rather than failing a
+ // download that actually happened.
+ let produced = part;
+ let st = await stat(part).catch(() => null);
+ if (!st?.isFile()) {
+ const base = path.basename(part);
+ const stray = (await readdir(clipsDir).catch(() => [] as string[])).find(
+ (f) => f.startsWith(`${base}.`),
+ );
+ if (!stray) {
+ throw new Error(
+ `yt-dlp reported success but produced no file for ` +
+ `${from.toFixed(2)}–${to.toFixed(2)} of ${opts.videoId}`,
+ );
+ }
+ produced = path.join(clipsDir, stray);
+ st = await stat(produced);
+ }
+ await rename(produced, dest);
+
+ const provenance: ClipProvenance = {
+ requestedBy: opts.provenance.requestedBy,
+ ...(opts.provenance.manifest ? { manifest: opts.provenance.manifest } : {}),
+ ...(opts.provenance.clipId ? { clipId: opts.provenance.clipId } : {}),
+ ...(opts.provenance.reason ? { reason: opts.provenance.reason } : {}),
+ requestedAt: opts.provenance.requestedAt ?? new Date().toISOString(),
+ ...(typeof opts.provenance.pad === "number"
+ ? { pad: opts.provenance.pad }
+ : {}),
+ bytes: st.size,
+ fetchedAt: new Date().toISOString(),
+ ytdlp: { args },
+ };
+ await writeClipProvenance(videoDir, from, to, provenance);
+ opts.onLog(
+ `Fetched ${from.toFixed(2)}–${to.toFixed(2)} of ${opts.videoId} into ` +
+ `clips/${clipWindowFile(from, to)} (${st.size} bytes).\n`,
+ );
+
+ return {
+ file: dest,
+ from,
+ to,
+ bytes: st.size,
+ cached: false,
+ provenance,
+ };
+}
diff --git a/common/ytdlp/runOneYtdlp.ts b/common/ytdlp/runOneYtdlp.ts
@@ -0,0 +1,84 @@
+// ONE yt-dlp invocation, with this repo's logging contract around it.
+//
+// Lifted verbatim out of ytdlp/downloadOneManaged.ts, which is still its
+// biggest caller and now delegates to it. It moved because the clip-window
+// fetch (ytdlp/fetchWindowManaged.ts) needs exactly this — the log tee, the
+// bounded stderr tail a failure is classified from, and the archive-marker
+// scrape — and the alternative was a second copy that would drift.
+//
+// Its dependencies are three values, not a ManagedDownloadOpts: the binary, a
+// log sink and an abort signal. That is the whole reason it lifts cleanly.
+
+import { execa } from "execa";
+
+export const STDERR_TAIL_BYTES = 64 * 1024;
+
+// yt-dlp is asked to `--print` this marker after each video so the caller can
+// learn the archive line without re-reading the archive file.
+export const ARCHIVE_MARKER = "DLOM_ARCHIVE";
+
+export type YtdlpRunOpts = {
+ ytdlpBin: string;
+ onLog: (s: string) => void;
+ signal: AbortSignal;
+};
+
+export type AttemptOutcome = {
+ exitCode: number | null;
+ // The last STDERR_TAIL_BYTES of stderr. Bounded because a failing download
+ // can produce megabytes of it and the only consumer is a set of regexes
+ // (lib/availability.ts classifyDownloadFailure).
+ stderrTail: string;
+ archiveLine: string | null;
+};
+
+export async function runOneYtdlp(
+ opts: YtdlpRunOpts,
+ cwd: string,
+ args: string[],
+): Promise<AttemptOutcome> {
+ opts.onLog(`$ ${opts.ytdlpBin} ${args.join(" ")}\n`);
+ const child = execa(opts.ytdlpBin, args, {
+ cwd,
+ cancelSignal: opts.signal,
+ all: false,
+ buffer: false,
+ reject: false,
+ });
+
+ let stderrTail = "";
+ child.stderr?.on("data", (c: Buffer) => {
+ const chunk = c.toString("utf8");
+ opts.onLog(chunk);
+ stderrTail = (stderrTail + chunk).slice(-STDERR_TAIL_BYTES);
+ });
+
+ let stdoutBuf = "";
+ let archiveLine: string | null = null;
+ child.stdout?.on("data", (c: Buffer) => {
+ const chunk = c.toString("utf8");
+ opts.onLog(chunk);
+ stdoutBuf += chunk;
+ // Pull whole lines out of the buffer; keep the trailing partial line.
+ let nl: number;
+ while ((nl = stdoutBuf.indexOf("\n")) !== -1) {
+ const line = stdoutBuf.slice(0, nl).trim();
+ stdoutBuf = stdoutBuf.slice(nl + 1);
+ if (line.startsWith(`${ARCHIVE_MARKER} `)) {
+ archiveLine = line.slice(ARCHIVE_MARKER.length + 1).trim();
+ }
+ }
+ });
+
+ const result = await child;
+ // Flush any final partial line.
+ const trailing = stdoutBuf.trim();
+ if (trailing.startsWith(`${ARCHIVE_MARKER} `)) {
+ archiveLine = trailing.slice(ARCHIVE_MARKER.length + 1).trim();
+ }
+ return {
+ exitCode: result.exitCode ?? null,
+ stderrTail,
+ archiveLine,
+ };
+}
diff --git a/export/public/public b/export/public/public
@@ -0,0 +1 @@
+/home/user/Projects/yt-dlp-transcript-browser/export/public
+\ No newline at end of file