commit b702f016c65853a65ad5a21555b9a181dcd283db
parent a7a407ac6855df87e68dfad6ace0d1b5b1be680e
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sun, 4 Oct 2026 16:34:59 -0400
Merge persist-videos-batch (pnpm ops persist-videos: persist an explicit list of videos at a chosen quality, one paced, disk- and memory-gated job per channel, dry-run buckets, above-height replace)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
# Conflicts:
# editor/CHANGELOG.md
Diffstat:
16 files changed, 1534 insertions(+), 12 deletions(-)
diff --git a/RUNNING_IN_DOCKER.md b/RUNNING_IN_DOCKER.md
@@ -256,6 +256,7 @@ pnpm ops refresh-report --json '{"all":true}'
pnpm ops keep-videos --json '{"slug":"paramount-tactical","match":"TheQuartering","dryRun":true}'
pnpm ops fetch-posts --json '{"slug":"example-x","older":true}' --wait
pnpm ops capture-posts --json '{"slug":"example-x","ids":["1234567890"]}' --wait
+pnpm ops persist-videos --json '{"items":[{"slug":"example-channel","id":"abc123"}],"dryRun":true}'
pnpm ops get channel the-quartering
pnpm ops list # every action name
```
diff --git a/common/controller/persistVideos.test.ts b/common/controller/persistVideos.test.ts
@@ -0,0 +1,511 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readFile, rm, stat, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { SiteSettings } from "../lib/settings";
+import type { DiskGateStatus } from "../lib/diskSpace";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import { loadSavedVideo, persistSourceVideo } from "../lib/savedVideo-server";
+import {
+ persistVideos,
+ type PersistVideoDownload,
+ type PersistVideosDeps,
+} from "./persistVideos";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test controller/persistVideos.test.ts
+//
+// Every dependency that would reach the network, the clock or the machine's
+// memory is injected: the download writes a fake container and persists it the
+// way downloadOneManaged does, sleeps are recorded, MemAvailable is scripted.
+
+const CONFIG: ChannelConfig = {
+ handling: "transcribe",
+ platform: "youtube",
+ url: "https://www.youtube.com/@demo",
+};
+
+const SETTINGS = {
+ sourceVideoQuality: "original",
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ inlineTranscribeOnFallback: false,
+ skipLiveDownloads: false,
+} as unknown as SiteSettings;
+
+const OPEN_GATE: DiskGateStatus = {
+ ok: true,
+ enabled: false,
+ freeBytes: Number.POSITIVE_INFINITY,
+ thresholdBytes: 0,
+ resumeBytes: 0,
+ reason: "ok",
+ message: "",
+};
+
+const CLOSED_GATE: DiskGateStatus = {
+ ...OPEN_GATE,
+ ok: false,
+ enabled: true,
+ freeBytes: 1,
+ thresholdBytes: 10,
+ reason: "below-floor",
+ message: "Low disk space",
+} as DiskGateStatus;
+
+async function withPaths(fn: (paths: Paths) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-persistvideos-"));
+ const paths = {
+ transcriptsDir: dir,
+ channelsDir: path.join(dir, "channels"),
+ savedVideosDir: path.join(dir, "saved"),
+ } as Paths;
+ try {
+ await fn(paths);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+function videoDir(paths: Paths, slug: string, id: string): string {
+ return path.join(paths.channelsDir, slug, "data", id);
+}
+
+// A downloaded video: its dir and the metadata that names its source URL.
+async function seedVideo(paths: Paths, slug: string, id: string): Promise<void> {
+ const dir = videoDir(paths, slug, id);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ path.join(dir, "metadata.info.json"),
+ JSON.stringify({ id, webpage_url: `https://www.youtube.com/watch?v=${id}` }),
+ );
+}
+
+async function seedSaved(
+ paths: Paths,
+ slug: string,
+ id: string,
+ opts: {
+ height?: number;
+ file?: string;
+ keepReason?: "keep-latest" | "pin";
+ requestedBy?: string;
+ } = {},
+): Promise<void> {
+ await seedVideo(paths, slug, id);
+ const dir = videoDir(paths, slug, id);
+ const file = opts.file ?? "source-media.mp4";
+ await writeFile(path.join(dir, file), `old ${id}`);
+ await persistSourceVideo({
+ videoDir: dir,
+ sourceFilename: file,
+ storeDir: path.join(paths.savedVideosDir, slug, id),
+ keepReason: opts.keepReason ?? "keep-latest",
+ ...(opts.requestedBy ? { origin: { requestedBy: opts.requestedBy } } : {}),
+ format: {
+ preset: "original",
+ ...(opts.height ? { height: opts.height } : {}),
+ },
+ });
+}
+
+type Harness = {
+ deps: Partial<PersistVideosDeps>;
+ downloads: { slug: string; id: string; quality: string }[];
+ sleeps: number[];
+ events: string[];
+};
+
+// The default harness: every channel exists, the gate is open, memory is
+// plentiful, and a download persists a 720p container named `file`.
+function harness(
+ paths: Paths,
+ over: Partial<PersistVideosDeps> & { file?: string; height?: number } = {},
+): Harness {
+ const downloads: Harness["downloads"] = [];
+ const sleeps: number[] = [];
+ const events: string[] = [];
+ const download: PersistVideoDownload = async (o) => {
+ const id = o.videoUrl.split("v=")[1];
+ downloads.push({ slug: o.channelSlug, id, quality: o.quality });
+ events.push(`download ${id}`);
+ const dir = videoDir(paths, o.channelSlug, id);
+ const file = over.file ?? "source-media.mp4";
+ await writeFile(path.join(dir, file), `new ${id}`);
+ await persistSourceVideo({
+ videoDir: dir,
+ sourceFilename: file,
+ storeDir: path.join(paths.savedVideosDir, o.channelSlug, id),
+ keepReason: "override",
+ format: { preset: o.quality, height: over.height ?? 720 },
+ });
+ return {
+ videoId: id,
+ status: "ok",
+ startedAt: "",
+ finishedAt: "",
+ attempts: [],
+ } satisfies DownloadOutcomeRecord;
+ };
+ return {
+ downloads,
+ sleeps,
+ events,
+ deps: {
+ download,
+ sleep: async (ms) => {
+ sleeps.push(ms);
+ events.push(`sleep ${ms}`);
+ },
+ readMemAvailableMb: async () => 99_999,
+ diskGate: async () => OPEN_GATE,
+ getSettings: () => SETTINGS,
+ readChannelConfig: async (_p, slug) => (slug === "nope" ? null : CONFIG),
+ onPlatformBackoff: async () => {},
+ ...over,
+ },
+ };
+}
+
+test("already-saved items are skipped; the rest are fetched in order", async () => {
+ await withPaths(async (paths) => {
+ await seedSaved(paths, "demo-channel", "aaa111");
+ await seedVideo(paths, "demo-channel", "bbb222");
+ await seedVideo(paths, "other-channel", "ccc333");
+ const h = harness(paths);
+ const res = await persistVideos({
+ paths,
+ items: [
+ { slug: "demo-channel", id: "aaa111" },
+ { slug: "demo-channel", id: "bbb222" },
+ { slug: "other-channel", id: "ccc333" },
+ ],
+ deps: h.deps,
+ });
+ assert.deepEqual(
+ h.downloads.map((d) => d.id),
+ ["bbb222", "ccc333"],
+ );
+ assert.equal(res.plan.saved.count, 1);
+ assert.equal(res.persisted.length, 2);
+ assert.equal(res.failed.length, 0);
+ assert.equal(res.stopped, null);
+ // Re-running the same list is the resume: everything is saved now.
+ const again = await persistVideos({
+ paths,
+ items: [
+ { slug: "demo-channel", id: "aaa111" },
+ { slug: "demo-channel", id: "bbb222" },
+ { slug: "other-channel", id: "ccc333" },
+ ],
+ deps: harness(paths).deps,
+ });
+ assert.equal(again.plan.saved.count, 3);
+ assert.equal(again.plan.willFetch, 0);
+ });
+});
+
+test("format defaults to each channel's quality, and an explicit one wins", async () => {
+ await withPaths(async (paths) => {
+ await seedVideo(paths, "demo-channel", "aaa111");
+ await seedVideo(paths, "hd-channel", "bbb222");
+ const configs: Record<string, ChannelConfig> = {
+ "demo-channel": CONFIG,
+ "hd-channel": { ...CONFIG, sourceVideoQuality: "video_720" },
+ };
+ const h = harness(paths, { readChannelConfig: async (_p, s) => configs[s] ?? null });
+ await persistVideos({
+ paths,
+ items: [
+ { slug: "demo-channel", id: "aaa111" },
+ { slug: "hd-channel", id: "bbb222" },
+ ],
+ dryRun: false,
+ deps: h.deps,
+ });
+ assert.deepEqual(
+ h.downloads.map((d) => d.quality),
+ ["original", "video_720"],
+ );
+
+ await seedVideo(paths, "demo-channel", "ccc333");
+ const h2 = harness(paths, { readChannelConfig: async (_p, s) => configs[s] ?? null });
+ await persistVideos({
+ paths,
+ items: [{ slug: "demo-channel", id: "ccc333" }],
+ format: "video_720",
+ deps: h2.deps,
+ });
+ assert.deepEqual(h2.downloads.map((d) => d.quality), ["video_720"]);
+ });
+});
+
+test("above-height replaces a too-tall or unmeasured container, keeping its provenance", async () => {
+ await withPaths(async (paths) => {
+ await seedSaved(paths, "demo-channel", "tall111", { height: 1080, file: "source-media.webm" });
+ await seedSaved(paths, "demo-channel", "nohgt22", { keepReason: "pin", requestedBy: "umtool" });
+ await seedSaved(paths, "demo-channel", "ok33333", { height: 720 });
+ const items = [
+ { slug: "demo-channel", id: "tall111" },
+ { slug: "demo-channel", id: "nohgt22" },
+ { slug: "demo-channel", id: "ok33333" },
+ ];
+
+ // "never" (the default) leaves every saved item alone.
+ const never = harness(paths);
+ const left = await persistVideos({ paths, items, format: "video_720", deps: never.deps });
+ assert.equal(never.downloads.length, 0);
+ assert.equal(left.plan.wrongHeight.count, 2);
+ assert.equal(left.plan.saved.count, 1);
+ assert.equal(left.plan.willFetch, 0);
+
+ const oldTall = await loadSavedVideo(videoDir(paths, "demo-channel", "tall111"));
+ const h = harness(paths);
+ const res = await persistVideos({
+ paths,
+ items,
+ format: "video_720",
+ replace: "above-height",
+ deps: h.deps,
+ });
+ assert.deepEqual(
+ h.downloads.map((d) => d.id),
+ ["tall111", "nohgt22"],
+ );
+ assert.equal(res.replaced.length, 2);
+ const tall = await loadSavedVideo(videoDir(paths, "demo-channel", "tall111"));
+ assert.equal(tall?.file, "source-media.mp4");
+ assert.equal(tall?.format?.height, 720);
+ // The replacement is still the keep-latest container it replaced.
+ assert.equal(tall?.keepReason, "keep-latest");
+ const nohgt = await loadSavedVideo(videoDir(paths, "demo-channel", "nohgt22"));
+ assert.equal(nohgt?.keepReason, "pin");
+ assert.deepEqual(nohgt?.origin, { requestedBy: "umtool" });
+ assert.equal(nohgt?.format?.height, 720);
+ // The old, differently-named container went only after the new one landed.
+ await assert.rejects(stat(path.join(oldTall!.dir, oldTall!.file)));
+ assert.equal(
+ await readFile(path.join(tall!.dir, tall!.file), "utf8"),
+ "new tall111",
+ );
+ });
+});
+
+test("a replacement that does not land leaves the old container alone", async () => {
+ await withPaths(async (paths) => {
+ await seedSaved(paths, "demo-channel", "tall111", { height: 1080, file: "source-media.webm" });
+ const h = harness(paths, {
+ // A download that "succeeds" without moving anything into the store.
+ download: async () => ({
+ videoId: "tall111",
+ status: "ok",
+ startedAt: "",
+ finishedAt: "",
+ attempts: [],
+ }),
+ });
+ const res = await persistVideos({
+ paths,
+ items: [{ slug: "demo-channel", id: "tall111" }],
+ format: "video_720",
+ replace: "above-height",
+ deps: h.deps,
+ });
+ assert.equal(res.failed.length, 1);
+ assert.match(res.failed[0].error, /not saved to the store/);
+ const ptr = await loadSavedVideo(videoDir(paths, "demo-channel", "tall111"));
+ assert.equal(await readFile(path.join(ptr!.dir, ptr!.file), "utf8"), "old tall111");
+ });
+});
+
+test("a closed disk gate stops the run; the rest are not attempted", async () => {
+ await withPaths(async (paths) => {
+ for (const id of ["aaa111", "bbb222", "ccc333"]) {
+ await seedVideo(paths, "demo-channel", id);
+ }
+ let calls = 0;
+ const h = harness(paths, {
+ diskGate: async () => (++calls >= 2 ? CLOSED_GATE : OPEN_GATE),
+ });
+ const res = await persistVideos({
+ paths,
+ items: ["aaa111", "bbb222", "ccc333"].map((id) => ({ slug: "demo-channel", id })),
+ deps: h.deps,
+ });
+ assert.deepEqual(h.downloads.map((d) => d.id), ["aaa111"]);
+ assert.equal(res.stopped, "low-disk");
+ assert.deepEqual(
+ res.notAttempted.map((i) => i.id),
+ ["bbb222", "ccc333"],
+ );
+ });
+});
+
+test("a rate limit stops the run and records the platform's backoff", async () => {
+ await withPaths(async (paths) => {
+ for (const id of ["aaa111", "bbb222"]) await seedVideo(paths, "demo-channel", id);
+ const backoffs: string[] = [];
+ const h = harness(paths, {
+ download: async () => ({
+ videoId: "aaa111",
+ status: "failed",
+ startedAt: "",
+ finishedAt: "",
+ attempts: [],
+ failureClass: "rate_limit",
+ }),
+ onPlatformBackoff: async (platform, _p, cls) => {
+ backoffs.push(`${platform}:${cls}`);
+ },
+ });
+ const res = await persistVideos({
+ paths,
+ items: ["aaa111", "bbb222"].map((id) => ({ slug: "demo-channel", id })),
+ deps: h.deps,
+ });
+ assert.equal(res.stopped, "rate-limit");
+ assert.equal(res.failed.length, 1);
+ assert.deepEqual(res.notAttempted.map((i) => i.id), ["bbb222"]);
+ assert.deepEqual(backoffs, ["youtube:rate_limit"]);
+ });
+});
+
+test("the gap is paid between downloads, not before the first", async () => {
+ await withPaths(async (paths) => {
+ for (const id of ["aaa111", "bbb222", "ccc333"]) await seedVideo(paths, "demo-channel", id);
+ await seedSaved(paths, "demo-channel", "saved11");
+ const h = harness(paths);
+ await persistVideos({
+ paths,
+ items: ["aaa111", "saved11", "bbb222", "ccc333"].map((id) => ({
+ slug: "demo-channel",
+ id,
+ })),
+ gapMs: 5000,
+ deps: h.deps,
+ });
+ assert.deepEqual(h.events, [
+ "download aaa111",
+ "sleep 5000",
+ "download bbb222",
+ "sleep 5000",
+ "download ccc333",
+ ]);
+ });
+});
+
+test("the default gap is the batch downloads' own (sleepBetweenDownloadsSeconds)", async () => {
+ await withPaths(async (paths) => {
+ for (const id of ["aaa111", "bbb222"]) await seedVideo(paths, "demo-channel", id);
+ const h = harness(paths, {
+ readChannelConfig: async () => ({ ...CONFIG, sleepBetweenDownloadsSeconds: 7 }),
+ });
+ await persistVideos({
+ paths,
+ items: ["aaa111", "bbb222"].map((id) => ({ slug: "demo-channel", id })),
+ deps: h.deps,
+ });
+ assert.deepEqual(h.sleeps, [7000]);
+ });
+});
+
+test("the memory wait holds each download until MemAvailable reaches the floor", async () => {
+ await withPaths(async (paths) => {
+ await seedVideo(paths, "demo-channel", "aaa111");
+ const readings = [1000, 2000, 5000];
+ const h = harness(paths, {
+ readMemAvailableMb: async () => readings.shift() ?? 5000,
+ });
+ const res = await persistVideos({
+ paths,
+ items: [{ slug: "demo-channel", id: "aaa111" }],
+ minFreeMemMb: 4096,
+ deps: h.deps,
+ });
+ assert.deepEqual(h.events, ["sleep 10000", "sleep 10000", "download aaa111"]);
+ assert.equal(res.persisted.length, 1);
+ });
+});
+
+test("cancel and drain stop between items", async () => {
+ await withPaths(async (paths) => {
+ for (const id of ["aaa111", "bbb222", "ccc333"]) await seedVideo(paths, "demo-channel", id);
+ const items = ["aaa111", "bbb222", "ccc333"].map((id) => ({ slug: "demo-channel", id }));
+
+ const abort = new AbortController();
+ const h = harness(paths);
+ const inner = h.deps.download!;
+ h.deps.download = async (o) => {
+ const r = await inner(o);
+ abort.abort();
+ return r;
+ };
+ const cancelled = await persistVideos({ paths, items, signal: abort.signal, deps: h.deps });
+ assert.equal(cancelled.stopped, "cancelled");
+ assert.equal(cancelled.persisted.length, 1);
+ assert.deepEqual(cancelled.notAttempted.map((i) => i.id), ["bbb222", "ccc333"]);
+
+ const drain = new AbortController();
+ const h2 = harness(paths);
+ const inner2 = h2.deps.download!;
+ h2.deps.download = async (o) => {
+ const r = await inner2(o);
+ drain.abort();
+ return r;
+ };
+ const drained = await persistVideos({ paths, items, drainSignal: drain.signal, deps: h2.deps });
+ assert.equal(drained.stopped, "drained");
+ assert.deepEqual(h2.downloads.map((d) => d.id), ["bbb222"]);
+ assert.deepEqual(drained.notAttempted.map((i) => i.id), ["ccc333"]);
+ });
+});
+
+test("a dry run buckets every item and fetches nothing", async () => {
+ await withPaths(async (paths) => {
+ await seedSaved(paths, "demo-channel", "saved11", { height: 720 });
+ await seedSaved(paths, "demo-channel", "tall111", { height: 1080 });
+ await seedVideo(paths, "demo-channel", "fetch11");
+ // A video dir with no metadata, on a channel with no platform: no URL.
+ await mkdir(videoDir(paths, "bare-channel", "nourl11"), { recursive: true });
+ const h = harness(paths, {
+ readChannelConfig: async (_p, slug) =>
+ slug === "nope"
+ ? null
+ : slug === "bare-channel"
+ ? { handling: "transcribe" }
+ : CONFIG,
+ });
+ const res = await persistVideos({
+ paths,
+ items: [
+ { slug: "demo-channel", id: "saved11" },
+ { slug: "demo-channel", id: "tall111" },
+ { slug: "demo-channel", id: "fetch11" },
+ { slug: "bare-channel", id: "nourl11" },
+ { slug: "nope", id: "xyz" },
+ { slug: "demo-channel", id: "missing" },
+ { slug: "demo-channel", id: "../escape" },
+ // A duplicate collapses.
+ { slug: "demo-channel", id: "fetch11" },
+ ],
+ format: "video_720",
+ replace: "above-height",
+ dryRun: true,
+ deps: h.deps,
+ });
+ assert.equal(h.downloads.length, 0);
+ assert.equal(res.dryRun, true);
+ assert.deepEqual(res.plan.saved, { count: 1, items: [{ slug: "demo-channel", id: "saved11" }] });
+ assert.deepEqual(res.plan.wrongHeight.items, [{ slug: "demo-channel", id: "tall111" }]);
+ assert.deepEqual(res.plan.toFetch.items, [{ slug: "demo-channel", id: "fetch11" }]);
+ assert.deepEqual(res.plan.noUrl.items, [{ slug: "bare-channel", id: "nourl11" }]);
+ assert.deepEqual(
+ res.plan.unknown.items.map((i) => `${i.slug}/${i.id}`),
+ ["nope/xyz", "demo-channel/missing", "demo-channel/../escape"],
+ );
+ assert.equal(res.plan.willFetch, 2);
+ });
+});
diff --git a/common/controller/persistVideos.ts b/common/controller/persistVideos.ts
@@ -0,0 +1,523 @@
+import path from "node:path";
+import { readFile, rm, stat } from "node:fs/promises";
+import type { Paths } from "../lib/paths";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { SiteSettings } from "../lib/settings";
+import { getSettings } from "../lib/settings";
+import { diskGate, type DiskGateStatus } from "../lib/diskSpace";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import type { DownloadOutcomeRecord } from "../lib/downloadOutcome";
+import {
+ savedVideoPath,
+ type SavedVideoPointer,
+} from "../lib/savedVideo";
+import {
+ loadSavedVideo,
+ restoreSavedVideoProvenance,
+} from "../lib/savedVideo-server";
+import { readChannelConfig } from "./channels";
+import { findVideoSourceUrl } from "./undownloadedVideos";
+import {
+ downloadOneManaged,
+ sourceFetchFailure,
+} from "../ytdlp/downloadOneManaged";
+import {
+ resolveSourceVideoQuality,
+ VIDEO_720_MAX_HEIGHT,
+ type SourceVideoQuality,
+} from "../ytdlp/downloadFormat";
+import {
+ channelPaceSeconds,
+ channelPlatform,
+ pacingPlatformKey,
+} from "../ytdlp/channelArgs";
+import { staticSleepRequestsSeconds } from "../ytdlp/platformArgs.mjs";
+import { downloadGapMs } from "../jobs/platformBackoff";
+import { recordDownloadBackoff } from "../jobs/downloadBackoff";
+
+// PERSIST A LIST OF SPECIFIC VIDEOS, across channels, to the saved-video store.
+//
+// The general form of persistKept: that pass walks one channel's keep-latest
+// window, this one walks an explicit `[{slug, id}]` list — a report's cited
+// videos, a hand-picked set. Each video not yet saved has its source container
+// re-fetched through downloadOneManaged exactly as "Persist source video" does
+// (keepSourceVideoOverride + forceMedia), at the requested quality.
+//
+// RESUMABLE BY RE-RUNNING. Nothing is remembered between runs: a video already
+// saved is skipped, so the second run of the same list fetches only what the
+// first did not finish. That is also why the run stops — not skips — at a disk
+// floor or a rate limit: the rest of the list is left for the next run.
+//
+// PACED AND GATED, ONE VIDEO AT A TIME. Before each download: the gap since the
+// previous one (`gapMs`, default the batch downloads' own gap), the memory wait
+// (`minFreeMemMb`), and the disk floor. Cancel and drain are honoured between
+// videos; the one in flight is a cancel's business, as everywhere else.
+//
+// REPLACING NEVER DELETES FIRST. `replace: "above-height"` re-fetches a saved
+// video whose recorded height is unknown or above the requested quality's
+// ceiling. The old container stays where it is until the new pointer is
+// written; only then is it removed (when its path differs — a same-named one
+// was already replaced by the store's atomic move). The replacement keeps the
+// old pointer's `keepReason` and `origin`, so a keep-latest container stays
+// evictable and a requested one keeps its requester.
+
+export type PersistVideoItem = { slug: string; id: string };
+
+export const PERSIST_REPLACE_POLICIES = ["never", "above-height"] as const;
+export type PersistReplacePolicy = (typeof PERSIST_REPLACE_POLICIES)[number];
+
+export function isPersistReplacePolicy(v: unknown): v is PersistReplacePolicy {
+ return (
+ typeof v === "string" &&
+ (PERSIST_REPLACE_POLICIES as readonly string[]).includes(v)
+ );
+}
+
+export type PersistVideosBucket = { count: number; items: PersistVideoItem[] };
+
+// What a run would do with each item, decided before anything is fetched.
+// saved already in the store at an acceptable height — left alone
+// wrongHeight in the store, but its height is unknown or above the ceiling
+// of the requested quality; re-fetched only under "above-height"
+// toFetch not saved, with a resolvable source URL
+// noUrl not saved, and no source URL can be resolved
+// unknown no such channel, or no data/<id>/ for that video
+export type PersistVideosPlan = {
+ saved: PersistVideosBucket;
+ wrongHeight: PersistVideosBucket;
+ toFetch: PersistVideosBucket;
+ noUrl: PersistVideosBucket;
+ unknown: PersistVideosBucket;
+ // How many downloads a run would attempt: toFetch, plus wrongHeight when
+ // the replace policy re-fetches it.
+ willFetch: number;
+};
+
+export type PersistVideosStop = "low-disk" | "rate-limit" | "cancelled" | "drained";
+
+export type PersistVideosResult = {
+ plan: PersistVideosPlan;
+ dryRun: boolean;
+ // Newly saved this run (replacements included).
+ persisted: PersistVideoItem[];
+ // Of `persisted`, the ones that replaced an existing container.
+ replaced: PersistVideoItem[];
+ failed: (PersistVideoItem & { error: string })[];
+ // Due a download but not attempted because the run stopped first.
+ notAttempted: PersistVideoItem[];
+ // Why the run ended before its list did; null when it did not.
+ stopped: PersistVideosStop | null;
+};
+
+// The download one item takes. Injectable so the tests never reach yt-dlp.
+export type PersistVideoDownload = (opts: {
+ paths: Paths;
+ channelSlug: string;
+ channelConfig: ChannelConfig;
+ settings: SiteSettings;
+ videoUrl: string;
+ quality: SourceVideoQuality;
+ onLog: (line: string) => void;
+ signal: AbortSignal;
+}) => Promise<DownloadOutcomeRecord>;
+
+export type PersistVideosDeps = {
+ download: PersistVideoDownload;
+ sleep: (ms: number, signal?: AbortSignal) => Promise<void>;
+ // MemAvailable in MiB, or null when it cannot be read.
+ readMemAvailableMb: () => Promise<number | null>;
+ diskGate: (paths: Paths, settings: SiteSettings) => Promise<DiskGateStatus>;
+ getSettings: () => SiteSettings;
+ readChannelConfig: (paths: Paths, slug: string) => Promise<ChannelConfig | null>;
+ // A rate-limit or network failure, recorded against the platform's pacing.
+ onPlatformBackoff: (
+ platform: string,
+ paths: Paths,
+ failureClass: "rate_limit" | "network",
+ ) => Promise<void>;
+ removeFile: (file: string) => Promise<void>;
+};
+
+// How often the memory wait re-reads /proc/meminfo.
+export const MEM_POLL_MS = 10_000;
+
+const defaultDownload: PersistVideoDownload = (o) =>
+ downloadOneManaged({
+ channelSlug: o.channelSlug,
+ channelConfig: o.channelConfig,
+ paths: o.paths,
+ videoUrl: o.videoUrl,
+ onLog: o.onLog,
+ signal: o.signal,
+ cookiePolicy: resolveCookiePolicy(o.settings, o.channelConfig),
+ inlineTranscribeOnFallback: o.settings.inlineTranscribeOnFallback,
+ globalSkipLiveDownloads: o.settings.skipLiveDownloads,
+ appendArchive: true,
+ keepSourceVideoOverride: true,
+ // Persist means persist — see persistKept.
+ forceMedia: true,
+ persistFormatPreset: o.quality,
+ });
+
+function abortableSleep(ms: number, signal?: AbortSignal): Promise<void> {
+ if (ms <= 0 || signal?.aborted) return Promise.resolve();
+ return new Promise((resolve) => {
+ const onAbort = () => {
+ clearTimeout(t);
+ resolve();
+ };
+ const t = setTimeout(() => {
+ signal?.removeEventListener("abort", onAbort);
+ resolve();
+ }, ms);
+ signal?.addEventListener("abort", onAbort, { once: true });
+ });
+}
+
+export async function readMemAvailableMb(): Promise<number | null> {
+ try {
+ const raw = await readFile("/proc/meminfo", "utf8");
+ const m = /^MemAvailable:\s+(\d+)\s+kB/m.exec(raw);
+ return m ? Math.floor(Number(m[1]) / 1024) : null;
+ } catch {
+ return null;
+ }
+}
+
+const DEFAULT_DEPS: PersistVideosDeps = {
+ download: defaultDownload,
+ sleep: abortableSleep,
+ readMemAvailableMb,
+ diskGate: (paths, settings) => diskGate(paths, settings),
+ getSettings,
+ readChannelConfig,
+ onPlatformBackoff: (platform, paths, failureClass) =>
+ recordDownloadBackoff(platform, paths, failureClass),
+ removeFile: (file) => rm(file, { force: true }),
+};
+
+// The tallest height a quality accepts, or null for "original" (no ceiling).
+export function sourceVideoQualityMaxHeight(
+ quality: SourceVideoQuality,
+): number | null {
+ return quality === "video_720" ? VIDEO_720_MAX_HEIGHT : null;
+}
+
+// A video id is a directory name under data/: one path segment, never a walk.
+function isPlainVideoId(id: string): boolean {
+ return (
+ id !== "" &&
+ id !== "." &&
+ id !== ".." &&
+ !id.includes("/") &&
+ !id.includes("\\") &&
+ !id.includes("\0")
+ );
+}
+
+async function isDirectory(p: string): Promise<boolean> {
+ try {
+ return (await stat(p)).isDirectory();
+ } catch {
+ return false;
+ }
+}
+
+type PlannedItem = PersistVideoItem & {
+ bucket: Exclude<keyof PersistVideosPlan, "willFetch">;
+ config?: ChannelConfig;
+ quality?: SourceVideoQuality;
+ url?: string;
+ pointer?: SavedVideoPointer;
+};
+
+function emptyBucket(): PersistVideosBucket {
+ return { count: 0, items: [] };
+}
+
+// Classify every item. Duplicates collapse (first occurrence wins the order).
+async function classify(
+ paths: Paths,
+ items: PersistVideoItem[],
+ format: SourceVideoQuality | undefined,
+ settings: SiteSettings,
+ deps: PersistVideosDeps,
+): Promise<PlannedItem[]> {
+ const configs = new Map<string, ChannelConfig | null>();
+ const seen = new Set<string>();
+ const out: PlannedItem[] = [];
+ for (const raw of items) {
+ const item = { slug: raw.slug, id: raw.id };
+ const key = `${item.slug}\0${item.id}`;
+ if (seen.has(key)) continue;
+ seen.add(key);
+ if (!configs.has(item.slug)) {
+ configs.set(item.slug, await deps.readChannelConfig(paths, item.slug));
+ }
+ const config = configs.get(item.slug) ?? null;
+ if (!config || !isPlainVideoId(item.id)) {
+ out.push({ ...item, bucket: "unknown" });
+ continue;
+ }
+ const videoDir = path.join(paths.channelsDir, item.slug, "data", item.id);
+ if (!(await isDirectory(videoDir))) {
+ out.push({ ...item, bucket: "unknown" });
+ continue;
+ }
+ const quality = resolveSourceVideoQuality({
+ override: format,
+ channel: config.sourceVideoQuality,
+ global: settings.sourceVideoQuality,
+ });
+ const pointer = await loadSavedVideo(videoDir);
+ const url =
+ (await findVideoSourceUrl(paths, item.slug, item.id, config)) ?? undefined;
+ if (pointer) {
+ const ceiling = sourceVideoQualityMaxHeight(quality);
+ const height = pointer.format?.height;
+ const tooTall = ceiling !== null && (height === undefined || height > ceiling);
+ out.push({
+ ...item,
+ bucket: tooTall ? "wrongHeight" : "saved",
+ config,
+ quality,
+ url,
+ pointer,
+ });
+ continue;
+ }
+ out.push({ ...item, bucket: url ? "toFetch" : "noUrl", config, quality, url });
+ }
+ return out;
+}
+
+function planOf(
+ planned: PlannedItem[],
+ replace: PersistReplacePolicy,
+): PersistVideosPlan {
+ const plan: PersistVideosPlan = {
+ saved: emptyBucket(),
+ wrongHeight: emptyBucket(),
+ toFetch: emptyBucket(),
+ noUrl: emptyBucket(),
+ unknown: emptyBucket(),
+ willFetch: 0,
+ };
+ for (const p of planned) {
+ plan[p.bucket].items.push({ slug: p.slug, id: p.id });
+ plan[p.bucket].count += 1;
+ }
+ plan.willFetch =
+ plan.toFetch.count + (replace === "above-height" ? plan.wrongHeight.count : 0);
+ return plan;
+}
+
+function label(item: PersistVideoItem): string {
+ return `${item.slug}/${item.id}`;
+}
+
+export async function persistVideos({
+ paths,
+ items,
+ format,
+ replace = "never",
+ gapMs,
+ minFreeMemMb = 0,
+ dryRun = false,
+ onLog,
+ signal,
+ drainSignal,
+ deps: depsOverride,
+}: {
+ paths: Paths;
+ items: PersistVideoItem[];
+ // Absent = each channel's effective quality (resolveSourceVideoQuality).
+ format?: SourceVideoQuality;
+ replace?: PersistReplacePolicy;
+ // The pause before each download after the first. Absent = the batch
+ // downloads' own gap for the item's channel (downloadGapMs).
+ gapMs?: number;
+ // Before each download, wait until MemAvailable is at least this. 0 = off.
+ minFreeMemMb?: number;
+ dryRun?: boolean;
+ onLog?: (line: string) => void;
+ signal?: AbortSignal;
+ drainSignal?: AbortSignal;
+ deps?: Partial<PersistVideosDeps>;
+}): Promise<PersistVideosResult> {
+ const deps: PersistVideosDeps = { ...DEFAULT_DEPS, ...depsOverride };
+ const log = (line: string) => onLog?.(line.endsWith("\n") ? line : `${line}\n`);
+ const downloadLog = (line: string) => onLog?.(line);
+ const downloadSignal = signal ?? new AbortController().signal;
+ const settings = deps.getSettings();
+
+ const planned = await classify(paths, items, format, settings, deps);
+ const plan = planOf(planned, replace);
+ const result: PersistVideosResult = {
+ plan,
+ dryRun,
+ persisted: [],
+ replaced: [],
+ failed: [],
+ notAttempted: [],
+ stopped: null,
+ };
+ log(
+ `Persist videos: ${planned.length} item(s) — ${plan.toFetch.count} to fetch, ` +
+ `${plan.saved.count} already saved, ${plan.wrongHeight.count} saved above the ` +
+ `requested height${replace === "above-height" ? " (to replace)" : " (left alone)"}, ` +
+ `${plan.noUrl.count} with no source URL, ${plan.unknown.count} unknown.`,
+ );
+ for (const p of planned) {
+ if (p.bucket === "unknown") log(` ${label(p)}: unknown channel or video, skipping.`);
+ else if (p.bucket === "noUrl") log(` ${label(p)}: no resolvable source URL, skipping.`);
+ else if (p.bucket === "wrongHeight" && replace === "above-height" && !p.url) {
+ log(` ${label(p)}: no resolvable source URL to replace it from, skipping.`);
+ }
+ }
+ if (dryRun) return result;
+
+ const work = planned.filter(
+ (p) =>
+ p.bucket === "toFetch" ||
+ (p.bucket === "wrongHeight" && replace === "above-height" && p.url),
+ );
+ let attempted = 0;
+ const stop = (why: PersistVideosStop, from: number) => {
+ result.stopped = why;
+ for (const p of work.slice(from)) {
+ result.notAttempted.push({ slug: p.slug, id: p.id });
+ }
+ };
+ const interrupted = (): PersistVideosStop | null =>
+ signal?.aborted ? "cancelled" : drainSignal?.aborted ? "drained" : null;
+
+ for (let i = 0; i < work.length; i++) {
+ const p = work[i];
+ const config = p.config!;
+ const quality = p.quality!;
+ const why = interrupted();
+ if (why) {
+ stop(why, i);
+ break;
+ }
+ if (attempted > 0) {
+ const gap =
+ gapMs ??
+ downloadGapMs(
+ config.sleepBetweenDownloadsSeconds ?? settings.sleepBetweenDownloadsSeconds,
+ channelPaceSeconds(config),
+ staticSleepRequestsSeconds(channelPlatform(config)),
+ );
+ if (gap > 0) {
+ log(`Sleeping ${gap / 1000}s before the next download...`);
+ await deps.sleep(gap, signal);
+ const after = interrupted();
+ if (after) {
+ stop(after, i);
+ break;
+ }
+ }
+ }
+ if (minFreeMemMb > 0) {
+ let waited = false;
+ for (;;) {
+ const mem = await deps.readMemAvailableMb();
+ // Unreadable is not low: a platform with no /proc/meminfo is not held.
+ if (mem === null || mem >= minFreeMemMb) break;
+ if (!waited) {
+ log(` ${label(p)}: ${mem} MiB available, waiting for ${minFreeMemMb} MiB...`);
+ waited = true;
+ }
+ await deps.sleep(MEM_POLL_MS, signal);
+ if (interrupted()) break;
+ }
+ const after = interrupted();
+ if (after) {
+ stop(after, i);
+ break;
+ }
+ }
+ // PER ITEM, as persistKept: each download writes a full container, so the
+ // tenth must not inherit the first one's headroom. A closed gate ends the
+ // run; the rest are left for the next one.
+ const gate = await deps.diskGate(paths, settings);
+ if (!gate.ok) {
+ log(` ${label(p)}: ${gate.message} — stopping; the rest are left for a later run.`);
+ stop("low-disk", i);
+ break;
+ }
+ const replacing = p.bucket === "wrongHeight";
+ log(
+ ` ${label(p)}: ${replacing ? "replacing the saved container" : "fetching the source container"} (${quality})…`,
+ );
+ attempted += 1;
+ const videoDir = path.join(paths.channelsDir, p.slug, "data", p.id);
+ let record: DownloadOutcomeRecord;
+ try {
+ record = await deps.download({
+ paths,
+ channelSlug: p.slug,
+ channelConfig: config,
+ settings,
+ videoUrl: p.url!,
+ quality,
+ onLog: downloadLog,
+ signal: downloadSignal,
+ });
+ } catch (e) {
+ result.failed.push({ slug: p.slug, id: p.id, error: (e as Error).message });
+ log(` ${label(p)}: persist failed — ${(e as Error).message}`);
+ continue;
+ }
+ const failure = sourceFetchFailure(record);
+ if (failure) {
+ result.failed.push({ slug: p.slug, id: p.id, error: failure });
+ log(` ${label(p)}: persist failed — ${failure}`);
+ const cls = record.failureClass;
+ if (cls === "rate_limit" || cls === "network") {
+ await deps
+ .onPlatformBackoff(pacingPlatformKey(config), paths, cls)
+ .catch(() => {});
+ log(` ${label(p)}: ${cls.replace("_", " ")} — stopping; the rest are left for a later run.`);
+ stop("rate-limit", i + 1);
+ break;
+ }
+ continue;
+ }
+ // A download that returned is not a saved container: the move into the
+ // store can fail and leave it in the data dir. The pointer is the proof —
+ // and for a replacement, a NEW pointer.
+ const pointer = await loadSavedVideo(videoDir);
+ if (!pointer || (p.pointer && pointer.storedAt === p.pointer.storedAt)) {
+ const error = "the container was not saved to the store";
+ result.failed.push({ slug: p.slug, id: p.id, error });
+ log(` ${label(p)}: persist failed — ${error}`);
+ continue;
+ }
+ if (p.pointer) {
+ // The same video's container, so it keeps the old pointer's retention
+ // class and requester: the forced persist writes "override", which
+ // would make a keep-latest container permanent.
+ await restoreSavedVideoProvenance(videoDir, p.pointer);
+ const oldFile = savedVideoPath(p.pointer);
+ if (oldFile !== savedVideoPath(pointer)) {
+ await deps.removeFile(oldFile).catch(() => {});
+ log(` ${label(p)}: removed the replaced container ${oldFile}.`);
+ }
+ result.replaced.push({ slug: p.slug, id: p.id });
+ }
+ result.persisted.push({ slug: p.slug, id: p.id });
+ }
+
+ log(
+ `Persist videos: ${result.persisted.length} persisted` +
+ (result.replaced.length ? ` (${result.replaced.length} replaced)` : "") +
+ `, ${result.failed.length} failed, ${plan.saved.count} already saved` +
+ (result.notAttempted.length
+ ? `, ${result.notAttempted.length} not attempted (stopped: ${result.stopped})`
+ : "") +
+ ".",
+ );
+ return result;
+}
diff --git a/common/jobs/jobKinds.test.ts b/common/jobs/jobKinds.test.ts
@@ -82,6 +82,8 @@ const ADDED_KINDS: Record<string, { label: string; drainable: boolean }> = {
label: "Build & deploy homepage",
drainable: false,
},
+ // A list of videos persisted one at a time: a drain stops between them.
+ "persist-videos": { label: "Persist videos", drainable: true },
};
test("added kinds carry their pinned label and drainability", () => {
@@ -162,6 +164,7 @@ const STILL_MEDIA = [
"redownload-incomplete-bucket",
"retry-bucket",
"persist-kept",
+ "persist-videos",
"whisper-video",
"transcribe-one",
"download-one-pipeline",
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -388,6 +388,17 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
queueKeyStrategy: "custom",
needsMedia: true,
},
+ // Persist an explicit list of videos (controller/persistVideos.ts): one job
+ // per channel, on the channel's download queue as persist-kept is. Drainable:
+ // it stops between videos and leaves the rest for a re-run.
+ "persist-videos": {
+ kind: "persist-videos",
+ label: "Persist videos",
+ drainable: true,
+ replayable: true,
+ queueKeyStrategy: "custom",
+ needsMedia: true,
+ },
"backup-saved-videos": {
kind: "backup-saved-videos",
label: "Back up saved videos",
diff --git a/common/lib/savedVideo-server.ts b/common/lib/savedVideo-server.ts
@@ -151,6 +151,25 @@ export async function updateSavedVideoChecksum(
await writePointer(videoDir, { ...pointer, sha256 });
}
+// Put a replaced pointer's provenance back on its replacement: the old
+// `keepReason` (absent stays absent — a legacy pointer the prune already treats
+// as non-evictable) and `origin`. A persist that re-fetched a container is the
+// same video's container, not a new decision about why it is kept. No-op when
+// there is no pointer.
+export async function restoreSavedVideoProvenance(
+ videoDir: string,
+ from: Pick<SavedVideoPointer, "keepReason" | "origin">,
+): Promise<void> {
+ const pointer = await loadSavedVideo(videoDir);
+ if (!pointer) return;
+ const { keepReason: _k, origin: _o, ...rest } = pointer;
+ await writePointer(videoDir, {
+ ...rest,
+ ...(from.keepReason ? { keepReason: from.keepReason } : {}),
+ ...(from.origin ?? pointer.origin ? { origin: from.origin ?? pointer.origin } : {}),
+ });
+}
+
// Best-effort removal of a now-empty store dir (and its empty <slug> parent).
async function pruneEmptyStoreDirs(storeDir: string): Promise<void> {
await rm(storeDir, { recursive: false, force: true }).catch(() => {});
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -3,6 +3,7 @@
## [Unreleased]
- **umtool's report videos can show a highlighted sentence from a saved article.** `node umtool/report-to-video/shoot-page.mjs --page <saved page.html> --quote "<sentence>" --out <shot.png>` opens a web page saved to disk, finds the sentence in its text, highlights it and saves a PNG of the paragraph that holds it, ready to be a report manifest's `image` entry. `--batch <items.json> --out <dir>` does a list of `{ id, page, quote, context? }` at once and writes `<id>.png` for each plus a `results.json` recording each shot's crop, the matched text and the block it shot. The page is opened offline: nothing is fetched except files saved beside it, and its own scripts do not run unless `--js` is given. The sentence is found whether its quotes and apostrophes are curly or straight, across links and emphasis, and through non-breaking spaces, soft hyphens and line breaks in the page's source. A sentence that is not on the page is listed in `results.json` and on the terminal, and the run ends with an error rather than leaving it out. `--color` sets the highlight; `context` picks one occurrence of a sentence that appears more than once.
- **A clip or whole-recording fetch can name the tallest source video it wants.** The MCP's `fetch_clip` takes `maxHeight`, `fetch-via-editor.mjs` takes `--max-height`, and the editor's fetch endpoint takes `maxHeight`: a whole number of pixels from 144 to 2160; anything else is refused before anything is fetched. A window is fetched at or under that height (720 when none is given, as before). A whole recording asked for at 720 or less is saved as the **Video 720p** quality, and above 720 at the original quality; with no height it follows the channel's, else the global, source video quality, as before. umtool's whole-source fetch from the clip bench now asks at the report's `render.maxHeightSource`. A file already on disk is returned as it is and never fetched again for a different height; the answer now gives its height (a window's is read from the file, a whole recording's from what its persist recorded) and says when it is taller than the height asked for.
+- **Persist a list of videos, across channels, to the saved-video store.** `pnpm ops persist-videos --json '{"items":[{"slug":"<channel>","id":"<video id>"}, …]}'` (or `--file list.json` for a long list) re-fetches the source container of each video that is not saved yet, the way **Persist source video** does on a video's page. `"format": "original"` or `"video_720"` picks the quality (default: each channel's own), and `"replace": "above-height"` also re-fetches a saved video whose recorded height is unknown or above that quality — the old file is removed only after the new one is saved, and the saved video keeps its retention class. Downloads run one at a time, one job per channel on that channel's download queue, with the usual gap between them (`"gapMs"` overrides it); `"minFreeMemMb"` holds each download until that much memory is free. A low disk or a rate limit stops the job, and running the same list again picks up where it left off: saved videos are skipped. `"dryRun": true` answers with what a run would do — saved, saved above the height, to fetch, no source URL, unknown — and starts nothing. Jobs show as **Persist videos**, can be drained, and can be retried from /jobs.
- **Capture specific X posts: a screenshot of each, and its attached media.** `pnpm ops capture-posts --json '{"slug":"<channel>","ids":["<post id>", …]}'` shoots each post as X shows it, through the connected X profile, and downloads its pictures and videos with gallery-dl, into the channel's `posts-media/<post id>/` beside a `capture.json` that records when, from which URLs, and each file's size and SHA-256. Every id must already be in the channel's posts archive; one that is not is refused by name and nothing runs. `"shots": false` or `"media": false` skips that half, and posts already captured are skipped unless `"force": true`. The job runs on the X queue with a post fetch, so the two never run at once, and waits a random 4–10 seconds before each request to X, as fetches do. A deleted post, or one behind its account's wall (protected, suspended, gone), is recorded as such in the channel's deleted-post record; a post behind a sensitive-media warning is opened and shot. If X asks to log in, or answers "Something went wrong", the job stops at that post and leaves the rest for a later run. Captures are never published: the export does not read them.
- **A source video can be saved at 720p for clip and editing work.** **Persist source video** on a video's page has a **Quality** select: **Original** (the best video and audio, as every persist has been) or **Video 720p (H.264, for clips/editing)**, which saves an H.264 mp4 at most 720p tall — smaller, and quick to cut. When a source has nothing at or under 720p in H.264 it takes 480p, and when it has neither it takes whatever is best and says so in the job's log, with the height it got. The default is the new **Source video quality** under **Settings**, which a channel can override in its Advanced settings; the whole-recording fetch (`full: true`) and **Persist kept now** follow the channel's, else the global, choice. A persisted video's **Source video** card now shows the format that was saved (height and codec) and the quality asked for; videos persisted before this show nothing new. "Video 720p" is also offered as a download format.
- **A report cut can be a fact-check: each claim gets a verdict stamp over the footage, and a tally in the on-screen deck counts them.** Any clip, still or card entry of a report manifest can say which claim it is evidence for and the verdict, `"claim": { "id": "k3", "verdict": "CONTRADICTED" }` (one of CORROBORATED, PARTLY, CONTRADICTED, NOT_FOUND, UNTESTABLE). At the end of the last entry carrying each claim, the verdict slams in over the picture as a stamp in its colour and leaves with the transition, and a row of counts in the deck, one per verdict the cut uses, steps up as each stamp lands. `render.chrome.factcheck` sets each verdict's label and colour, how long the stamp is up (1–10 seconds, 3 by default) and which corner of the footage it sits in, and whether the tally is drawn beside the QR or before the title. A claim is a new key, separate from the clip review's `verdict`; one claim id given two verdicts, a claim on a teaser, or an unknown setting is refused with a sentence before a build fetches anything. umtool's On-screen table has a **claim** column (an id and a verdict per row), saved with the titles, and its preview shows the stamps and the tally before a build. The deck's QR can link each clip's original instead of the archive with `render.chrome.deck.qr.links: "original"` — a YouTube video at the clip's second, a Rumble page or an X post as they are; a clip's `citeUrl` still wins. A manifest post can carry `shot`, a screenshot (PNG, JPEG or WebP, beside the manifest), which its card draws in place of the post's text, its QR kept. A cut with none of these builds exactly as before.
diff --git a/editor/app/api/ops/_lib.test.ts b/editor/app/api/ops/_lib.test.ts
@@ -1,6 +1,12 @@
import test from "node:test";
import assert from "node:assert/strict";
-import { OpsInputError, optPositiveInt, optSubset } from "./_lib";
+import {
+ OpsInputError,
+ optNonNegativeInt,
+ optPositiveInt,
+ optSubset,
+ reqVideoItems,
+} from "./_lib";
// Run with:
// pnpm -C editor exec tsx --test "app/**/*.test.ts"
@@ -52,3 +58,57 @@ test("optPositiveInt: zero, negatives, fractions and strings are refused", () =>
);
}
});
+
+test("optNonNegativeInt: absent is undefined; zero and whole numbers come back", () => {
+ assert.equal(optNonNegativeInt({}, "gapMs"), undefined);
+ assert.equal(optNonNegativeInt({ gapMs: 0 }, "gapMs"), 0);
+ assert.equal(optNonNegativeInt({ gapMs: 30000 }, "gapMs"), 30000);
+});
+
+test("optNonNegativeInt: negatives, fractions and strings are refused", () => {
+ for (const gapMs of [-1, 1.5, "10", null, Number.NaN]) {
+ assert.throws(
+ () => optNonNegativeInt({ gapMs }, "gapMs"),
+ (e: unknown) =>
+ e instanceof OpsInputError && /"gapMs" must be a whole number, zero or above/.test(e.message),
+ JSON.stringify(gapMs),
+ );
+ }
+});
+
+test("reqVideoItems: a list of { slug, id } comes back trimmed, in order", () => {
+ assert.deepEqual(
+ reqVideoItems(
+ { items: [{ slug: "demo-channel", id: " abc123 " }, { slug: "other", id: "x" }] },
+ "items",
+ ),
+ [
+ { slug: "demo-channel", id: "abc123" },
+ { slug: "other", id: "x" },
+ ],
+ );
+});
+
+test("reqVideoItems: an empty list, a non-object, an extra key or a missing id is refused", () => {
+ for (const items of [
+ undefined,
+ [],
+ "demo-channel/abc123",
+ ["abc123"],
+ [{ slug: "demo-channel" }],
+ [{ slug: "demo-channel", id: "abc123", height: 720 }],
+ ]) {
+ assert.throws(() => reqVideoItems({ items }, "items"), OpsInputError, JSON.stringify(items));
+ }
+});
+
+test("reqVideoItems: a traversing slug or id is refused at the door", () => {
+ for (const item of [
+ { slug: "../../escape", id: "abc123" },
+ { slug: "demo-channel", id: "../abc123" },
+ { slug: "demo-channel", id: ".." },
+ { slug: "demo-channel", id: "a\\b" },
+ ]) {
+ assert.throws(() => reqVideoItems({ items: [item] }, "items"), OpsInputError, JSON.stringify(item));
+ }
+});
diff --git a/editor/app/api/ops/_lib.ts b/editor/app/api/ops/_lib.ts
@@ -157,6 +157,49 @@ export function optPositiveInt(body: OpsBody, key: string): number | undefined {
return v;
}
+// A whole number, zero or above, or undefined when absent — for a duration or
+// a floor where 0 means "off".
+export function optNonNegativeInt(body: OpsBody, key: string): number | undefined {
+ const v = body[key];
+ if (v === undefined) return undefined;
+ if (typeof v !== "number" || !Number.isInteger(v) || v < 0) {
+ throw new OpsInputError(`"${key}" must be a whole number, zero or above`);
+ }
+ return v;
+}
+
+// A list of videos across channels: `[{ slug, id }, …]`. Every slug is a
+// CHANNEL SLUG (see reqSlug) and every id one path segment, checked here so a
+// traversing entry is refused at the door rather than read as an unknown video.
+export function reqVideoItems(
+ body: OpsBody,
+ key: string,
+): { slug: string; id: string }[] {
+ const v = body[key];
+ if (!Array.isArray(v) || v.length === 0) {
+ throw new OpsInputError(
+ `"${key}" is required and must be a non-empty array of { "slug", "id" }`,
+ );
+ }
+ return v.map((entry, i) => {
+ if (typeof entry !== "object" || entry === null || Array.isArray(entry)) {
+ throw new OpsInputError(`"${key}[${i}]" must be an object { "slug", "id" }`);
+ }
+ const extra = Object.keys(entry).filter((k) => k !== "slug" && k !== "id");
+ if (extra.length) {
+ throw new OpsInputError(
+ `"${key}[${i}]" has unknown key(s): ${extra.join(", ")} — an item is { "slug", "id" }`,
+ );
+ }
+ const slug = reqSlug(entry as OpsBody, "slug");
+ const id = reqString(entry as OpsBody, "id");
+ if (id === "." || id === ".." || /[/\\\0]/.test(id)) {
+ throw new OpsInputError(`"${id}" is not a video id (one path segment)`);
+ }
+ return { slug, id };
+ });
+}
+
export function reqStringArray(body: OpsBody, key: string): string[] {
const v = body[key];
if (
diff --git a/editor/app/api/ops/persist-videos/route.ts b/editor/app/api/ops/persist-videos/route.ts
@@ -0,0 +1,65 @@
+import { NextResponse } from "next/server";
+import { SOURCE_VIDEO_QUALITIES } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
+import { PERSIST_REPLACE_POLICIES } from "yt-dlp-transcript-common/controller/persistVideos";
+import { persistVideosAction } from "../../../channels/[slug]/persistActions";
+import {
+ oneOf,
+ ops,
+ opsFail,
+ optBool,
+ optNonNegativeInt,
+ optString,
+ reqVideoItems,
+} from "../_lib";
+
+export const dynamic = "force-dynamic";
+
+// POST { items: [{ slug, id }, …], format?, replace?, gapMs?, minFreeMemMb?,
+// dryRun?, queueKey? }
+// dryRun -> { ok: true, dryRun: true, plan }
+// else -> { ok: true, dryRun: false, plan, jobs: [{ slug, jobId }], jobIds,
+// jobId (one job only), skipped: [{ slug, reason }] }
+//
+// Persist specific videos, across channels, to the saved-video store: the
+// video page's "Persist source video" over a list. `plan` buckets every item —
+// saved, wrongHeight, toFetch, noUrl, unknown — with counts and ids, and
+// `willFetch` is how many downloads a run makes. A real run starts one job per
+// channel with something to fetch, on that channel's download queue.
+//
+// `format` ("original" | "video_720") defaults to each channel's own quality.
+// `replace: "above-height"` also re-fetches a saved container whose height is
+// unknown or above that quality's ceiling (default "never"). `gapMs` is the
+// pause between downloads (default the batch downloads' gap), `minFreeMemMb`
+// holds each download until that much memory is available (default 0, off).
+// Re-running the same body is the resume: saved videos are skipped.
+//
+// A big list goes in a file: `pnpm ops persist-videos --file list.json`.
+export async function POST(request: Request) {
+ return ops(
+ request,
+ ["items", "format", "replace", "gapMs", "minFreeMemMb", "dryRun", "queueKey"],
+ async (body) => {
+ const result = await persistVideosAction({
+ items: reqVideoItems(body, "items"),
+ format:
+ body.format === undefined
+ ? undefined
+ : oneOf(body, "format", SOURCE_VIDEO_QUALITIES),
+ replace:
+ body.replace === undefined
+ ? undefined
+ : oneOf(body, "replace", PERSIST_REPLACE_POLICIES),
+ gapMs: optNonNegativeInt(body, "gapMs"),
+ minFreeMemMb: optNonNegativeInt(body, "minFreeMemMb"),
+ dryRun: optBool(body, "dryRun"),
+ queueKey: optString(body, "queueKey"),
+ });
+ if (!result.ok) return opsFail(result.error);
+ if (result.dryRun) return NextResponse.json(result);
+ return NextResponse.json({
+ ...result,
+ ...(result.jobIds.length === 1 ? { jobId: result.jobIds[0] } : {}),
+ });
+ },
+ );
+}
diff --git a/editor/app/channels/[slug]/persistActions.ts b/editor/app/channels/[slug]/persistActions.ts
@@ -12,10 +12,38 @@ import {
import { readChannelConfig } from "yt-dlp-transcript-common/controller/channels";
import { persistKept } from "yt-dlp-transcript-common/controller/persistKept";
import {
+ persistVideos,
+ type PersistReplacePolicy,
+ type PersistVideoItem,
+ type PersistVideosPlan,
+} from "yt-dlp-transcript-common/controller/persistVideos";
+import type { SourceVideoQuality } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
+import {
runManagedFunction,
type StreamActionResult,
} from "yt-dlp-transcript-common/jobs/streamCommand";
+export type PersistVideosOptions = {
+ // Absent = each channel's effective source-video quality.
+ format?: SourceVideoQuality;
+ replace?: PersistReplacePolicy;
+ gapMs?: number;
+ minFreeMemMb?: number;
+ queueKey?: string;
+};
+
+export type PersistVideosActionResult =
+ | { ok: false; error: string }
+ | { ok: true; dryRun: true; plan: PersistVideosPlan }
+ | {
+ ok: true;
+ dryRun: false;
+ plan: PersistVideosPlan;
+ jobs: { slug: string; jobId: string }[];
+ jobIds: string[];
+ skipped: { slug: string; reason: string }[];
+ };
+
// Bulk catch-up: ensure every video in the channel's keep-latest window has its
// source container saved to the store. Re-downloads only those not already
// saved. Runs on the channel's download queue since it issues real downloads.
@@ -29,16 +57,8 @@ export async function persistKeptAction(
// The operator clicked this: only the floor applies and the shared hysteresis
// latch is left alone (see diskGate). This is a preflight only — persistKept
// re-checks per video, because it writes full containers in a loop.
- const disk = await diskGate(paths, getSettings(), { mode: "manual" });
- if (!disk.ok) {
- return {
- ok: false,
- error:
- `Low disk space: ${formatBytes(disk.freeBytes)} free, ` +
- `${formatBytes(disk.thresholdBytes)} required. Free up space or ` +
- `lower the floor in Settings.`,
- };
- }
+ const disk = await persistDiskPreflight();
+ if (disk) return { ok: false, error: disk };
return runManagedFunction({
kind: "persist-kept",
queueKey: resolveQueueKey(downloadQueueKey(config), queueKey),
@@ -57,3 +77,142 @@ export async function persistKeptAction(
},
});
}
+
+// Persist an explicit list of one channel's videos (controller/persistVideos.ts)
+// as one job on the channel's download queue — persist-kept's queue, since it
+// issues real downloads. The job carries its ids and options in its spec, so a
+// replay re-runs the same list, and a re-run fetches only what is still not
+// saved.
+export async function persistChannelVideosAction(
+ slug: string,
+ ids: string[],
+ opts: PersistVideosOptions = {},
+): Promise<StreamActionResult> {
+ const wanted = [...new Set(ids)];
+ if (wanted.length === 0) return { ok: false, error: "No video ids to persist." };
+ const paths = getPaths();
+ const config = await readChannelConfig(paths, slug);
+ if (!config) return { ok: false, error: `Channel "${slug}" not found` };
+ const disk = await persistDiskPreflight();
+ if (disk) return { ok: false, error: disk };
+ const { format, replace, gapMs, minFreeMemMb, queueKey } = opts;
+ return runManagedFunction({
+ kind: "persist-videos",
+ queueKey: resolveQueueKey(downloadQueueKey(config), queueKey),
+ paths,
+ channelSlug: slug,
+ spec: {
+ kind: "persist-videos",
+ slug,
+ params: { queueKey, ids: wanted, format, replace, gapMs, minFreeMemMb },
+ },
+ fn: async (onLog, signal, _progress, ctx) => {
+ const result = await persistVideos({
+ paths,
+ items: wanted.map((id) => ({ slug, id })),
+ format,
+ replace,
+ gapMs,
+ minFreeMemMb,
+ onLog,
+ signal,
+ drainSignal: ctx.drainSignal,
+ });
+ safeRevalidate([`/channels/${slug}`, "/saved-videos"]);
+ // A run that stopped short or lost a video did not do what it was asked:
+ // the job says so, and the re-run picks up the rest.
+ if (result.stopped === "low-disk" || result.stopped === "rate-limit") {
+ throw new Error(
+ `Stopped (${result.stopped}) with ${result.notAttempted.length} video(s) not attempted — run it again later.`,
+ );
+ }
+ if (result.failed.length > 0) {
+ throw new Error(`${result.failed.length} video(s) failed to persist.`);
+ }
+ },
+ });
+}
+
+// The disk floor a persist asks before it queues (manual mode: the operator
+// asked). A preflight only — the controllers re-check per video.
+async function persistDiskPreflight(): Promise<string | null> {
+ const disk = await diskGate(getPaths(), getSettings(), { mode: "manual" });
+ if (disk.ok) return null;
+ return (
+ `Low disk space: ${formatBytes(disk.freeBytes)} free, ` +
+ `${formatBytes(disk.thresholdBytes)} required. Free up space or ` +
+ `lower the floor in Settings.`
+ );
+}
+
+// Persist a list of videos ACROSS channels. A dry run answers with the buckets
+// and starts nothing. A real run starts one persist-videos job per channel that
+// has something to fetch — so each runs on its channel's download queue, behind
+// that platform's other downloads, under the channel's media guard — and
+// reports the channels it refused, with the action's own sentence.
+export async function persistVideosAction(
+ input: { items: PersistVideoItem[]; dryRun?: boolean } & PersistVideosOptions,
+): Promise<PersistVideosActionResult> {
+ const { items, dryRun, ...opts } = input;
+ if (items.length === 0) return { ok: false, error: "No videos to persist." };
+ const paths = getPaths();
+ const planned = await persistVideos({
+ paths,
+ items,
+ format: opts.format,
+ replace: opts.replace,
+ dryRun: true,
+ });
+ const plan = planned.plan;
+ if (dryRun) return { ok: true, dryRun: true, plan };
+ if (plan.willFetch === 0) {
+ return { ok: true, dryRun: false, plan, jobs: [], jobIds: [], skipped: [] };
+ }
+ const disk = await persistDiskPreflight();
+ if (disk) return { ok: false, error: disk };
+ // Only the channels with work get a job, and each job gets that channel's
+ // WHOLE slice of the list: its own re-run skips what is saved.
+ const due = new Set(
+ [
+ ...plan.toFetch.items,
+ ...(opts.replace === "above-height" ? plan.wrongHeight.items : []),
+ ].map((i) => i.slug),
+ );
+ const bySlug = new Map<string, string[]>();
+ for (const { slug, id } of items) {
+ if (!due.has(slug)) continue;
+ bySlug.set(slug, [...(bySlug.get(slug) ?? []), id]);
+ }
+ const jobs: { slug: string; jobId: string }[] = [];
+ const skipped: { slug: string; reason: string }[] = [];
+ for (const [slug, ids] of bySlug) {
+ let result: StreamActionResult;
+ try {
+ result = await persistChannelVideosAction(slug, ids, opts);
+ } catch (e) {
+ skipped.push({ slug, reason: (e as Error).message });
+ continue;
+ }
+ if (!result.ok) {
+ skipped.push({ slug, reason: result.error });
+ continue;
+ }
+ // Nobody reads the stream: the job's log is on disk (see the ops layer).
+ void result.stream.cancel();
+ jobs.push({ slug, jobId: result.jobId });
+ }
+ if (jobs.length === 0) {
+ return {
+ ok: false,
+ error: skipped.map((s) => `${s.slug}: ${s.reason}`).join("; "),
+ };
+ }
+ return {
+ ok: true,
+ dryRun: false,
+ plan,
+ jobs,
+ jobIds: jobs.map((j) => j.jobId),
+ skipped,
+ };
+}
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -12,6 +12,8 @@ import { readChannelSnapshot } from "yt-dlp-transcript-common/controller/channel
import type { JobSpec, ReplayBucket } from "yt-dlp-transcript-common/jobs/jobSpec";
import type { StreamActionResult } from "yt-dlp-transcript-common/jobs/streamCommand";
import type { AudioFormat } from "yt-dlp-transcript-common/lib/channelConfig";
+import { isSourceVideoQuality } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
+import { isPersistReplacePolicy } from "yt-dlp-transcript-common/controller/persistVideos";
import {
downloadAction,
downloadMissingAction,
@@ -38,7 +40,10 @@ import {
type DigestLaneChoice,
} from "../channels/[slug]/digestActions";
import { backfillChannelAction } from "../channels/[slug]/backfillActions";
-import { persistKeptAction } from "../channels/[slug]/persistActions";
+import {
+ persistChannelVideosAction,
+ persistKeptAction,
+} from "../channels/[slug]/persistActions";
import {
capturePostsAction,
checkPostAvailabilityAction,
@@ -341,4 +346,16 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
const { queueKey } = params(spec);
return persistKeptAction(spec.slug, queueKey);
},
+ // The same list, the same options: a replay fetches whatever of it is still
+ // not saved (the controller skips the rest).
+ "persist-videos": (spec) => {
+ const { p, queueKey } = params(spec);
+ return persistChannelVideosAction(spec.slug, strings(p.ids) ?? [], {
+ queueKey,
+ format: isSourceVideoQuality(p.format) ? p.format : undefined,
+ replace: isPersistReplacePolicy(p.replace) ? p.replace : undefined,
+ gapMs: num(p.gapMs),
+ minFreeMemMb: num(p.minFreeMemMb),
+ });
+ },
};
diff --git a/editor/app/storage/lib/storeBusy.ts b/editor/app/storage/lib/storeBusy.ts
@@ -47,6 +47,7 @@ const STORE_TOUCHING_KINDS = new Set([
"sync",
// The store's own jobs.
"persist-kept",
+ "persist-videos",
"check-kept-deleted",
"backup-saved-videos",
"verify-saved-video-backup",
diff --git a/editor/e2e/ops-api.spec.ts b/editor/e2e/ops-api.spec.ts
@@ -194,6 +194,7 @@ test("a traversing slug is refused at the door, on every route that takes one",
["relocate-back", { slugs: ["../../escape"] }],
["fetch-posts", { slug: "../../escape", older: true }],
["capture-posts", { slug: "../../escape", ids: ["1"] }],
+ ["persist-videos", { items: [{ slug: "../../escape", id: "abc123" }] }],
];
for (const [action, data] of cases) {
const { status, body } = await ops(request, action, data);
@@ -1042,6 +1043,79 @@ test("keep-videos marks the matching video only, after a dry run that writes not
expect(badRegex.body.error).toContain("invalid pattern");
});
+test("persist-videos buckets a list in a dry run, and refuses a bad body and a low disk — before any job", async ({
+ request,
+}) => {
+ await resetData("one-youtube-channel-with-data");
+ await settings();
+ const slug = "test-youtube";
+ const video = "20240101_test1234567";
+ const before = await listJobIds();
+
+ type Bucket = { count: number; items: { slug: string; id: string }[] };
+ type PersistBody = OpsResponse & {
+ dryRun?: boolean;
+ plan?: Record<"saved" | "wrongHeight" | "toFetch" | "noUrl" | "unknown", Bucket> & {
+ willFetch: number;
+ };
+ };
+ const items = [
+ { slug, id: video },
+ { slug, id: "missing12345" },
+ { slug: "no-such-channel", id: "abc123" },
+ ];
+ const dry = await ops(request, "persist-videos", {
+ items,
+ format: "video_720",
+ dryRun: true,
+ });
+ expect(dry.status, JSON.stringify(dry.body)).toBe(200);
+ const plan = (dry.body as PersistBody).plan!;
+ expect((dry.body as PersistBody).dryRun).toBe(true);
+ expect(plan.toFetch.items).toEqual([{ slug, id: video }]);
+ expect(plan.unknown.count).toBe(2);
+ expect(plan.saved.count).toBe(0);
+ expect(plan.willFetch).toBe(1);
+
+ // Nothing to fetch is an answer, not a job.
+ const none = await ops(request, "persist-videos", {
+ items: [{ slug: "no-such-channel", id: "abc123" }],
+ });
+ expect(none.status, JSON.stringify(none.body)).toBe(200);
+ expect(none.body.jobIds).toEqual([]);
+
+ // The body's shape.
+ const cases: [Record<string, unknown>, RegExp][] = [
+ [{}, /"items" is required/],
+ [{ items: [] }, /"items" is required/],
+ [{ items: [{ slug, id: video, height: 720 }] }, /unknown key\(s\): height/],
+ [{ items: [{ slug, id: "../escape" }] }, /is not a video id/],
+ [{ items, format: "1080p" }, /"format" must be one of original, video_720/],
+ [{ items, replace: "always" }, /"replace" must be one of never, above-height/],
+ [{ items, gapMs: -1 }, /"gapMs" must be a whole number, zero or above/],
+ [{ items, minFreeMemMb: "4096" }, /"minFreeMemMb" must be a whole number/],
+ [{ items, dryRun: "yes" }, /"dryRun" must be a boolean/],
+ [{ items, ids: ["x"] }, /unknown key\(s\): ids/],
+ ];
+ for (const [data, error] of cases) {
+ const res = await ops(request, "persist-videos", data);
+ expect(res.status, JSON.stringify(data)).toBe(400);
+ expect(res.body.error, JSON.stringify(data)).toMatch(error);
+ }
+
+ // A real run asks the disk floor before it queues anything.
+ await writeSettings({ minFreeDiskGB: 1_000_000 });
+ try {
+ const low = await ops(request, "persist-videos", { items });
+ expect(low.status, JSON.stringify(low.body)).toBe(400);
+ expect(low.body.error).toMatch(/^Low disk space/);
+ } finally {
+ await settings();
+ }
+
+ expect(await listJobIds()).toEqual(before);
+});
+
// --- the two build routes speak the same body ---------------------------------
//
// They did not. build-site took `siteIds` (a list) and build-deploy took
diff --git a/scripts/archilyzer-ops.mjs b/scripts/archilyzer-ops.mjs
@@ -42,6 +42,7 @@
// pnpm ops get channel the-quartering
// pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}'
// pnpm ops tag-videos --file ids.json
+// pnpm ops persist-videos --file list.json --wait
// pnpm ops get tags eva-collab
// pnpm ops cut-release --json '{"workspace":"all","version":"next","commit":true}'
//
@@ -132,6 +133,9 @@ const ACTIONS = [
// Set the per-video do-not-clean marker on every video of a channel whose
// title/description matches a download-filter pattern ({slug, match}).
"keep-videos",
+ // Persist specific videos, across channels, to the saved-video store
+ // ({items: [{slug, id}]}), paced and gated; a re-run resumes.
+ "persist-videos",
// Cut a changelog's [Unreleased] into a dated release heading (release 10
// slice P). Synchronous. The same writer as `archilyzer release cut`, which
// needs no editor at all — this route exists only on an editor built from
@@ -335,6 +339,17 @@ export function usage() {
' "shots": false or "media": false skips that half; posts already captured',
' are skipped unless "force": true. Paced like a post fetch, on its queue.',
"",
+ 'persist-videos saves specific videos, across channels, to the saved-video',
+ ' store: {"items": [{"slug", "id"}, ...]}. "format": "original" |',
+ ' "video_720" (default: each channel\'s own). "replace": "above-height" also',
+ ' re-fetches a saved one whose height is unknown or above that quality',
+ ' (default "never"). "gapMs" pauses between downloads (default the batch',
+ ' gap), "minFreeMemMb" waits for that much free memory before each.',
+ ' "dryRun": true answers with the buckets (saved, wrongHeight, toFetch,',
+ ' noUrl, unknown) and starts nothing. One job per channel, on its download',
+ " queue; a low disk or a rate limit stops it, and running the same body",
+ " again resumes — saved videos are skipped.",
+ "",
'"preview": "<branch>" on deploy-site or build-deploy makes it a Cloudflare',
" Pages PREVIEW instead of production: the same bundle goes to a branch",
" alias, https://<branch>.<project>.pages.dev, and the live site is left",
diff --git a/scripts/archilyzer-ops.test.mjs b/scripts/archilyzer-ops.test.mjs
@@ -405,3 +405,22 @@ test("capture-posts is a POST to its route, named in the usage", () => {
assert.match(usage(), /Actions:.*fetch-posts, capture-posts/);
assert.match(usage(), /Every id must be in the channel's posts archive/);
});
+
+test("persist-videos is a POST to its route, named in the usage", () => {
+ const p = parseArgs([
+ "persist-videos",
+ "--json",
+ '{"items":[{"slug":"demo-channel","id":"abc123"}],"format":"video_720","dryRun":true}',
+ ]);
+ assert.equal(p.method, "POST");
+ assert.equal(p.path, "/api/ops/persist-videos");
+ assert.deepEqual(p.body, {
+ items: [{ slug: "demo-channel", id: "abc123" }],
+ format: "video_720",
+ dryRun: true,
+ });
+ assert.equal(p.defaultSource, undefined);
+ // A long list arrives from a file.
+ assert.equal(parseArgs(["persist-videos", "--file", "list.json"]).bodyFile, "list.json");
+ assert.match(usage(), /persist-videos/);
+});