commit 07c587f904d9a7359fed4965e4a6b1102ba4dc3c
parent 0c0598528a81d7ded837b6544fa7267ea840a9b2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 8 Oct 2026 23:13:12 -0400
umtool song: link-sources replaces fetch-sources -- the editor's persist-videos fetches, this links
media/<id>.mp4 becomes a symlink to the saved file, found through the
video's own saved-video.json pointer, and wav48/<id>.wav is written once
from it. The fetch itself is pnpm ops persist-videos: one paced job on the
channel's download queue, with its cookies and backoff.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
2 files changed, 113 insertions(+), 139 deletions(-)
diff --git a/umtool/song/fetch-sources.mjs b/umtool/song/fetch-sources.mjs
@@ -1,139 +0,0 @@
-#!/usr/bin/env node
-// Re-derive a song's SOURCE media -- media/<id>.mp4 and wav48/<id>.wav -- by
-// asking the editor for each whole recording, one at a time.
-//
-// render-poly reads both by absolute source seconds: wav48/<id>.wav whole into
-// memory (a missing one aborts the render) and media/<id>.mp4 per clip (a
-// missing or short one silently drops or slides that voice's layer). So these
-// are FULL recordings, not windows -- a window's t=0 is its `from`, and the plan
-// would have to be rewritten to use it.
-//
-// Every fetch goes through the editor (POST /api/media/fetch-window, full:
-// true): cookie policy, per-platform sleeps, the 429 cooldown, and the bytes
-// land in the archive's saved-video store with a note of who asked and why. The
-// song's copy is a HARDLINK to that file, not a second copy: the two trees share
-// a filesystem, the store is the durable one, and each name survives the other
-// being deleted. The link is named <id>.mp4 whatever the container is, because
-// render-poly asks for that name; ffmpeg reads the content, not the extension.
-//
-// Slow and steady, never a burst: one job in flight, PAUSE seconds between
-// them, a 409 cooldown waited out rather than retried. Three failures in a row
-// stops the run -- that pattern is a bot check, and it wants a person.
-//
-// WORKER_TOKEN=... node fetch-sources.mjs <plan.json>... [--channel the-quartering]
-//
-// EDITOR_URL the editor (http://localhost:3001)
-// PAUSE seconds between fetches (20)
-// LIMIT stop after this many fetches (all)
-//
-// Resumable: an id whose media/ and wav48/ both exist is skipped, and
-// sources.json (in SONG_DATA) records each one as it lands.
-import { existsSync, linkSync, mkdirSync, readFileSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs";
-import { execFileSync } from "node:child_process";
-import path from "node:path";
-import { SONG_DATA } from "./paths.mjs";
-
-const argv = process.argv.slice(2);
-const flag = (name, dflt) => {
- const i = argv.indexOf(name);
- return i < 0 ? dflt : argv[i + 1];
-};
-const CHANNEL = flag("--channel", "the-quartering");
-const plans = argv.filter((a, i) => a.endsWith(".json") && argv[i - 1] !== "--channel");
-const TOKEN = process.env.WORKER_TOKEN;
-const EDITOR = (process.env.EDITOR_URL ?? "http://localhost:3001").replace(/\/$/, "");
-const PAUSE = Number(process.env.PAUSE ?? 20);
-const LIMIT = Number(process.env.LIMIT ?? Infinity);
-if (!plans.length || !TOKEN) {
- console.error("usage: WORKER_TOKEN=... fetch-sources.mjs <plan.json>... [--channel <slug>]");
- process.exit(2);
-}
-
-const ids = [...new Set(plans.flatMap((p) =>
- JSON.parse(readFileSync(p, "utf8")).voices.flatMap((v) => v.plan.map((n) => n.video)).filter(Boolean)))].sort();
-const MEDIA = path.join(SONG_DATA, "media"), WAV = path.join(SONG_DATA, "wav48");
-mkdirSync(MEDIA, { recursive: true });
-mkdirSync(WAV, { recursive: true });
-const LEDGER = path.join(SONG_DATA, "sources.json");
-const ledger = existsSync(LEDGER) ? JSON.parse(readFileSync(LEDGER, "utf8")) : {};
-const save = () => {
- writeFileSync(`${LEDGER}.tmp`, JSON.stringify(ledger, null, 1) + "\n");
- renameSync(`${LEDGER}.tmp`, LEDGER);
-};
-const sleep = (s) => new Promise((r) => setTimeout(r, s * 1000));
-const log = (msg) => console.log(`${new Date().toISOString().slice(11, 19)} ${msg}`);
-const headers = { authorization: `Bearer ${TOKEN}`, "content-type": "application/json" };
-
-async function fetchFull(id) {
- for (;;) {
- const res = await fetch(`${EDITOR}/api/media/fetch-window`, {
- method: "POST",
- headers,
- body: JSON.stringify({
- channelSlug: CHANNEL,
- videoId: id,
- full: true,
- requestedBy: "umtool song/fetch-sources.mjs",
- manifest: plans.map((p) => path.basename(p)).join(","),
- reason: "re-derive um-song source media (wav48/media) lost with a job temp dir",
- }),
- });
- const j = await res.json().catch(() => ({}));
- if (res.status === 409 && j.cooldownMs) {
- log(`${id}: ${j.platform} cooldown, waiting ${Math.ceil(j.cooldownMs / 1000)}s`);
- await sleep(j.cooldownMs / 1000 + 5);
- continue;
- }
- if (res.status === 507) throw Object.assign(new Error(`editor refused for disk space: ${j.error}`), { fatal: true });
- if (!res.ok) throw new Error(`HTTP ${res.status}: ${j.error ?? ""}`);
- if (j.cached) return j.file;
- for (;;) {
- await sleep(5);
- const p = await fetch(`${EDITOR}/api/media/fetch-window/${j.jobId}`, { headers });
- const s = await p.json().catch(() => ({}));
- if (!p.ok) throw new Error(`poll HTTP ${p.status}: ${s.error ?? ""}`);
- if (s.status === "done") {
- if (!s.file) throw new Error(`job ${j.jobId} done but no saved file`);
- return s.file;
- }
- if (s.status === "failed" || s.status === "cancelled") {
- throw new Error(`job ${j.jobId} ${s.status}: ${(s.error ?? "").split("\n").slice(-3).join(" | ")}`);
- }
- }
- }
-}
-
-const todo = ids.filter((id) => !(existsSync(path.join(MEDIA, `${id}.mp4`)) && existsSync(path.join(WAV, `${id}.wav`))));
-log(`${ids.length} sources in ${plans.length} plan(s); ${ids.length - todo.length} already here, ${todo.length} to fetch from ${CHANNEL}`);
-let fails = 0, done = 0;
-for (const id of todo) {
- if (done >= LIMIT) break;
- try {
- const t0 = Date.now();
- const file = await fetchFull(id);
- const media = path.join(MEDIA, `${id}.mp4`);
- if (existsSync(media)) unlinkSync(media);
- linkSync(file, media);
- const wav = path.join(WAV, `${id}.wav`);
- execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-max_error_rate", "1.0", "-i", media,
- "-vn", "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", `${wav}.part.wav`]);
- renameSync(`${wav}.part.wav`, wav);
- const secs = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration",
- "-of", "csv=p=0", wav]).toString().trim());
- ledger[id] = { channel: CHANNEL, saved: file, bytes: statSync(file).size, seconds: +secs.toFixed(2),
- fetchedAt: new Date().toISOString() };
- save();
- done += 1;
- fails = 0;
- log(`${id}: ${(statSync(file).size / 1e9).toFixed(2)} GB, ${(secs / 60).toFixed(1)} min, ` +
- `${Math.round((Date.now() - t0) / 1000)}s [${done}/${todo.length}]`);
- } catch (e) {
- log(`${id}: FAILED ${e.message}`);
- if (e.fatal || ++fails >= 3) {
- log(e.fatal ? "stopping" : "three failures in a row -- stopping (bot check?)");
- process.exit(1);
- }
- }
- await sleep(PAUSE);
-}
-log(`done: ${done} fetched`);
diff --git a/umtool/song/link-sources.mjs b/umtool/song/link-sources.mjs
@@ -0,0 +1,113 @@
+#!/usr/bin/env node
+// Point a song at its source recordings in the archive's saved-video store:
+// media/<id>.mp4 and wav48/<id>.wav, for every video the plans use.
+//
+// render-poly reads both by absolute source seconds: wav48/<id>.wav whole into
+// memory (a missing one aborts the render) and media/<id>.mp4 per clip (a
+// missing or short one silently drops or slides that voice's layer). So these
+// are FULL recordings, and they are fetched by the editor, not here:
+//
+// pnpm ops persist-videos --file <{"items": [{"slug","id"}, ...], "format": "video_720"}>
+//
+// one paced job on the channel's download queue, with its cookies and backoff.
+// This step only reads what that job saved, through each video's own
+// saved-video.json pointer -- the same way the editor finds it, so a store that
+// is moved to another drive later is followed, not copied.
+//
+// media/<id>.mp4 a SYMLINK to the saved file (no second copy; named .mp4
+// whatever the container, because render-poly asks for that
+// name and ffmpeg reads the content). An older real file
+// there is replaced.
+// wav48/<id>.wav 48 kHz mono s16 PCM, written once from the saved file.
+//
+// node link-sources.mjs <plan.json>... [--channel the-quartering]
+//
+// CHANNELS_DIR the corpus's channels/ (default: <repo>/transcripts/channels)
+//
+// Re-run as the persist job progresses: an id not saved yet is skipped and
+// counted, and one already linked is left alone. sources.json in SONG_DATA
+// records each one.
+import { existsSync, lstatSync, mkdirSync, readFileSync, readlinkSync, renameSync, statSync, symlinkSync, unlinkSync, writeFileSync } from "node:fs";
+import { execFileSync } from "node:child_process";
+import path from "node:path";
+import { SONG_DATA } from "./paths.mjs";
+
+const argv = process.argv.slice(2);
+const ci = argv.indexOf("--channel");
+const CHANNEL = ci < 0 ? "the-quartering" : argv[ci + 1];
+const plans = argv.filter((a, i) => a.endsWith(".json") && !(ci >= 0 && i === ci + 1));
+const DIR = path.dirname(new URL(import.meta.url).pathname);
+const CHANNELS = path.resolve(process.env.CHANNELS_DIR ?? path.join(DIR, "..", "..", "transcripts", "channels"));
+if (!plans.length) {
+ console.error("usage: link-sources.mjs <plan.json>... [--channel <slug>]");
+ process.exit(2);
+}
+if (!existsSync(path.join(CHANNELS, CHANNEL))) {
+ console.error(`no channel ${CHANNEL} under ${CHANNELS} -- set CHANNELS_DIR to the corpus's channels/`);
+ process.exit(2);
+}
+
+const ids = [...new Set(plans.flatMap((p) =>
+ JSON.parse(readFileSync(p, "utf8")).voices.flatMap((v) => v.plan.map((n) => n.video)).filter(Boolean)))].sort();
+const MEDIA = path.join(SONG_DATA, "media"), WAV = path.join(SONG_DATA, "wav48");
+mkdirSync(MEDIA, { recursive: true });
+mkdirSync(WAV, { recursive: true });
+const LEDGER = path.join(SONG_DATA, "sources.json");
+const ledger = existsSync(LEDGER) ? JSON.parse(readFileSync(LEDGER, "utf8")) : {};
+
+const savedFile = (id) => {
+ try {
+ const p = JSON.parse(readFileSync(path.join(CHANNELS, CHANNEL, "data", id, "saved-video.json"), "utf8"));
+ const f = path.join(p.dir, p.file);
+ return existsSync(f) ? f : null;
+ } catch {
+ return null;
+ }
+};
+const lexists = (p) => {
+ try {
+ lstatSync(p);
+ return true;
+ } catch {
+ return false;
+ }
+};
+
+let linked = 0, wavs = 0, waiting = 0, already = 0;
+for (const id of ids) {
+ const file = savedFile(id);
+ if (!file) {
+ waiting += 1;
+ continue;
+ }
+ const media = path.join(MEDIA, `${id}.mp4`);
+ const isLink = lexists(media) && lstatSync(media).isSymbolicLink();
+ let changed = false;
+ if (!isLink || path.resolve(MEDIA, readlinkSync(media)) !== file) {
+ if (lexists(media)) unlinkSync(media);
+ symlinkSync(file, media);
+ linked += 1;
+ changed = true;
+ }
+ const wav = path.join(WAV, `${id}.wav`);
+ if (!existsSync(wav)) {
+ execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-max_error_rate", "1.0", "-i", file,
+ "-vn", "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", `${wav}.part.wav`]);
+ renameSync(`${wav}.part.wav`, wav);
+ wavs += 1;
+ changed = true;
+ }
+ if (!changed) {
+ already += 1;
+ continue;
+ }
+ const secs = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration",
+ "-of", "csv=p=0", wav]).toString().trim());
+ ledger[id] = { channel: CHANNEL, saved: file, bytes: statSync(file).size, seconds: +secs.toFixed(2),
+ linkedAt: new Date().toISOString() };
+ writeFileSync(`${LEDGER}.tmp`, JSON.stringify(ledger, null, 1) + "\n");
+ renameSync(`${LEDGER}.tmp`, LEDGER);
+ console.log(`${id}: ${(secs / 60).toFixed(1)} min`);
+}
+console.log(`${ids.length} sources: ${linked} linked, ${wavs} wav48 written, ${already} already done, ${waiting} not saved yet`);
+