Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 07c587f904d9a7359fed4965e4a6b1102ba4dc3c
parent 0c0598528a81d7ded837b6544fa7267ea840a9b2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu,  8 Oct 2026 23:13:12 -0400

umtool song: link-sources replaces fetch-sources -- the editor's persist-videos fetches, this links

media/<id>.mp4 becomes a symlink to the saved file, found through the
video's own saved-video.json pointer, and wav48/<id>.wav is written once
from it. The fetch itself is pnpm ops persist-videos: one paced job on the
channel's download queue, with its cookies and backoff.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>

Diffstat:
Dumtool/song/fetch-sources.mjs | 139-------------------------------------------------------------------------------
Aumtool/song/link-sources.mjs | 113+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 113 insertions(+), 139 deletions(-)

diff --git a/umtool/song/fetch-sources.mjs b/umtool/song/fetch-sources.mjs @@ -1,139 +0,0 @@ -#!/usr/bin/env node -// Re-derive a song's SOURCE media -- media/<id>.mp4 and wav48/<id>.wav -- by -// asking the editor for each whole recording, one at a time. -// -// render-poly reads both by absolute source seconds: wav48/<id>.wav whole into -// memory (a missing one aborts the render) and media/<id>.mp4 per clip (a -// missing or short one silently drops or slides that voice's layer). So these -// are FULL recordings, not windows -- a window's t=0 is its `from`, and the plan -// would have to be rewritten to use it. -// -// Every fetch goes through the editor (POST /api/media/fetch-window, full: -// true): cookie policy, per-platform sleeps, the 429 cooldown, and the bytes -// land in the archive's saved-video store with a note of who asked and why. The -// song's copy is a HARDLINK to that file, not a second copy: the two trees share -// a filesystem, the store is the durable one, and each name survives the other -// being deleted. The link is named <id>.mp4 whatever the container is, because -// render-poly asks for that name; ffmpeg reads the content, not the extension. -// -// Slow and steady, never a burst: one job in flight, PAUSE seconds between -// them, a 409 cooldown waited out rather than retried. Three failures in a row -// stops the run -- that pattern is a bot check, and it wants a person. -// -// WORKER_TOKEN=... node fetch-sources.mjs <plan.json>... [--channel the-quartering] -// -// EDITOR_URL the editor (http://localhost:3001) -// PAUSE seconds between fetches (20) -// LIMIT stop after this many fetches (all) -// -// Resumable: an id whose media/ and wav48/ both exist is skipped, and -// sources.json (in SONG_DATA) records each one as it lands. -import { existsSync, linkSync, mkdirSync, readFileSync, renameSync, statSync, unlinkSync, writeFileSync } from "node:fs"; -import { execFileSync } from "node:child_process"; -import path from "node:path"; -import { SONG_DATA } from "./paths.mjs"; - -const argv = process.argv.slice(2); -const flag = (name, dflt) => { - const i = argv.indexOf(name); - return i < 0 ? dflt : argv[i + 1]; -}; -const CHANNEL = flag("--channel", "the-quartering"); -const plans = argv.filter((a, i) => a.endsWith(".json") && argv[i - 1] !== "--channel"); -const TOKEN = process.env.WORKER_TOKEN; -const EDITOR = (process.env.EDITOR_URL ?? "http://localhost:3001").replace(/\/$/, ""); -const PAUSE = Number(process.env.PAUSE ?? 20); -const LIMIT = Number(process.env.LIMIT ?? Infinity); -if (!plans.length || !TOKEN) { - console.error("usage: WORKER_TOKEN=... fetch-sources.mjs <plan.json>... [--channel <slug>]"); - process.exit(2); -} - -const ids = [...new Set(plans.flatMap((p) => - JSON.parse(readFileSync(p, "utf8")).voices.flatMap((v) => v.plan.map((n) => n.video)).filter(Boolean)))].sort(); -const MEDIA = path.join(SONG_DATA, "media"), WAV = path.join(SONG_DATA, "wav48"); -mkdirSync(MEDIA, { recursive: true }); -mkdirSync(WAV, { recursive: true }); -const LEDGER = path.join(SONG_DATA, "sources.json"); -const ledger = existsSync(LEDGER) ? JSON.parse(readFileSync(LEDGER, "utf8")) : {}; -const save = () => { - writeFileSync(`${LEDGER}.tmp`, JSON.stringify(ledger, null, 1) + "\n"); - renameSync(`${LEDGER}.tmp`, LEDGER); -}; -const sleep = (s) => new Promise((r) => setTimeout(r, s * 1000)); -const log = (msg) => console.log(`${new Date().toISOString().slice(11, 19)} ${msg}`); -const headers = { authorization: `Bearer ${TOKEN}`, "content-type": "application/json" }; - -async function fetchFull(id) { - for (;;) { - const res = await fetch(`${EDITOR}/api/media/fetch-window`, { - method: "POST", - headers, - body: JSON.stringify({ - channelSlug: CHANNEL, - videoId: id, - full: true, - requestedBy: "umtool song/fetch-sources.mjs", - manifest: plans.map((p) => path.basename(p)).join(","), - reason: "re-derive um-song source media (wav48/media) lost with a job temp dir", - }), - }); - const j = await res.json().catch(() => ({})); - if (res.status === 409 && j.cooldownMs) { - log(`${id}: ${j.platform} cooldown, waiting ${Math.ceil(j.cooldownMs / 1000)}s`); - await sleep(j.cooldownMs / 1000 + 5); - continue; - } - if (res.status === 507) throw Object.assign(new Error(`editor refused for disk space: ${j.error}`), { fatal: true }); - if (!res.ok) throw new Error(`HTTP ${res.status}: ${j.error ?? ""}`); - if (j.cached) return j.file; - for (;;) { - await sleep(5); - const p = await fetch(`${EDITOR}/api/media/fetch-window/${j.jobId}`, { headers }); - const s = await p.json().catch(() => ({})); - if (!p.ok) throw new Error(`poll HTTP ${p.status}: ${s.error ?? ""}`); - if (s.status === "done") { - if (!s.file) throw new Error(`job ${j.jobId} done but no saved file`); - return s.file; - } - if (s.status === "failed" || s.status === "cancelled") { - throw new Error(`job ${j.jobId} ${s.status}: ${(s.error ?? "").split("\n").slice(-3).join(" | ")}`); - } - } - } -} - -const todo = ids.filter((id) => !(existsSync(path.join(MEDIA, `${id}.mp4`)) && existsSync(path.join(WAV, `${id}.wav`)))); -log(`${ids.length} sources in ${plans.length} plan(s); ${ids.length - todo.length} already here, ${todo.length} to fetch from ${CHANNEL}`); -let fails = 0, done = 0; -for (const id of todo) { - if (done >= LIMIT) break; - try { - const t0 = Date.now(); - const file = await fetchFull(id); - const media = path.join(MEDIA, `${id}.mp4`); - if (existsSync(media)) unlinkSync(media); - linkSync(file, media); - const wav = path.join(WAV, `${id}.wav`); - execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-max_error_rate", "1.0", "-i", media, - "-vn", "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", `${wav}.part.wav`]); - renameSync(`${wav}.part.wav`, wav); - const secs = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration", - "-of", "csv=p=0", wav]).toString().trim()); - ledger[id] = { channel: CHANNEL, saved: file, bytes: statSync(file).size, seconds: +secs.toFixed(2), - fetchedAt: new Date().toISOString() }; - save(); - done += 1; - fails = 0; - log(`${id}: ${(statSync(file).size / 1e9).toFixed(2)} GB, ${(secs / 60).toFixed(1)} min, ` + - `${Math.round((Date.now() - t0) / 1000)}s [${done}/${todo.length}]`); - } catch (e) { - log(`${id}: FAILED ${e.message}`); - if (e.fatal || ++fails >= 3) { - log(e.fatal ? "stopping" : "three failures in a row -- stopping (bot check?)"); - process.exit(1); - } - } - await sleep(PAUSE); -} -log(`done: ${done} fetched`); diff --git a/umtool/song/link-sources.mjs b/umtool/song/link-sources.mjs @@ -0,0 +1,113 @@ +#!/usr/bin/env node +// Point a song at its source recordings in the archive's saved-video store: +// media/<id>.mp4 and wav48/<id>.wav, for every video the plans use. +// +// render-poly reads both by absolute source seconds: wav48/<id>.wav whole into +// memory (a missing one aborts the render) and media/<id>.mp4 per clip (a +// missing or short one silently drops or slides that voice's layer). So these +// are FULL recordings, and they are fetched by the editor, not here: +// +// pnpm ops persist-videos --file <{"items": [{"slug","id"}, ...], "format": "video_720"}> +// +// one paced job on the channel's download queue, with its cookies and backoff. +// This step only reads what that job saved, through each video's own +// saved-video.json pointer -- the same way the editor finds it, so a store that +// is moved to another drive later is followed, not copied. +// +// media/<id>.mp4 a SYMLINK to the saved file (no second copy; named .mp4 +// whatever the container, because render-poly asks for that +// name and ffmpeg reads the content). An older real file +// there is replaced. +// wav48/<id>.wav 48 kHz mono s16 PCM, written once from the saved file. +// +// node link-sources.mjs <plan.json>... [--channel the-quartering] +// +// CHANNELS_DIR the corpus's channels/ (default: <repo>/transcripts/channels) +// +// Re-run as the persist job progresses: an id not saved yet is skipped and +// counted, and one already linked is left alone. sources.json in SONG_DATA +// records each one. +import { existsSync, lstatSync, mkdirSync, readFileSync, readlinkSync, renameSync, statSync, symlinkSync, unlinkSync, writeFileSync } from "node:fs"; +import { execFileSync } from "node:child_process"; +import path from "node:path"; +import { SONG_DATA } from "./paths.mjs"; + +const argv = process.argv.slice(2); +const ci = argv.indexOf("--channel"); +const CHANNEL = ci < 0 ? "the-quartering" : argv[ci + 1]; +const plans = argv.filter((a, i) => a.endsWith(".json") && !(ci >= 0 && i === ci + 1)); +const DIR = path.dirname(new URL(import.meta.url).pathname); +const CHANNELS = path.resolve(process.env.CHANNELS_DIR ?? path.join(DIR, "..", "..", "transcripts", "channels")); +if (!plans.length) { + console.error("usage: link-sources.mjs <plan.json>... [--channel <slug>]"); + process.exit(2); +} +if (!existsSync(path.join(CHANNELS, CHANNEL))) { + console.error(`no channel ${CHANNEL} under ${CHANNELS} -- set CHANNELS_DIR to the corpus's channels/`); + process.exit(2); +} + +const ids = [...new Set(plans.flatMap((p) => + JSON.parse(readFileSync(p, "utf8")).voices.flatMap((v) => v.plan.map((n) => n.video)).filter(Boolean)))].sort(); +const MEDIA = path.join(SONG_DATA, "media"), WAV = path.join(SONG_DATA, "wav48"); +mkdirSync(MEDIA, { recursive: true }); +mkdirSync(WAV, { recursive: true }); +const LEDGER = path.join(SONG_DATA, "sources.json"); +const ledger = existsSync(LEDGER) ? JSON.parse(readFileSync(LEDGER, "utf8")) : {}; + +const savedFile = (id) => { + try { + const p = JSON.parse(readFileSync(path.join(CHANNELS, CHANNEL, "data", id, "saved-video.json"), "utf8")); + const f = path.join(p.dir, p.file); + return existsSync(f) ? f : null; + } catch { + return null; + } +}; +const lexists = (p) => { + try { + lstatSync(p); + return true; + } catch { + return false; + } +}; + +let linked = 0, wavs = 0, waiting = 0, already = 0; +for (const id of ids) { + const file = savedFile(id); + if (!file) { + waiting += 1; + continue; + } + const media = path.join(MEDIA, `${id}.mp4`); + const isLink = lexists(media) && lstatSync(media).isSymbolicLink(); + let changed = false; + if (!isLink || path.resolve(MEDIA, readlinkSync(media)) !== file) { + if (lexists(media)) unlinkSync(media); + symlinkSync(file, media); + linked += 1; + changed = true; + } + const wav = path.join(WAV, `${id}.wav`); + if (!existsSync(wav)) { + execFileSync("ffmpeg", ["-nostdin", "-v", "error", "-y", "-max_error_rate", "1.0", "-i", file, + "-vn", "-ac", "1", "-ar", "48000", "-c:a", "pcm_s16le", `${wav}.part.wav`]); + renameSync(`${wav}.part.wav`, wav); + wavs += 1; + changed = true; + } + if (!changed) { + already += 1; + continue; + } + const secs = Number(execFileSync("ffprobe", ["-v", "error", "-show_entries", "format=duration", + "-of", "csv=p=0", wav]).toString().trim()); + ledger[id] = { channel: CHANNEL, saved: file, bytes: statSync(file).size, seconds: +secs.toFixed(2), + linkedAt: new Date().toISOString() }; + writeFileSync(`${LEDGER}.tmp`, JSON.stringify(ledger, null, 1) + "\n"); + renameSync(`${LEDGER}.tmp`, LEDGER); + console.log(`${id}: ${(secs / 60).toFixed(1)} min`); +} +console.log(`${ids.length} sources: ${linked} linked, ${wavs} wav48 written, ${already} already done, ${waiting} not saved yet`); +