#!/usr/bin/env node // archilyzer-ops — drive a running editor over HTTP, without a browser. // // Every editor gesture used to be reachable only as a server action, which meant // an agent that wanted to sync a channel or fix a download filter had to drive // Playwright. /api/ops is a thin adapter layer over those same actions, and this // is its client. // // USAGE // // pnpm ops [--json '' | --file ] [--wait] // [--wait-timeout ] [--quiet] // pnpm ops get channel [--counts] // pnpm ops get channels // pnpm ops get tags [] // pnpm ops get job [--tail []] // pnpm ops get jobs [--active | --failed] [--kind ] [--slug ] [--limit ] // pnpm ops get remote-listing // pnpm ops get transcript [--slug ] // pnpm ops get coverage // pnpm ops job ... [--wait] // pnpm ops job retry-failed | job wait ... // pnpm ops list // // ARCHILYZER_EDITOR_URL editor base URL (default http://localhost:3001) // ARCHILYZER_AGENT who is asking, recorded as the provenance of a // curated-tag write (`agent:`, default // `agent:cli`). Ignored by every other action. // WORKER_TOKEN the shared secret the editor is running with. // Unset on the SERVER => every route 503s; unset here // => every route 401s. // // Either variable, when unset, is read from the editor's own env files — // `editor/.env.local`, then `editor/.env` — of THIS checkout (found from the // script's path, never the cwd), then of the main worktree when this is a // linked one (a worktree has no `editor/.env`; the editor it talks to by // default is the primary's). See loadEditorEnv. // // EXAMPLES // // pnpm ops sync --json '{"slug":"the-quartering"}' --wait // pnpm ops metadata-scan --json '{"slug":"the-quartering"}' // pnpm ops refresh-metadata --json '{"slug":"the-quartering","id":""}' --wait // pnpm ops feed-metadata --json '{"slug":"demo-podcast","dryRun":true}' --wait // pnpm ops import-video --json '{"slug":"demo-archive","url":"https://archive.org/details/example-item"}' // pnpm ops import-archive-org --json '{"slug":"demo-archive","item":"example-item","match":"\\.mp4$"}' --wait // pnpm ops import-archive-org --json '{"slug":"demo-archive","query":"collection:example-collection","dryRun":true}' --wait // pnpm ops get remote-listing demo-odysee // pnpm ops build-cues --json '{"slug":"demo-yt","ids":[""]}' --wait // pnpm ops get transcript --slug demo-yt // pnpm ops channel-config --json '{"slug":"x","patch":{"downloadFilterExclude":"rerun"}}' // pnpm ops channel-config --json '{"slug":"x","sites":[{"siteId":"anilyzer"}]}' // pnpm ops channel-config --json '{"slug":"x","sites":[],"excludeFromBuild":true}' // pnpm ops create-channel --json '{"fields":{"name":"Example (X)","handling":"transcribe","url":"https://x.com/example"}}' // pnpm ops rename-channel --json '{"slug":"old-slug","newSlug":"new-slug"}' // pnpm ops delete-channel --json '{"slug":"x","confirm":"x"}' // pnpm ops channel-priority --json '{"slugs":["x"],"operation":"download","tier":"paused"}' // pnpm ops lane --json '{"lane":"download","held":true}' // pnpm ops refresh-report --json '{"all":true}' // pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter"}' // pnpm ops relocate --json '{"slugs":["x"],"locationId":"platter","dryRun":true}' // pnpm ops settings --json '{"patch":{"minFreeDiskGB":20}}' // pnpm ops lane --json '{"lane":"publish","action":"drain"}' // pnpm ops cleanup --json '{"slug":"x","sweep":"transcribed"}' --wait // pnpm ops workers --json '{"op":"disable","ids":["parakeet-cpu"]}' // pnpm ops get job --tail 20 // pnpm ops get jobs --failed --slug x // pnpm ops build-site --json '{"siteId":"anilyzer"}' --wait // pnpm ops build-deploy --json '{"siteIds":["anilyzer","jeralyzer"]}' --wait // pnpm ops deploy-site --json '{"siteId":"anilyzer","preview":"tags-exclude"}' --wait // pnpm ops build-hub --wait // pnpm ops build-hub --json '{"deploy":true}' --wait // pnpm ops deploy-hub --wait // pnpm ops build-homepage --json '{"deploy":true}' --wait // pnpm ops deploy-homepage --json '{"preview":"refresh"}' --wait // pnpm ops reports-prepare --json '{"siteId":"demo-site"}' --wait // pnpm ops reports-export --json '{"siteId":"demo-site","formats":["html","md"]}' --wait // pnpm ops get channel the-quartering // pnpm ops get channels // pnpm ops tags --json '{"op":"define","tag":{"id":"eva-collab","label":"Collab"}}' // (a define is the WHOLE def: rules, and `sites` — the site ids the // tag exists on, absent = every site — included) // pnpm ops tag-videos --file ids.json // pnpm ops persist-videos --file list.json --wait // pnpm ops fetch-windows --json '{"siteId":"demo-site","dryRun":true}' // pnpm ops fetch-windows --file windows.json --wait // pnpm ops get tags eva-collab // pnpm ops cut-release --json '{"workspace":"all","version":"next","commit":true}' // pnpm ops transcribe --json '{"path":"/abs/clip.mp4","start":120,"end":150}' --wait // pnpm ops transcribe --json '{"path":"/abs/a.wav","workerId":"parakeet-cpu","out":"/tmp/a.json"}' --wait // // --file reads the BODY from a JSON file, which is how a big one gets sent: a // four-thousand-id tag-videos body is written by a script, not typed by a model // into a shell argument. --json and --file are mutually exclusive. // // --wait follows /api/jobs//log to the end for a job-starting action and // exits 0 only if the job finished `done`. Without it the command returns as // soon as the job is QUEUED, which is the honest answer: the queue may hold it // behind other work for hours. A poll that fails does NOT end the follow — see // followJob — and --wait-timeout is there for a caller that cannot // wait indefinitely. // // The response JSON is printed verbatim on stdout (log lines from --wait go to // stderr), so `pnpm ops … | jq` works. import { readFileSync } from "node:fs"; import { readFile } from "node:fs/promises"; import path from "node:path"; import { fileURLToPath, pathToFileURL } from "node:url"; const DEFAULT_URL = "http://localhost:3001"; // Consecutive polls where NEITHER the job's log NOR its ops record answered, // after which --wait gives up. At the 30 s backoff ceiling that is ~5 minutes // of an editor saying nothing at all, which is not a busy server — it is a // server that is gone. const MAX_PROBE_FAILURES = 10; // The read-side routes, reachable as `get `. Kept tiny and explicit: // an ops API that let a caller assemble arbitrary GET paths would be a proxy, // not an adapter. const GETTERS = { // --counts adds the LIVE on-disk counts, which walk every video directory — // opt-in for the same reason the route makes it opt-in. channel: (slug, { counts }) => `/api/ops/channel/${encodeURIComponent(slug)}${counts ? "?counts=1" : ""}`, // Every channel, one line each: slug, name, kind, platform and the sites // that carry it ([] = private to this editor). channels: () => "/api/ops/channels", // No argument: every definition with its pin/suppression counts. With one: a // single tag's assignments, each carrying the provenance of the pin. tags: (tag) => tag ? `/api/ops/tags?tag=${encodeURIComponent(tag)}` : "/api/ops/tags", // The publish status (release 18): the index, the lane, a row per site, the // hub and the homepage with their chips, and the plan Publish now would run. publish: () => "/api/ops/publish", // One job: its record or sidecar, where it waits, and with --tail [N] the // last N lines of its log (default 40). job: (id, { tail }) => `/api/ops/job/${encodeURIComponent(id)}${tail ? `?tail=${tail}` : ""}`, // The /jobs list: --active (the live head, in queue order), --failed, // --kind , --slug , --limit . jobs: (_arg, { jobs }) => { const q = new URLSearchParams(); if (jobs.active) q.set("active", "1"); if (jobs.failed) q.set("failed", "1"); if (jobs.kind) q.set("kind", jobs.kind); if (jobs.slug) q.set("slug", jobs.slug); if (jobs.limit) q.set("limit", String(jobs.limit)); const s = q.toString(); return `/api/ops/jobs${s ? `?${s}` : ""}`; }, // THE READ SIDE (release 19, A3): what each page draws, behind the token. // settings.json as the editor reads it (migrated, defaulted, secrets // redacted), or one top-level block of it. settings: (key) => key ? `/api/ops/settings?key=${encodeURIComponent(key)}` : "/api/ops/settings", // The /storage page: each location, mounted or not, its free space and tiers. storage: () => "/api/ops/storage", // Every site: id, title, siteUrl, Pages project, audience, policy, channels. sites: () => "/api/ops/sites", // The transcription workers and what each is running. workers: () => "/api/ops/workers", // The four lanes: runner, policy, cooldowns, picks, pending. "auto-queue": () => "/api/ops/auto-queue", // The sync scheduler: each channel's schedule, the tick log. scheduler: () => "/api/ops/scheduler", // One channel's cleanup row: what each sweep would reclaim, what holds the rest. cleanup: (slug) => `/api/ops/cleanup/${encodeURIComponent(slug)}`, // What a channel holds, by date, and its gaps (release 19 A9): each video // dated by its recorded date when the channel has a title rule, else upload. coverage: (slug) => `/api/ops/coverage?slug=${encodeURIComponent(slug)}`, // One video's cues off disk, with no index (release 19 A7): a fresh // cues.json, else what normalize would write, else the VTT alone. --slug // names the channel (else the one holding data//). transcript: (id, { jobs }) => { const q = new URLSearchParams({ id }); if (jobs.slug) q.set("slug", jobs.slug); return `/api/ops/transcript?${q.toString()}`; }, }; // READS THAT ARE JOBS: the answer needs a request upstream, so the editor // runs it on the platform's queue and the job's log carries the result. `get` // POSTs the action, waits, and prints the result on stdout (--wait-timeout // bounds the wait). const GET_JOBS = { // An Odysee or BitChute channel's listing, diffed against what it holds // (release 19 A6): {notHeld: [{id, url}], heldNotListed: [id], …}. "remote-listing": (slug) => ({ path: "/api/ops/remote-listing", body: { slug }, resultMarker: REMOTE_LISTING_RESULT_MARKER, }), }; // Nouns whose read takes no argument. `get channel` without a slug is a // mistake; `get tags` without one is the whole vocabulary. const GET_ARG_OPTIONAL = new Set([ "tags", "channels", "publish", "jobs", "settings", "storage", "sites", "workers", "auto-queue", "scheduler", ]); // The flags each noun takes beyond the shared ones; any other is refused. const GET_FLAGS = { transcript: ["slug"], channel: ["counts"], job: ["tail"], jobs: ["active", "failed", "kind", "slug", "limit"], }; // The row buttons of /jobs (`POST /api/ops/job`), as `job `. // `retry-failed` takes no ids; `wait` is not a request at all — it follows the // ids as --wait would. export const JOB_VERBS = [ "cancel", "drain", "promote", "force-release", "retry", "retry-failed", "wait", ]; const ACTIONS = [ "channel-priority", "channel-config", // The channel's lifecycle: the New channel form, and the channel page's // Danger → Rename and Danger → Delete. Synchronous; no job. "create-channel", "rename-channel", "delete-channel", "metadata-scan", // ONE video's metadata.info.json re-read from its source ({slug, id}): no // subtitles, no media; the rewrite lands in metadata.history.json. "refresh-metadata", "import-video", // archive.org media into a channel: one item ({slug, item, files: [...] | // match: ""}), many ({items: [...]}) or a search ({query, limit?}); // dryRun? lists held / RESTRICTED / would-get. One job, one file at a time, // paced, a pause between items. "import-archive-org", // An Odysee or BitChute channel's listing diffed against what it holds // ({slug}); `get remote-listing ` is the same, waited on. "remote-listing", // A channel's held videos given their media from a LOCAL archive — a // directory, a zip read in place, a 7z ({slug, source, items?, match?, // createRecords?, replace?, dryRun?}): one job, nothing fetched. "attach-media", // A channel's saved containers remuxed losslessly into browser-playable // copies, one torrent each ({slug, ids?, trackers?, root?, dryRun?}). "prepare-playable", // Write transcript.cues.json from each video's raw transcript ({slug, ids?, // force?}) — what a reader with no index build wants. "build-cues", // A podcast channel's records completed from its RSS feed ({slug, dryRun?}): // one fetch of the feed, no media. "feed-metadata", "refresh-report", "sync", "download-missing", "retry-bucket", // The TRANSCRIBE half of a channel's "downloaded, not transcribed" bucket // (retry-bucket is the download retry and skips audio already on disk). "transcribe-bucket", // A social channel's post fetch: new posts, the full re-walk ("full"), or // the walk back below the oldest archived post ("older"). "fetch-posts", // A screenshot and the attached media of specific archived posts. "capture-posts", // Publishing as stages (release 18): {"verb": "index" | "build" | "deploy" | // "hub" | "homepage" | "now" | "stale", …} — one run of stages on the // editor's publish queue. The eight rows below are its aliases. "publish", "build-index", "build-deploy", "build-site", // Deploy the site's ALREADY-BUILT bundle — build-deploy's other half, and // the one a preview is for: build once, look at the preview, then ship the // same bundle to production without rebuilding it. "deploy-site", // The HUB (the export app in hub mode, into its own bundle) and its deploy // to homepage.json's Pages project. "build-hub", "deploy-hub", // The HOMEPAGE (the `homepage` package, Archilyzer's own site, into // homepage/out) and its deploy to the constant Pages project `archilyzer`. "build-homepage", "deploy-homepage", "relocate", "relocate-back", "evict-clips", // A report site's evidence media: cut every cited clip and copy every cited // post capture into the site's report-media cache ({siteId}). "reports-prepare", // A report site's exports: each published report as HTML, PDF, Markdown and // an evidence pack ({siteId, reportId?, formats?}). "reports-export", "lane", // The curated-tag writers. `tags` edits the vocabulary (define/remove); // `tag-videos` pins, unpins, suppresses or unsuppresses one tag over a batch // of videos in ONE write. "tags", "tag-videos", // Set the per-video do-not-clean marker on every video of a channel whose // title/description matches a download-filter pattern ({slug, match}). "keep-videos", // Persist specific videos, across channels, to the saved-video store // ({items: [{slug, id}]}), paced and gated; a re-run resumes. "persist-videos", // Fetch clip windows through the managed path, one paced job per platform // queue ({siteId} = a site's missing evidence, or {items, requestedBy}). "fetch-windows", // Cut a changelog's [Unreleased] into a dated release heading (release 10 // slice P). Synchronous. The same writer as `archilyzer release cut`, which // needs no editor at all — this route exists only on an editor built from // release 10 or later. "cut-release", // ONE local file, or a window of it, through a local transcription worker // ({path, start?, end?, workerId?, out?}) — the corpus's own engine and // model, as a job. With --wait the result JSON is what stdout carries. "transcribe", // ARCHIVAL WRITES (release 19, A4). settings.json through the editor's one // writer ({patch}); a platform's rate-limit hold cleared ({platform}); // workers switched on, off or drained ({op, ids}); one video transcribed // ({slug, id, file?}), one of its files deleted ({slug, id, file}), its // do-not-clean marker set ({slug, id, keep?}); a channel's cleanup sweep // ({slug, sweep}). "settings", "clear-platform-hold", "workers", "transcribe-one", "delete-file", "do-not-clean", "cleanup", ]; // The log line a transcribe job ends with: this marker, then the result as // compact JSON. The same string is TRANSCRIBE_RESULT_MARKER in // common/controller/transcribeFile.ts. export const TRANSCRIBE_RESULT_MARKER = "@@transcribe-result "; // The same for a remote listing (REMOTE_LISTING_RESULT_MARKER in // common/controller/remoteListing.ts). export const REMOTE_LISTING_RESULT_MARKER = "@@remote-listing "; // The provenance a tag write from this CLI carries. Everything else ignores it. function agentSource() { return `agent:${process.env.ARCHILYZER_AGENT || "cli"}`; } export function parseArgs(argv) { const positional = []; let json = null; let file = null; let wait = false; let quiet = false; let counts = false; let tail = 0; const jobs = {}; // The noun-specific flags seen, checked against GET_FLAGS once the noun is // known: a flag a read does not take is refused, never ignored. const used = new Set(); let waitTimeout = null; const readTimeout = (raw) => { const n = Number(raw); if (!Number.isFinite(n) || n <= 0) { return { error: "--wait-timeout needs a positive number of seconds" }; } waitTimeout = n; // A timeout on a wait nobody asked for is not a preference, it is a typo // with no effect — so it IMPLIES --wait rather than being ignored. wait = true; return null; }; for (let i = 0; i < argv.length; i++) { const arg = argv[i]; if (arg === "--wait") { wait = true; } else if (arg === "--wait-timeout") { const raw = argv[++i]; if (raw === undefined) { return { error: "--wait-timeout needs a number of seconds" }; } const err = readTimeout(raw); if (err) return err; } else if (arg.startsWith("--wait-timeout=")) { const err = readTimeout(arg.slice("--wait-timeout=".length)); if (err) return err; } else if (arg === "--quiet") { quiet = true; } else if (arg === "--counts") { counts = true; used.add("counts"); } else if (arg === "--active" || arg === "--failed") { jobs[arg.slice(2)] = true; used.add(arg.slice(2)); } else if (/^--(kind|slug|limit)(=|$)/.test(arg)) { const [, key, eq] = /^--(kind|slug|limit)(=?)/.exec(arg); const raw = eq ? arg.slice(key.length + 3) : argv[++i]; if (raw === undefined || raw === "") { return { error: `--${key} needs a value` }; } if (key === "limit") { const n = Number(raw); if (!Number.isInteger(n) || n <= 0) { return { error: "--limit needs a whole number above zero" }; } jobs.limit = n; } else { jobs[key] = raw; } used.add(key); } else if (arg === "--tail" || arg.startsWith("--tail=")) { // An optional count: `--tail` alone is the last 40 lines. const raw = arg.startsWith("--tail=") ? arg.slice("--tail=".length) : /^\d+$/.test(argv[i + 1] ?? "") ? argv[++i] : "40"; const n = Number(raw); if (!Number.isInteger(n) || n <= 0) { return { error: "--tail takes a whole number of lines above zero" }; } tail = n; used.add("tail"); } else if (arg === "--json") { json = argv[++i]; if (json === undefined) { return { error: "--json needs a JSON object argument" }; } } else if (arg.startsWith("--json=")) { json = arg.slice("--json=".length); } else if (arg === "--file") { file = argv[++i]; if (file === undefined) { return { error: "--file needs a path to a JSON file" }; } } else if (arg.startsWith("--file=")) { file = arg.slice("--file=".length); } else if (arg === "--help" || arg === "-h") { return { help: true }; } else if (arg.startsWith("-")) { return { error: `unknown flag: ${arg}` }; } else { positional.push(arg); } } if (positional.length === 0) return { help: true }; if (json !== null && file !== null) { return { error: "--json and --file are mutually exclusive" }; } let body = {}; if (json !== null) { try { body = JSON.parse(json); } catch (e) { return { error: `--json is not valid JSON: ${e.message}` }; } if (typeof body !== "object" || body === null || Array.isArray(body)) { return { error: "--json must be a JSON object" }; } } if (positional[0] === "list") { return { list: true }; } if (positional[0] === "get") { const noun = positional[1]; if (noun && GET_JOBS[noun]) { if (!positional[2]) return { error: `get ${noun}: needs an argument` }; if (used.size) { return { error: `get ${noun} does not take ${[...used].map((f) => `--${f}`).join(", ")}` }; } const job = GET_JOBS[noun](positional[2]); return { method: "POST", path: job.path, body: job.body, resultMarker: job.resultMarker, wait: true, quiet, waitTimeout, }; } if (!noun || !GETTERS[noun]) { return { error: `get: unknown noun "${noun ?? ""}" — known: ${[...Object.keys(GETTERS), ...Object.keys(GET_JOBS)].join(", ")}`, }; } if (!positional[2] && !GET_ARG_OPTIONAL.has(noun)) { return { error: `get ${noun}: needs an argument` }; } const allowed = GET_FLAGS[noun] ?? []; const stray = [...used].filter((f) => !allowed.includes(f)); if (stray.length) { return { error: `get ${noun} does not take ${stray.map((f) => `--${f}`).join(", ")}`, }; } return { method: "GET", path: GETTERS[noun](positional[2], { counts, tail, jobs }), wait: false, quiet, waitTimeout, }; } if (used.size) { return { error: `${[...used].map((f) => `--${f}`).join(", ")} belongs to a read (get …), not to "${positional[0]}"`, }; } if (positional[0] === "job") { const verb = positional[1]; if (!JOB_VERBS.includes(verb)) { return { error: `job: unknown verb "${verb ?? ""}" — known: ${JOB_VERBS.join(", ")}`, }; } const ids = positional.slice(2); if (verb === "retry-failed") { if (ids.length) return { error: "job retry-failed takes no ids" }; } else if (ids.length === 0) { return { error: `job ${verb}: needs one or more job ids` }; } if (json !== null || file !== null) { return { error: `job ${verb} takes its ids as arguments, not a body` }; } if (verb === "wait") { // Nothing to send: follow each id to its end, as --wait would. return { waitFor: ids, wait: true, quiet, waitTimeout }; } return { method: "POST", path: "/api/ops/job", body: verb === "retry-failed" ? { verb } : { verb, ids }, wait, quiet, waitTimeout, }; } const action = positional[0]; if (!ACTIONS.includes(action)) { return { error: `unknown action "${action}" — known: ${ACTIONS.join(", ")}`, }; } if (positional.length > 1) { return { error: `"${action}" takes no positional arguments — pass its body with --json`, }; } // The body may still arrive from --file; main() reads it, because parseArgs // is pure (and unit-tested without a filesystem). return { method: "POST", path: `/api/ops/${action}`, body, ...(file !== null ? { bodyFile: file } : {}), // WHO IS ASKING, for the one pair of actions that records it. Defaulted // here rather than in the route so a human at a terminal and a script under // a name are told apart; the route's own default (`agent:ops`) covers a // caller that is neither. ...(action === "tag-videos" ? { defaultSource: agentSource() } : {}), // A job whose log carries a RESULT, which --wait prints on stdout. ...(action === "transcribe" ? { resultMarker: TRANSCRIBE_RESULT_MARKER } : {}), ...(action === "remote-listing" ? { resultMarker: REMOTE_LISTING_RESULT_MARKER } : {}), wait, quiet, waitTimeout, }; } // The preview alias(es) a response carries, in job order and de-duplicated. // // WHY IT IS REPRINTED AT ALL. The alias is already in the JSON, but the JSON is // what a pipe consumes and the URL is what a person needs, and with --wait it // scrolls off behind minutes of build log. Printed on stderr, not stdout, so // `pnpm ops … | jq` still sees nothing but the response — the same split the // --wait log lines already use. export function previewUrlsIn(payload) { const urls = []; if (payload && Array.isArray(payload.jobs)) { for (const job of payload.jobs) { if (job && typeof job.previewUrl === "string") urls.push(job.previewUrl); } } // Only as a fallback: the single-job case repeats jobs[0].previewUrl at the // top level, and printing it twice would read as two previews. if (urls.length === 0 && payload && typeof payload.previewUrl === "string") { urls.push(payload.previewUrl); } return [...new Set(urls)]; } function printPreviewUrls(payload) { for (const url of previewUrlsIn(payload)) console.error(`preview: ${url}`); } // Pull a job's RESULT line out of its log as the log streams by. `feed` takes // each chunk and returns the text to echo — every complete line except the // result line; `finish` returns what is left and the parsed result (null when // the log never carried one). Line-buffered, because a poll may end mid-line. export function makeResultCapture(marker) { let pending = ""; let result = null; const take = (line) => { if (line.startsWith(marker)) { try { result = JSON.parse(line.slice(marker.length)); return ""; } catch { return `${line}\n`; } } return `${line}\n`; }; return { feed(chunk) { pending += chunk; const lines = pending.split("\n"); pending = lines.pop() ?? ""; return lines.map(take).join(""); }, finish() { const echo = pending ? take(pending).replace(/\n$/, "") : ""; pending = ""; return { echo, result }; }, }; } export function usage() { return [ "Usage: pnpm ops [--json '' | --file ] [--wait]", " [--wait-timeout ] [--quiet]", " pnpm ops get channel [--counts]", " pnpm ops get channels", " pnpm ops get tags []", " pnpm ops get publish", " pnpm ops get job [--tail []]", " pnpm ops get jobs [--active | --failed] [--kind ] [--slug ]", " [--limit ]", " pnpm ops job cancel|drain|promote|force-release|retry ... [--wait]", " pnpm ops job retry-failed [--wait]", " pnpm ops job wait ... [--wait-timeout ]", " pnpm ops get settings []", " pnpm ops get storage | sites | workers | auto-queue | scheduler", " pnpm ops get cleanup ", " pnpm ops get remote-listing [--wait-timeout ]", " pnpm ops get transcript [--slug ]", " pnpm ops get coverage ", " pnpm ops list", "", `Actions: ${ACTIONS.join(", ")}`, "", "--wait follows the job's log and survives a poll that fails (a busy", " in-process build starves the server): it backs off and, after three", " failures, asks /api/ops/job/ whether the job is still there and how", " it ended. While the job waits it prints its queue position on stderr.", "--wait-timeout gives up and exits 1 instead of waiting forever.", " Default: no timeout — the queue may legitimately hold a job for hours.", "", "get job is one job: kind, channel, status, times, exit code, and while", " it waits its queue (key, position, how many wait, the job at the head);", " --tail [] adds the last lines of its log (40, at most 500).", "get jobs lists jobs newest first (50; --limit, at most 500): --active is", " every queued and running job in queue order, --failed the failed ones;", " --kind and --slug narrow either. A filtered list looks through the", " newest 2000 jobs.", "job ... is a /jobs row's button: cancel (a running job is", " stopped), drain (finish the sub-operations in flight, start no more),", " promote (a queued job to the front of its queue), force-release (free a", " wedged queue slot), retry (re-run from its replay descriptor, ahead of", " the queue). retry-failed retries every failed job the editor still holds.", " Each id is answered in \"results\"; one that could not be acted on makes", " the answer ok: false without stopping the rest. --wait follows the jobs a", " retry started. job wait ... follows jobs already running.", "", "get settings [] is settings.json as the editor reads it (migrated and", " defaulted; a secret is \"\"), or one top-level key of it.", "get storage is /storage: each location, mounted or not, free space, tiers.", "get sites lists every site: id, title, siteUrl, Pages project, audience,", " listed, search, publish policy, channels. (`get publish` is the status.)", "get workers, get auto-queue and get scheduler are /workers, the four lanes", " of /operations, and /operations/sync — the payloads those pages poll.", "get cleanup is one channel's /cleanup row: what each sweep would", " reclaim (they overlap — never add them), what holds the rest, and the", " failed-transcriptions count. measured: false means unknown, not zero.", "", 'remote-listing lists an Odysee or BitChute channel upstream and diffs it', ' against what it holds: {"slug"}. One flat-playlist read on the platform\'s', " queue (refused while it is held or cooling down; a 429 backs it off),", " nothing written. `get remote-listing ` waits for the job and prints", " {listed, held, notHeld: [{id, url}], heldNotListed: [id], ...} on stdout.", "", 'build-cues writes each video\'s transcript.cues.json from its raw', " transcript (the caption-track rule for VTTs), as the digest card's", ' Normalize button does: {"slug"}, "ids": [...] for those videos only', ' (every one held), "force": true to rewrite a fresh one. A job on the', " channel's queue. The file every reader without an index build wants.", "", "get coverage is what a channel holds by date: held, dated (by", " recorded date — the channel's recordedDate title rule — else by upload", " date), undated, first and last day, per year and month, and every gap", " over 30 days with nothing held. Off disk; nothing written. (The MCP's", " channel_coverage takes gap days, a date window and a video list.)", "", "get transcript [--slug ] reads one video's cues off disk", " with no index: a fresh cues.json, else what build-cues would write, else", " the English VTT alone — {source, cuesJson, title?, ..., cues}. Nothing is", " written. Without --slug, the channel holding data// is found.", "", 'publish runs publish stages on the editor\'s publish queue, one at a time,', ' under one run id: {"verb": …}. "index" updates the index; "build" builds', ' "siteId"/"siteIds" (forced; the index first when stale; "runner":', ' "docker" builds every site in containers; "allowMissingMedia": true lets', ' a report citation with no prepared media through compose, local builds', ' only); "deploy" ships their built', ' bundles (production, "preview": "", or "to": "local"; "force"', ' redeploys a bundle already shipped there); "hub" / "homepage" build', ' them, {"deploy": true} deploys after; "now" is Publish now (the stale', ' index, then each policy target); "stale" builds every stale site. The', ' answer lists every job ({target, kind, jobId}) and --wait follows them', ' all. A site never built is refused: "no build of in —', ' archilyzer publish build ". `get publish` is the status.', "", 'build-index, build-site, build-deploy, deploy-site, build-hub, deploy-hub,', ' build-homepage and deploy-homepage are publish\'s aliases, with their old', ' bodies and answers ("skipData" is accepted and ignored).', "", 'build-site, build-deploy and deploy-site all take "siteId" (one) or', ' "siteIds" (a list).', "", 'reports-prepare cuts every clip and copies every post capture a site\'s', ' published reports cite into its report-media cache, before its build:', ' {"siteId"}. The job fails, naming each one, when a citation lacks media.', ' When nothing is missing it then exports the reports, as reports-export.', "", 'reports-export writes each published report as report.html, report.pdf,', ' report.md, slides.html, slides.pdf and evidence-pack.zip for the site\'s', ' build to publish: {"siteId", "reportId"?, "formats"?: ["html","pdf","md",', ' "slides","slides-pdf","zip"]}.', "", 'settings patches settings.json through the editor\'s one writer:', ' {"patch": {"": value, …}} (SETTINGS.md). A block\'s keys', ' merge one level; an array, a scalar or an object nested in a block', ' replaces. Refused before anything is written: an unknown key; a key', ' another writer owns (channelPriority, autoQueue, workers, storage — the', ' refusal names it); a value the schema would not keep as sent (clamped,', ' dropped), named with what would be saved. Answers the saved values.', "", 'lane takes {"lane": "transcription"|"download"|"digest"|"backfill"|', ' "publish", "enabled"?, "held"?, "action"?: "start"|"stop"|"drain"}.', "", 'clear-platform-hold clears a platform\'s rate-limit hold, backoff and raised', ' pace, as the lane strip\'s Clear hold: {"platform"}. Answers the sentence.', "", 'workers switches transcription workers as /workers does: {"op": "enable" |', ' "disable" | "drain", "ids": [...]} — the live pool; a restart restores', ' the launch default. An unknown id is named; the others still apply.', "", 'transcribe-one transcribes one video: {"slug", "id", "file"?}. With "file"', ' (an audio file in its dir) that file; without, the video page\'s', ' Transcribe — the audio on disk, or downloaded first through the channel\'s', ' managed path. A job; --wait follows it.', "", 'delete-file deletes one file of one video, as its Files list does:', ' {"slug", "id", "file"}. A tiered media file goes with its bytes on the', ' media drive (never a bare link); refused while that drive is not', ' answering. No undo.', "", 'do-not-clean sets one video\'s "Do not clean" marker: {"slug", "id",', ' "keep"?: true}; "keep": false removes it. (keep-videos is the bulk form.)', "", 'cleanup runs one of a channel\'s cleanup sweeps as a job: {"slug", "sweep":', ' "transcribed" (audio of transcribed videos — the /cleanup total) |', ' "extra-formats" | "wrong-format" | "failed-transcriptions" (clears the', ' list; deletes nothing)}. Do-not-clean videos are skipped; `get cleanup', ' ` says what each would reclaim.', "", 'relocate moves channels\' media to a location: {"slugs", "locationId" |', ' "root"}. "dryRun": true answers each channel\'s preview (bytes to copy,', ' free space both sides) and moves nothing — though, as on the Storage', ' panel, a channel never tiered is first tiered in place on the corpus disk.', "", 'channel-priority sets channels\' priority, as the /channels deck\'s tier', ' control does: {"slugs", "tier": "normal" | "low" | "paused"} sets the', ' base tier; with "operation" (sync, transcription, download, digest,', ' backfill) it pins that operation\'s tier, "tier": null clearing it back', ' to the base; "preset": "sync-only" | "clear" is the deck\'s two shortcuts', ' and takes no tier. A manual change clears an automatic pause.', "", 'sync lists a channel and downloads what is new, as its Sync button does:', ' {"slug"}. "full": true runs the periodic whole-listing sweep now (else', ' the configured cadence decides); "queueKey" picks another queue. On the', " channel's platform queue, paced like its downloads.", "", 'download-missing downloads every listed video the channel does not hold:', ' {"slug"}. "ignoreArchive": true drops the download archive so listed', ' videos it names are fetched again (recovery from a stale archive);', ' "abortOnError": false carries on past a failure (default: stop at the', ' first that is not one video\'s own). Paced on the platform\'s queue.', "", 'metadata-scan reads every listed video\'s title, date and description', ' into the channel\'s scan file, fetching no media: {"slug"}. Not held by', " the download pause and needs no disk floor; one request per video, so", " it respects the platform's cooldown and records one when pushed back.", "", 'import-video imports one video by URL into a channel: {"slug", "url"}. An', " archive.org URL becomes its canonical item or file (an item of several", " media files is refused — see import-archive-org); a BitChute or Odysee", " URL runs on that platform's queue at its pace, refused while it is held", " or cooling down, and refused when already on disk; a Wayback capture is", " named by what it copies.", "", 'feed-metadata completes a podcast channel\'s records from its RSS feed: one', ' fetch of the channel\'s url, then title, date, description and duration', ' into each record that lacks them: {"slug"}. "dryRun": true counts', " matched, unmatched and already complete, and writes nothing.", "", 'refresh-report regenerates a channel\'s report (the buckets and counts its', ' page shows) on the one serial report queue: {"slug"} answers the job', ' queued, or the one already waiting ("started": false); {"all": true}', " queues every channel and answers {queued, jobIds, skipped}.", "", 'relocate-back moves channels\' media back onto the corpus disk, one job', ' per channel on the relocation queue: {"slugs"}. A busy channel is', ' answered in "skipped"; --wait follows the jobs started.', "", 'evict-clips deletes cached clip windows (data//clips/) older than a', ' number of days, as /storage\'s button does: {"olderThanDays", "slug"?,', ' "dryRun"?}. BY AGE: nothing knows whether a umtool report still cites a', " window; an evicted window is fetched again when asked for.", "", 'tags edits the curated-tag vocabulary through its one writer: {"op":', ' "define", "tag": {...}} (the WHOLE definition: id, label, rules and', ' "sites", the sites it exists on, absent = every site) or {"op":', ' "remove", "tag": ""}. `get tags []` reads it.', "", 'tag-videos pins, unpins, suppresses or unsuppresses one tag on many videos', ' in ONE write: {"tag", "op": "add" | "remove" | "suppress" | "unsuppress",', ' "videos": [{"slug", "id"}, ...]}. "remove" unpins only — a rule\'s hit', ' survives; "suppress" rejects it. The write records who asked', ' (ARCHILYZER_AGENT); a big list goes in --file.', "", 'keep-videos sets the "Do not clean" marker on every held video of a', ' channel whose title or description matches a download-filter pattern:', ' {"slug", "match", "fields"?: ["title" | "description"], "note"?,', ' "dryRun"?}. A match not held is reported in "notDownloaded", never', " created; a fresh channel wants metadata-scan first.", "", 'channel-config changes a channel as its Configure form does: {"slug"} and', ' any of "patch" (form field names; "" clears one), "sites" (the WHOLE', ' membership set: [{"siteId", "groupId"? | "newGroupName"?}], [] = on no', ' site; an unknown site id is refused), "excludeFromBuild" and', ' "excludeFromCleanup" (set to the value given, not toggled). A VOD mirror\'s', ' recorded date: "patch": {"recordedDateTitlePattern": ""} (CHANNEL.md, recordedDate).', "", 'create-channel is the New channel form: {"fields": {"name", "handling":', ' "youtube"|"transcribe", "url"?, "platform"?, "sourceKind"?, "postFetcher"?,', ' "socialHandle"?, …}} with channel-config\'s patch keys; "slug"? (else', ' derived from the name), "sites"? (absent = on no site). "fetchPlaylist",', ' "fetchPostsNow" and "prioritizeDownload" are the form\'s checkboxes, OFF', ' unless true; a job they start comes back as jobId(s), so --wait follows it.', "", 'rename-channel moves a channel to a new slug, as Danger → Rename does:', ' {"slug", "newSlug"}. Refused while the channel is busy (a job, a lane unit,', ' media in transition) or when the new slug is taken. Old links break.', "", 'delete-channel removes a channel\'s whole directory, as Danger → Delete does:', ' {"slug", "confirm"} — "confirm" must repeat the slug. No undo outside the', ' transcripts/ repo\'s own history.', "", 'refresh-metadata re-reads ONE video\'s metadata.info.json from its source', ' (no subtitles, no media) on the platform\'s queue: {"slug", "id"}. The', ' job\'s log ends with what the source now says — live_status, formats,', ' audio-only formats and whether any is non-fragmented, English captions,', ' the keys that changed. An id with no data// is refused (a refresh', ' re-reads a video already archived), as are archive.org and Wayback records.', "", 'retry-bucket runs one bucket of a channel\'s report as one job, past any', ' lane hold: {"slug", "bucket"}. "ids": [...] runs only those videos, and', " every one must be in the bucket (a stray id is refused, named); a job run", " with ids is not replayable, as a checkbox selection in the UI is not.", "", 'transcribe-bucket transcribes a channel\'s "downloaded, not transcribed"', ' bucket on the transcription queue, as the channel page\'s Transcribe button', ' does: {"slug"}. "ids": [...] narrows it the same way as on retry-bucket.', "", 'fetch-posts fetches a social channel\'s new posts: {"slug"}. "full": true', ' re-walks the whole timeline; "older": true walks back from the oldest', " archived post through search (X; needs a login), saving its place for", ' the next run, down to "floor": "YYYY-MM-DD" when given; "from":', ' "YYYY-MM-DD" starts the walk afresh there, replacing its saved place', ' (and a "complete") — for a gap above one surviving old post. "limit": N caps', ' the posts one run reads; "pages": N caps the pages (a forum thread: its', ' latest N pages). "full" and "older" together are refused. An', ' older walk over an account that shows no posts (nothing archived, and', ' the last timeline fetch read none) is refused unless "force": true. A', " drained fetch stops at its next resume point and the next run resumes.", "", 'capture-posts captures archived posts of a social channel (X, forum): a', ' screenshot of each through the connected X profile (a forum thread: its', ' host\'s forum profile), and its attached media through gallery-dl (a', ' forum thread: the same profile), into the channel\'s posts-media//:', ' {"slug", "ids": [...]}. Every id must be in the channel\'s posts archive.', ' "shots": false or "media": false skips that half; posts already captured', ' are skipped unless "force": true. A post that links to an X Article also', ' gets the article (article.json, .md, .png and its images) unless', ' "articles": false; both halves off with "articles": true reads only the', ' articles. Paced like a post fetch, on its queue.', "", 'persist-videos saves specific videos, across channels, to the saved-video', ' store: {"items": [{"slug", "id"}, ...]}. "format": "original" |', ' "video_720" (default: each channel\'s own). "replace": "above-height" also', ' re-fetches a saved one whose height is unknown or above that quality', ' (default "never"). "gapMs" pauses between downloads (default the batch', ' gap), "minFreeMemMb" waits for that much free memory before each.', ' "dryRun": true answers with the buckets (saved, wrongHeight, toFetch,', ' noUrl, unknown) and starts nothing. One job per channel, on its download', " queue; a low disk or a rate limit stops it, and running the same body", " again resumes — saved videos are skipped.", "", 'import-archive-org imports archive.org media into a channel, as one job on', ' archive.org\'s queue: {"slug", "item", "files": [...] | "match": ""}', ' for one item; {"slug", "items": ["" | {"item", "files"? | "match"?},', ' ...]} for many (a bare id takes "match", else every media original); or', ' {"slug", "query": "", "limit"?: 100} for the first', ' items a search finds (at most 500). One file at a time, a jittered pause', ' between files and between items; a record already held (on disk, or in', ' the saved-video store) is skipped. "dryRun": true lists each file as', ' held, RESTRICTED (archive.org marks it not for download: a fetch would', ' answer 401/403) or would get, and fetches nothing. A 401/403 skips the', ' rest of its item; three such items in a row, a 429 or three failures in', ' a row stop the job. The log ends with a summary: line; a re-run resumes.', "", 'attach-media copies each held video\'s file out of a LOCAL archive into the', ' saved-video store, as its source container (nothing fetched): {"slug",', ' "source"} — an absolute path to a directory, a .zip (read in place) or a', ' .7z. The id is the folder\'s trailing "()", else the file\'s yt-dlp', ' suffix; "items": [{"id", "path"}] names exact files (path inside the', ' source), "match" narrows the folders by regex. "createRecords": true', ' writes a record for a video the channel does not hold; "replace": true', ' re-attaches over a saved container. "dryRun": true logs each folder\'s', " class (attach, not-held, already-attached, lost, no-media, unmatched,", " ambiguous) and the held videos with no media in it, and writes nothing.", " The log ends with a summary: line; a re-run resumes.", "", 'prepare-playable remuxes each of a channel\'s saved containers, losslessly', ' (-c copy), into a browser-playable mp4 (+faststart) or webm, and makes', ' one single-file torrent per copy, under playable/ beside the saved-video', ' store: {"slug"}. "ids" narrows it; "trackers": [...] is each .torrent\'s', ' announce list (none by default; the infohash does not depend on it);', ' "root" names another playable root. A video prepared from the same', ' source (by sha256) is skipped; "dryRun": true logs each decision.', "", 'fetch-windows fetches clip windows, one paced job per platform queue', ' (YouTube and Rumble side by side): {"siteId"} fetches every window the', ' site\'s published reports cite and the disk does not hold; {"items":', ' [{"slug", "id", "from", "to", "clipId"?, "reason"?, "pad"?,', ' "webpageUrl"?}, ...], "requestedBy", "manifest"?} fetches a list.', ' "maxHeight" caps the source height (default 720). "dryRun": true lists', ' the windows per platform, the ones already on disk ("cached") and the', ' ones no fetch can fill ("unfetchable": deleted, off the site) and starts', ' nothing. A platform cooling down or held is refused for its group; a', ' 429, or two 403s in a row, backs the platform off and stops its job.', ' Running the same body again resumes — fetched windows are cached, and a', ' window a queued or running job will already write is answered in', ' "inFlight" with that job, whose id joins "jobIds" (so --wait follows it)', ' and no second job is queued. Windows run on the platform\'s clip queue', ' (clips:youtube), never behind its long downloads.', "", '"preview": "" on deploy-site or build-deploy makes it a Cloudflare', " Pages PREVIEW instead of production: the same bundle goes to a branch", " alias, https://..pages.dev, and the live site is left", " alone. The alias is printed after the response. Lowercase letters,", ' digits and dashes, up to 28 characters; "main" is refused.', "", 'build-hub builds the hub into its bundle; {"deploy": true} deploys it', ' after, and deploy-hub ships the one already built. Both deploy to the', " Pages project set on /sites under Hub, and take \"preview\" too.", "", 'build-homepage builds the homepage package into homepage/out;', ' {"deploy": true} deploys it after (only if the build succeeded), and', " deploy-homepage ships the one already built. Both deploy to the Pages", " project archilyzer (https://archilyzer.pages.dev), production unless", ' "preview" is given.', "", 'cut-release turns a changelog\'s [Unreleased] into "## [] - ":', ' {"workspace": "editor" | "export" | "all",', ' "version": "X.Y.Z" | "next" | "next-minor",', ' "commit": boolean (default false), "date": "YYYY-MM-DD" (default today)}.', ' "all" cuts both with ONE version and commits each ("Release ', ' ") — or neither: every check runs before either file is', " written. Only an editor built from release 10 or later has the route", " (an older one answers 404); with no editor running, `archilyzer release", " cut` does the same locally.", "", 'transcribe runs ONE local file through a local transcription worker, as a', ' job: {"path": "/abs/file"} (audio or video), "start"/"end" (seconds) for a', ' window, "workerId" (a settings worker id; default: the one auto-transcribe', ' would get), "out" (an absolute path for the result JSON, never inside the', " corpus). The result is {path, window, worker: {id, appId, model, device},", " cues: [{start, end, text}], text, ...}, cue times on the file's own clock.", ' "words": true adds words: [{w, start, end, conf?}] on the same clock -- from', " parakeet, which keeps its word timestamps; [] from an engine that does not.", " With --wait it is printed on stdout (the response and the log go to", " stderr), so `pnpm ops transcribe ... --wait | jq -r .text` works.", "", "Env: ARCHILYZER_EDITOR_URL (default http://localhost:3001), WORKER_TOKEN,", " ARCHILYZER_AGENT (provenance of a tag write; default \"cli\").", " WORKER_TOKEN and ARCHILYZER_EDITOR_URL, when unset, are read from", " editor/.env.local and editor/.env of this checkout (found from the", " script, not the cwd), then of the main worktree.", ].join("\n"); } // THE REPO ROOT, FROM THIS FILE — never from the cwd. `pnpm ops` runs with the // repo root as its cwd, but `node scripts/archilyzer-ops.mjs` from a subdir, // or a tool that shells out from its own project directory, does not; every // session used to write a wrapper that sourced editor/.env first. const REPO_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), ".."); // The two variables this client reads from the editor's env files. Nothing // else is taken from them: they also hold deploy credentials, which have no // business in this process. const EDITOR_ENV_KEYS = ["WORKER_TOKEN", "ARCHILYZER_EDITOR_URL"]; // `KEY=value` lines, as Next's own loader reads a .env: `#` comments, an // optional `export `, single or double quotes stripped. No interpolation. export function parseDotenv(text) { const out = {}; for (const raw of text.split(/\r?\n/)) { const line = raw.trim(); if (!line || line.startsWith("#")) continue; const m = /^(?:export\s+)?([A-Za-z_][A-Za-z0-9_]*)\s*=\s*(.*)$/.exec(line); if (!m) continue; let value = m[2].trim(); const q = value[0]; if ((q === '"' || q === "'") && value.length >= 2 && value.endsWith(q)) { value = value.slice(1, -1); } else { // An unquoted value ends at an inline comment. value = value.replace(/\s+#.*$/, ""); } out[m[1]] = value; } return out; } // The main worktree of a LINKED worktree, or null. A linked worktree's `.git` // is a file `gitdir:
/.git/worktrees/`, and that directory's // `commondir` names the main `.git`. Read off disk — no git process. export function mainWorktreeOf(root, read = (p) => readFileSync(p, "utf8")) { try { const m = /^gitdir:\s*(.+)$/m.exec(read(path.join(root, ".git"))); if (!m) return null; const gitdir = path.resolve(root, m[1].trim()); const common = path.resolve(gitdir, read(path.join(gitdir, "commondir")).trim()); return path.dirname(common); } catch { return null; } } // The env files consulted, in order: this checkout's, then the main worktree's. export function editorEnvFiles(root = REPO_ROOT, read) { const roots = [root]; const main = mainWorktreeOf(root, read); if (main && path.resolve(main) !== path.resolve(root)) roots.push(main); return roots.flatMap((r) => [ path.join(r, "editor", ".env.local"), path.join(r, "editor", ".env"), ]); } // FILL WHAT THE ENVIRONMENT LEFT UNSET from the editor's env files. A variable // already set — even to "" — wins: that is an explicit choice. The first file // that names a key supplies it. Returns where each came from, for the 401/503 // hint (never the value). export function loadEditorEnv( env = process.env, files = editorEnvFiles(), read = (p) => readFileSync(p, "utf8"), ) { const sources = {}; for (const key of EDITOR_ENV_KEYS) { if (env[key] !== undefined) sources[key] = "environment"; } for (const file of files) { let parsed; try { parsed = parseDotenv(read(file)); } catch { continue; } for (const key of EDITOR_ENV_KEYS) { if (sources[key] || parsed[key] === undefined) continue; env[key] = parsed[key]; sources[key] = file; } } return sources; } function baseUrl() { return (process.env.ARCHILYZER_EDITOR_URL || DEFAULT_URL).replace(/\/+$/, ""); } function authHeaders() { const token = process.env.WORKER_TOKEN ?? ""; return token ? { authorization: `Bearer ${token}` } : {}; } // What a 401 or a 503 means HERE, with where the token came from — never what // it is. A 503 is the server's own switch (it runs without WORKER_TOKEN); a 401 // is this side sending none, or one the server does not hold. export function tokenHint(status, sources, url = baseUrl()) { const from = sources.WORKER_TOKEN; if (status === 503) { return `hint: the editor at ${url} runs without WORKER_TOKEN, so /api/ops is off — set it in its editor/.env and restart it`; } if (!from) { return `hint: no WORKER_TOKEN in the environment or in ${editorEnvFiles().join(", ")}`; } return `hint: the WORKER_TOKEN from ${from} is not the one the editor at ${url} runs with`; } // Follow a job's log to its terminal state. Returns the status string. // Deliberately polls the SAME endpoint the editor's own log panel does, so a // job started here and a job started by a click are observed identically. // NO AUTH HEADER on the log poll, and that is not an omission. // `/api/jobs//log` is deliberately ungated (see its route for what it does // and does not leak, and why gating it would switch every run panel's log off // on a default install) — the browser polls it same-origin with no token. // Sending one here implied a gate that does not exist, which is worse than // sending nothing: the next person to read this would conclude the endpoint // was protected. // // A POLL FAILURE IS NOT A JOB FAILURE, and this used to treat them as the same // thing. The editor is single-process: a busy in-process build-index starves // the event loop for long enough that `fetch` rejects outright, and one // rejection ended the follow with a stack trace about a job that was running // fine and went on to finish. So every poll is caught, the wait backs off // 1→2→4…→30 s instead of hammering a server that is already struggling, and // `from` is kept across the failure so not one line of log text is lost. // // After three consecutive failures the endpoint is no longer trusted to answer // at all, and the question becomes a different one — IS THE JOB STILL THERE? // `GET /api/ops/job/` answers it (behind the token): one registry record // and one sidecar. It used to be `/api/jobs/active`, the UI's live view — a // next.config REWRITE onto /api/view/activeJobs, kept at its old path for the // pages and pinned widgets that poll it, which builds the whole live payload // (channel stats, the disk gate, every runner's status) to answer a yes/no, on // a server already too busy to answer the log poll. Queued or running ⇒ keep // waiting, however long that takes. Ended ⇒ the status it ended with, after // one last log poll for the lines we missed. Unknown (404) ⇒ the follow gives // up rather than claiming an outcome it never read. // // "archived" IS NOT AN OUTCOME EITHER. The log route says it for any id the // registry no longer holds (evicted past 100, or the editor restarted); the // job's sidecar still says how it ended, and the job route reads it. // // QUEUE POSITION is printed (stderr, unless --quiet) whenever it changes while // the job waits: "queued — position 3 on youtube". A job queued behind hours of // other work otherwise looks exactly like a hung command. // // `--wait-timeout` bounds the whole thing for a caller that cannot hang (CI, // an agent). Default none, because the honest default for a queue that may hold // a job behind hours of other work is to wait. export async function followJob(jobId, quiet, opts = {}) { const doFetch = opts.fetch ?? fetch; const sleep = opts.sleep ?? ((ms) => new Promise((r) => setTimeout(r, ms))); const now = opts.now ?? (() => Date.now()); const say = opts.say ?? ((line) => process.stderr.write(`${line}\n`)); const deadline = opts.timeoutSeconds > 0 ? now() + opts.timeoutSeconds * 1000 : null; const base = opts.baseUrl ?? baseUrl(); const headers = opts.headers ?? authHeaders(); const id = encodeURIComponent(jobId); let from = 0; let failures = 0; let probeFailures = 0; let backoff = 1000; let lastPlace = null; const notePlace = (status, queueKey, position) => { if (quiet || status !== "queued" || !(position > 0)) return; const place = `${position}@${queueKey ?? ""}`; if (place === lastPlace) return; lastPlace = place; say(`[${jobId}] queued — position ${position}${queueKey ? ` on ${queueKey}` : ""}`); }; // One log poll. Returns the status, or null when the poll itself failed — // never throws, so a transient fetch rejection cannot end the follow. const pollLog = async () => { try { const res = await doFetch(`${base}/api/jobs/${id}/log?from=${from}`); if (!res.ok) return null; const payload = await res.json(); if (payload.content) { // onContent sees every chunk, quiet or not, and says what to echo. const echo = opts.onContent ? opts.onContent(payload.content) : payload.content; if (echo && !quiet) process.stderr.write(echo); } // Only advance once the chunk is in hand: a poll that failed halfway // must re-ask for the same offset. from = payload.nextOffset ?? from; notePlace(payload.status, payload.queueKey, payload.queuePosition); return payload.status ?? null; } catch { return null; } }; // The job route: { status } when it answered, "gone" when the editor does // not know the id at all, null when the probe itself failed (or the editor // predates the route — an HTML 404 is not an answer). const probe = async () => { try { const res = await doFetch(`${base}/api/ops/job/${id}`, { headers }); const payload = await res.json().catch(() => null); if (!payload || typeof payload !== "object") return null; if (res.status === 404 && payload.ok === false) return "gone"; if (!res.ok || !payload.job) return null; return payload.job.status ?? null; } catch { return null; } }; // "queued" and "running" are the two NON-answers. Everything else is the job // having ended, which is the only thing worth returning. const terminal = (s) => s !== null && s !== "queued" && s !== "running"; const ended = (s) => s === "done" || s === "failed" || s === "cancelled"; for (;;) { const status = await pollLog(); if (status !== null) { failures = 0; probeFailures = 0; backoff = 1000; if (status === "archived") { // Not in the registry: the sidecar knows how it ended. const real = await probe(); return ended(real) ? real : status; } if (terminal(status)) return status; } else { failures++; if (failures >= 3) { const real = await probe(); if (ended(real)) { // It ended while the log endpoint was unreachable. One more try for // the lines we missed — its status does not override the record's. await pollLog(); return real; } if (real === "gone") { throw new Error( `lost contact with job ${jobId}: the editor no longer knows it and its log could not be read`, ); } if (real === "queued" || real === "running") { // A job we can still see is a job to wait for. failures = 0; probeFailures = 0; backoff = 1000; } else { // The probe failed too, so we now know nothing at all. Without a // bound this waits forever on an editor that has gone away; with one // it says so. Only CONSECUTIVE failures count — a single answer of // either kind resets it. probeFailures++; if (probeFailures >= MAX_PROBE_FAILURES) { throw new Error( `lost contact with the editor at ${base}: ${MAX_PROBE_FAILURES} consecutive failed polls of job ${jobId} and of /api/ops/job/${jobId}`, ); } } } backoff = Math.min(backoff * 2, 30_000); } if (deadline !== null && now() >= deadline) { throw new Error( `--wait-timeout: gave up after ${opts.timeoutSeconds}s waiting for job ${jobId}; it is still running and can be followed on /jobs`, ); } await sleep(status !== null ? 1000 : backoff); } } // Follow every id to its end, one after another, and say how each ended. // 0 only when every one finished `done`. A job still queued says where it waits // (followJob), so a list of ids behind one long job reads as a queue, not a // hang. In result mode (transcribe) each job's RESULT goes to stdout. async function followAll(jobIds, parsed, resultMode = false) { let worst = 0; for (const jobId of jobIds) { const capture = resultMode ? makeResultCapture(parsed.resultMarker) : null; const status = await followJob(jobId, parsed.quiet, { timeoutSeconds: parsed.waitTimeout ?? 0, ...(capture ? { onContent: capture.feed } : {}), }); if (capture) { const { echo, result } = capture.finish(); if (echo && !parsed.quiet) process.stderr.write(echo); if (result !== null) { console.log(JSON.stringify(result, null, 2)); } else if (status === "done") { console.error(`[${jobId}] finished but its log carries no result`); worst = 1; } } console.error(`[${jobId}] ${status}`); if (status !== "done") worst = 1; } return worst; } async function main() { const parsed = parseArgs(process.argv.slice(2)); if (parsed.help) { console.log(usage()); return 0; } if (parsed.error) { console.error(parsed.error); console.error(""); console.error(usage()); return 2; } if (parsed.list) { console.log(ACTIONS.join("\n")); return 0; } if (parsed.waitFor) { // `job wait `: no request of its own, only the follow. loadEditorEnv(); return followAll(parsed.waitFor, parsed); } if (parsed.bodyFile) { let raw; try { raw = await readFile(parsed.bodyFile, "utf8"); } catch (e) { console.error(`--file: ${e.message}`); return 2; } try { parsed.body = JSON.parse(raw); } catch (e) { console.error(`--file ${parsed.bodyFile} is not valid JSON: ${e.message}`); return 2; } if ( typeof parsed.body !== "object" || parsed.body === null || Array.isArray(parsed.body) ) { console.error(`--file ${parsed.bodyFile} must hold a JSON object`); return 2; } } if (parsed.defaultSource && parsed.body && parsed.body.source === undefined) { parsed.body = { ...parsed.body, source: parsed.defaultSource }; } const envSources = loadEditorEnv(); const url = `${baseUrl()}${parsed.path}`; const res = await fetch(url, { method: parsed.method, headers: { ...authHeaders(), ...(parsed.method === "POST" ? { "content-type": "application/json" } : {}), }, ...(parsed.method === "POST" ? { body: JSON.stringify(parsed.body) } : {}), }); const text = await res.text(); let payload; try { payload = JSON.parse(text); } catch { console.error(`HTTP ${res.status}: ${text.slice(0, 500)}`); return 1; } if (res.status === 401 || res.status === 503) { console.error(tokenHint(res.status, envSources)); } // A result-carrying action under --wait keeps stdout for the RESULT: the // response goes to stderr with the log. const resultMode = Boolean(parsed.wait && parsed.resultMarker); (resultMode ? console.error : console.log)(JSON.stringify(payload, null, 2)); if (!res.ok || payload.ok === false) return 1; if (!parsed.wait) { printPreviewUrls(payload); return 0; } // `jobIds` FIRST: a bulk fan-out (relocate, relocate-back) returns an array // and also a single `jobId` when it started exactly one, so reading `jobId` // first would follow one job out of five. Before `jobIds` existed those // routes carried no id at all and --wait returned immediately, reporting // success about a copy that had not begun. const jobIds = Array.isArray(payload.jobIds) ? payload.jobIds : payload.jobId ? [payload.jobId] : Array.isArray(payload.jobs) ? payload.jobs.map((j) => j.jobId) : []; if (jobIds.length === 0) { // Not a job-starting action (or it queued nothing). --wait is satisfied. printPreviewUrls(payload); return 0; } const worst = await followAll(jobIds, parsed, resultMode); // LAST, after the logs: with --wait the response scrolled off minutes ago, // and the alias is the one thing the operator came for. printPreviewUrls(payload); return worst; } // Importable for the arg-parsing tests; only the CLI entry point runs main(). if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { main().then( (code) => process.exit(code), (e) => { console.error(e.message); process.exit(1); }, ); }