// Next.js instrumentation hook. register() runs once when a server instance // starts (stable in Next 16). We use it to arm the in-process scheduler // heartbeat so auto-sync can run without an external cron job. // // See editor/app/scheduler/heartbeat.ts and SCHEDULED_SYNC.md. // // Everything armed here that starts or writes work is skipped when the process // boots idle (ARCHILYZER_IDLE_BOOT) — see isIdleBoot below. What stays armed // only stops work or only reads: the shutdown reaper, the persisted-pause // restore, the storage boot probe and the drive health pass. // // The one STATIC import in this file, and safe as one because idleBoot.ts // imports nothing and touches no Node API: the Edge bundle's static Node-API // scan has nothing to object to. Every other import stays lazy. import { isIdleBoot } from "yt-dlp-transcript-common/lib/idleBoot"; export async function register() { // register() is called in every runtime (Node.js and Edge). The heartbeat and // its transitive imports (runTick -> server actions, lmdb, fs) are Node-only, // so guard the import — and run nothing on Edge. if (process.env.NEXT_RUNTIME !== "nodejs") return; // Anything a previous process left `queued` was queued before this instant. const bootedAt = Date.now(); // Cancel in-flight children on a graceful shutdown. Armed FIRST, before any of // the runners below: a failure while starting those must not leave the server // without its only way to reap the children they spawn. // // Must be a lazy import like the rest — the module uses process.once/kill/ // listenerCount, and inlining it here fails the Edge bundle's static Node-API // scan even though the guard above means it never runs there. See // common/jobs/shutdownCancel.ts. try { const { armShutdownCancel } = await import( "yt-dlp-transcript-common/jobs/shutdownCancel" ); armShutdownCancel(); } catch { /* failing to arm the reaper must not block server readiness */ } // Boot idle: the operator asked for a server, not for whatever the corpus's // stored policies were in the middle of. Pointing a fresh container at an // existing corpus would otherwise resume a GPU-weeks digest sweep seconds // after `docker compose up`. See common/lib/idleBoot.ts. // // Two things stay armed regardless, because both only ever STOP work: the // shutdown reaper above (a server that cannot reap its children is never // correct) and the persisted-pause restore below. const idle = isIdleBoot(); if (!idle) { // Lazy import inside the guard keeps server-only code out of the Edge bundle. // startSyncHeartbeat only arms a timer (no synchronous tick), so it never // blocks the server from becoming ready. const { startSyncHeartbeat } = await import("./app/scheduler/heartbeat"); startSyncHeartbeat(); } // Restore a persisted transcription pause: if the operator paused // transcriptions and the server later restarted, re-pause the worker pool so // the pause survives the restart (it's otherwise a live-pool-only state). The // downloads pause is enforced by reading settings at dispatch time, so it // needs no boot step. Best-effort — never block server readiness. try { const { getSettings } = await import( "yt-dlp-transcript-common/lib/settings" ); const { isGateHeld } = await import( "yt-dlp-transcript-common/lib/pauseGates" ); // One of exactly two places the PERSISTED transcription gate is read (the // other is the action's "did this change anything" check). Everything that // asks "is transcription held right now" reads the pool instead. Since // slice 1.4 the stored value is `autoQueue.transcription.held`; this line // needs no edit for that, because isGateHeld is the one thing that knows // where a lane's gate lives. if (isGateHeld(getSettings(), "transcription")) { const { getWorkerPool } = await import( "yt-dlp-transcript-common/jobs/workerPool" ); getWorkerPool().pauseAll(); } } catch { /* a failed re-pause must not block server readiness */ } // THE DRIVE HEALTH PASS, every `storage.health.passIntervalMs` (15 s by // default; a save on /storage re-arms it) — and ON AN IDLE BOOT TOO, like the // probe below. It reads each location's block device counters (never the // drive) and keeps, in memory only, which drives are not answering; every // page and poll asks it before touching a drive, and its watchdog marks a // drive a page reached and got no answer from. It writes nothing and starts // no work, and without it nothing would ever clear such a mark. The // five-minute pass that may auto-pause channels is armed below the idle gate. // See common/controller/storageWatch.ts and common/lib/storageHealth.ts. try { const { startStorageHealthWatch } = await import( "yt-dlp-transcript-common/controller/storageWatch" ); startStorageHealthWatch({ log: (line) => console.log(line) }); } catch { /* a health pass that fails to arm must not block server readiness */ } // ONE PASS OVER THE STORAGE LOCATIONS, and it runs on an IDLE BOOT TOO: it // only reads, like the health pass above. // // The pass probes each location (is the disk here, and if not, where?) and, // for a location the operator armed with `autoRepoint`, re-points it to // wherever the volume actually came up. Probing is read-only — one findmnt // per location — and it is exactly what a machine that has just rebooted with // its platter on a different mountpoint needs done before anybody opens a // page. What idle boot refuses is STARTING WORK, so `enqueue: !idle` is the // whole of the gate: an idle container learns where its disks are and // enqueues nothing. // // Lazy-imported and `void`ed like everything else here: the controller pulls // in execa transitively (Node-only), and a disk that cannot be probed must // never block server readiness. // Kept so the queued-meta pass below can wait for it: a re-queue for a // channel whose location is mid-autoRepoint would be refused as unreachable. let storagePass: Promise = Promise.resolve(); try { const { runStorageBootPass } = await import( "yt-dlp-transcript-common/controller/storageLocations" ); storagePass = runStorageBootPass({ enqueue: !idle }).catch(() => {}); } catch { /* a storage probe that fails to start must not block server readiness */ } // STALE `queued` METAS FROM THE LAST PROCESS (release 9, B4b). A job still // waiting in its queue when the server stopped never got its terminal meta // write, so /jobs would read it as queued forever. Each is closed // `cancelled` with a reason; only a few are re-queued (through the path Retry // uses): never a sync (the heartbeat re-derives those, paced), nothing older // than 24 h, and only the newest of duplicate specs. Runs on an idle boot // too, but there it ONLY cancels — an idle boot must not resume work — and so // does the e2e test server, whose leftover metas belong to a previous run's // fixture. It waits for the storage pass above, so a channel being // re-pointed is reachable when its job is re-queued — for at most // STORAGE_PASS_WAIT_MS (60 s, release 10): a probe stuck on a hung mount // must not keep every stale meta `queued` for the life of the process. The // timeout is logged, and the storage pass itself runs on untouched. See // common/jobs/bootQueuedJobs.ts. Lazy, voided, best-effort: never blocks // readiness. try { const { settleAfterStoragePass, settleRunningJobMetas } = await import( "yt-dlp-transcript-common/jobs/bootQueuedJobs" ); const { getPaths } = await import("yt-dlp-transcript-common/lib/paths"); const { getRegistry } = await import( "yt-dlp-transcript-common/jobs/registry" ); const testServer = process.env.E2E_TEST_ROUTES === "1"; // STALE `running` METAS FROM A DEAD PROCESS (release 17 slice D0): closed // `cancelled` as interrupted, never re-run — on every boot, idle and test // server included, because closing one starts nothing. Does not wait for // the storage pass: it touches no channel. A meta another live process // still owns (`archilyzer run`) is left alone — see writerIsGone. void settleRunningJobMetas({ paths: getPaths(), bootedAt, isLive: (id) => getRegistry().get(id) !== undefined, log: (line) => console.log(line), }).catch(() => {}); const cancelOnly = idle || testServer; void settleAfterStoragePass(storagePass, { paths: getPaths(), bootedAt, isLive: (id) => getRegistry().get(id) !== undefined, log: (line) => console.log(line), idleReason: idle ? "idle boot" : testServer ? "test server" : undefined, requeue: cancelOnly ? null : async (spec) => { const { runJobSpec } = await import("./app/jobs/runJobSpec"); const res = await runJobSpec(spec); if (!res.ok) return { ok: false, error: res.error }; // Nobody reads this stream; release it as the ops routes do. void res.stream.cancel(); return { ok: true, jobId: res.jobId }; }, }).catch(() => {}); } catch { /* a boot pass that fails to start must not block server readiness */ } // Everything past here STARTS work. On an idle boot, nothing does. if (idle) return; // THE DRIVE WATCH. Every guard the corpus has for an unreachable channel runs // at the start of a piece of work — none of them is a DETECTOR, so a channel // whose drive vanished sits there being refused with nothing saying why. This // is the thing that looks, on a five-minute cadence, and auto-pauses (and // later restores) the channels on a location that is not there. // // BELOW THE IDLE GATE, deliberately, and unlike the boot probe and the // health pass above: this one WRITES settings.channelPriority, and a // container pointed at somebody else's corpus for the first time has no // business rewriting that corpus's priority document. See // common/controller/storageWatch.ts. try { const { startStorageWatch } = await import( "yt-dlp-transcript-common/controller/storageWatch" ); startStorageWatch({ log: (line) => console.log(line) }); } catch { /* a watch that fails to arm must not block server readiness */ } // Start the automatic priority-queue runners — all four lanes — if their // policies are enabled. Each is a self-managed registry job; this only // kicks them off and returns. Best-effort — a failure here must not stop the // server from starting. try { const { startAutoRunnersIfEnabled } = await import( "yt-dlp-transcript-common/controller/autoRunner" ); await startAutoRunnersIfEnabled(); } catch { /* a runner that fails to start must not block server readiness */ } // THE PUBLISH LANE (release 18): its runner, when settings.publish.enabled. // Beside the four above rather than inside startAutoRunnersIfEnabled — the // dispatch layer may not import the publish layer — and, like them, below // the idle gate: an idle boot leaves it off. try { const { startPublishRunnerIfEnabled } = await import( "yt-dlp-transcript-common/publish/publishRunner" ); await startPublishRunnerIfEnabled(); } catch { /* a runner that fails to start must not block server readiness */ } // NO SECOND RESUME HOOK. There used to be two more here, one per sweep, and // each re-launched a corpus-wide pass from its own persisted flag. Both lanes // are runner lanes now, so `startAutoRunnersIfEnabled` above resumes all four // from one switch — `autoQueue[lane].enabled` — which is what a lane being // armed has always meant on the transcription and download lanes. }