import path from "node:path"; import { lstat, readlink, stat, symlink, unlink } from "node:fs/promises"; import { getPaths, type Paths } from "../lib/paths"; import { getFreeBytes } from "../lib/diskSpace"; import { getSettings, writeSettings, type SiteSettings, } from "../lib/settings"; import { INTERNAL_LOCATION_ID, INTERNAL_LOCATION_LABEL, locationOfDataDir, type StorageLocation, type StorageVolume, } from "../lib/storageLocations"; import { probeLocation, probeLocationMemo, resetStorageProbeMemo, PROBE_MEMO_MS, type MemoizedProbe, type ProbeOptions, type StorageLocationProbe, type VolumeBins, } from "../lib/storageVolumes"; import { forgetChannelMedia, inspectChannelMedia, relocatedDataDir, } from "../lib/channelMedia"; import { MEDIA_LINK_NAME, relocatedMediaDir } from "../lib/mediaTier-server"; import { isDriveNotAnswering, onDrive, stalledLocation, } from "../lib/storageHealth"; import { relocationQueueKey } from "../lib/queueKeys"; import { runManagedFunction, type StreamActionResult, } from "../jobs/streamCommand"; import type { ChannelConfig } from "../lib/channelConfig"; import { listChannelConfigs, patchChannelConfig, readChannelConfig, } from "./channels"; import { relocationRootProblem } from "./relocateChannelMedia"; // WHAT A STORAGE LOCATION KNOWS ABOUT THE CHANNELS ON IT, AND HOW IT MOVES. // // The probe next door (`lib/storageVolumes.ts`) answers "is this disk here, and // if not, where?" for ONE location in isolation. This module is the half that // knows the corpus: which channels are on a location, whether a re-point is // safe, and the re-point itself. // // RE-POINT MOVES NO BYTES. It rewrites each channel's `media` symlink and its // `config.mediaDir`, then the location's `root` (release 17: a channel's media // tier is what lives on a location; its text never leaves the corpus disk). That is the entire operation — // the media is already where it is going, because the DISK came up somewhere // else and took it along. The relocate job (which does move bytes) is a // different thing entirely, and the two share a queue key precisely so they can // never run at once. // // THE BUSY CHECK IS INJECTED, and this is the one design choice worth naming. // `channelMediaBusyReason` (S1) lives in `editor/app/channels/lib/mediaBusy.ts` // because it reads two PROCESS singletons — the job registry and the auto-queue // runner status — and a view-model layer that constructs singletons is the rule // one-core phase 3 exists to hold (see `common/views/inputs.ts`). Moving it down // here would make every unit test of this controller create a registry and four // lane runners just to be told nothing is running. So the seam is a callback: // the editor passes its helper, the tests pass a stub, and the default is "not // busy" — which is correct for every non-editor caller, none of which has an // in-process runner to be busy with. // Why a channel cannot be touched right now, or null. The editor passes // `channelMediaBusyReason`; a bin script passes nothing. export type BusyCheck = (slug: string, what?: string) => string | null; // Reading and writing settings, injectable so a unit test neither reads the // operator's settings.json nor has to point the global Paths cache at a tmp // dir. Production callers pass nothing. export type SettingsIO = { read: () => SiteSettings; write: (next: SiteSettings) => Promise; }; const DEFAULT_SETTINGS_IO: SettingsIO = { read: getSettings, write: writeSettings, }; // --------------------------------------------------------------------------- // The roll-up: which channels are on which location, and how they read // --------------------------------------------------------------------------- export type LocationRollup = { locationId: string; // Every channel whose `config.mediaDir` is under this location's root (or, // on a `legacy` channel, its retired `config.dataDir`), sorted. slugs: string[]; total: number; // Sum of `snapshot.totalMediaBytes` over the channels on this location — the // MEDIA TIER's bytes, which is what a location holds (release 17) — and // how many of them could not contribute one (no snapshot, or one written // before the field existed). THE SECOND NUMBER IS WHY THE FIRST IS HONEST: a // location whose channels have never had a report reads `0 bytes` otherwise, // which on a storage page is a claim that a 500 GB drive is empty. Every // surface renders "+ n unknown" beside the total. bytes: number; unknownBytes: number; // THE CORPUS-DISK BYTES of these channels: `clips/` (the fetched clip // windows, a cache nothing prunes) and the text tier. A SIBLING of `bytes` // since release 17, not a share of it — both stay on the corpus volume // whatever location the media is on, so only the INTERNAL row adds them to // what it holds; a location row shows them as "on the corpus volume". // // No `unknownClipsBytes`: it would be the same set of channels // `unknownBytes` already counts (the fields are written by one snapshot // pass), and a second copy of one number is a second thing to keep in step. // `textBytes` has its own unknown count: a snapshot written before release // 17 carries clips and media but no text figure. clipsBytes: number; textBytes: number; unknownTextBytes: number; // `inspectChannelMedia` status, bucketed into the three numbers the page // shows. `unreachable` DELIBERATELY ABSORBS `inconsistent`: both mean "this // channel's media is not readable through its link right now", which is the // question a storage page is asking, and a fourth column for a state the // re-point preflight refuses by name anyway would be a number nobody acts on. ok: number; unreachable: number; // `in-transition`: a relocation marker is present. A location with any of // these is one no re-point may touch. moving: number; // `legacy` — the retired whole-directory layout, waiting for // `archilyzer storage migrate-tier`. Counted in `unreachable` too (its // text and media are held); this is the "(n to migrate)" beside it. legacy: number; }; function emptyRollup(locationId: string): LocationRollup { return { locationId, slugs: [], total: 0, bytes: 0, unknownBytes: 0, clipsBytes: 0, textBytes: 0, unknownTextBytes: 0, ok: 0, unreachable: 0, moving: 0, legacy: 0, }; } // THE SYNTHETIC LOCATION: the corpus volume itself. // // `paths.channelsDir` is where a channel's media lives when it has not been // moved anywhere, and that is the row the operator actually acts on — "what is // still on the internal disk, largest first" is the whole question a storage // page is asked. It is NOT stored in settings and never will be: there is // nothing to configure (the root is wherever the corpus is), nothing to // re-point (re-pointing the corpus is moving the corpus), and writing it into // `settings.storage.locations` would make it deletable and would make // `locationOfDataDir` match every unrelocated channel — which would break the // one rule the whole design rests on, that a channel is on a location iff its // `mediaDir` is under that location's root, and an in-place channel HAS no // `mediaDir`. // // So it is assembled where it is rendered, out of the same three facts every // other row carries, and it is the one row with no actions. // THE TWO CONSTANTS LIVE IN `lib/storageLocations.ts` and are re-exported // here, where every caller names them. They had to move: `lib/settings.ts` // must REFUSE "internal" as a stored location id (see `LOCATION_ID_RE` there), // and lib may not import controller. export { INTERNAL_LOCATION_ID, INTERNAL_LOCATION_LABEL }; // One pass over the corpus for EVERY location, not one pass per location: the // page draws a row per location and the channel list is the same list for all // of them. Cost is `listChannelConfigs` (one readdir + one config read per // channel) plus `inspectChannelMedia` (two stats) for the channels that are // actually on a location — an unrelocated channel has no `mediaDir` and is // skipped before it costs a stat. export async function channelsOnLocation(opts: { paths: Paths; locations: readonly StorageLocation[]; // Already-read configs, when the caller has them (the /storage shell does // not, the channels page does). configs?: ReadonlyArray<{ slug: string; config: ChannelConfig }>; // `snapshot.totalMediaBytes` by slug, INJECTED rather than read here. A // snapshot is up to a megabyte of JSON per channel and the two callers // already hold theirs (`listChannelBriefs`); a controller that went and read // 71 of them to add one integer each would be the third read of the same // file in one render. A slug that is absent (or maps to undefined) counts // towards `unknownBytes`, never towards `bytes`. mediaBytes?: Readonly>; // `snapshot.totalClipsBytes` per slug — the clip cache, on the corpus disk. // Absent for a snapshot written before the field existed, which is the same // set `mediaBytes` is absent for. clipsBytes?: Readonly>; // `snapshot.totalTextBytes` per slug — the text tier, on the corpus disk. // Absent for a snapshot written before release 17: unknown, never 0. textBytes?: Readonly>; // The in-place row (see INTERNAL_LOCATION_ID). When true, every channel with // NO `mediaDir` (and no retired `dataDir`) is rolled up under that id // alongside the configured ones. includeInternal?: boolean; }): Promise> { const out: Record = {}; for (const loc of opts.locations) out[loc.id] = emptyRollup(loc.id); if (opts.includeInternal) { out[INTERNAL_LOCATION_ID] = emptyRollup(INTERNAL_LOCATION_ID); } if (opts.locations.length === 0 && !opts.includeInternal) return out; const configs = opts.configs ?? (await listChannelConfigs(opts.paths)); for (const { slug, config } of configs) { // Where the channel's media is: `mediaDir`, or on a legacy channel the // retired `dataDir` (its whole tree is still there until migrate-tier). const mediaDir = config.mediaDir?.trim() || config.dataDir?.trim(); const loc = mediaDir ? locationOfDataDir(mediaDir, opts.locations as StorageLocation[]) : null; // In place: no recorded target at all. A target under a root NOBODY named // is neither in place nor on a location, and it is deliberately counted in // neither — /storage says so by the totals not adding up to the corpus, and // the remedy is to name that root as a location. const id = loc ? loc.id : !mediaDir && opts.includeInternal ? INTERNAL_LOCATION_ID : null; if (!id) continue; const roll = out[id]; roll.slugs.push(slug); roll.total += 1; const bytes = opts.mediaBytes?.[slug]; if (typeof bytes === "number") roll.bytes += bytes; else roll.unknownBytes += 1; const clips = opts.clipsBytes?.[slug]; if (typeof clips === "number") roll.clipsBytes += clips; const text = opts.textBytes?.[slug]; if (typeof text === "number") roll.textBytes += text; else roll.unknownTextBytes += 1; const media = await inspectChannelMedia(opts.paths, slug, config); if (media.status === "ok" || media.status === "in-place") roll.ok += 1; else if (media.status === "in-transition") roll.moving += 1; else { roll.unreachable += 1; if (media.status === "legacy") roll.legacy += 1; } } for (const roll of Object.values(out)) roll.slugs.sort(); return out; } // HOW MUCH ROOM EACH VOLUME HAS, WITHOUT PROBING. // // `probeLocation` is up to three subprocesses and is right for /storage, which // renders once per navigation. It is wrong for /channels, which renders on // every auto-refresh with 71 rows on it — the rule the badge already follows is // that TABLES NEVER PROBE. // // So this is two syscalls per location: one `stat` to find out whether the root // is there at all, and `getFreeBytes` only when it is. The stat is what makes // the answer honest — `getFreeBytes` walks up to the nearest existing ancestor // on ENOENT, so an unmounted root would otherwise report the free space of // whatever is mounted over its parent, which on this machine is the disk the // operator is trying to empty. Absent → undefined, rendered as "—". // // `internal` is always present: it is the corpus volume, and the process is // reading the corpus out of it. export async function volumeFreeBytes(opts: { paths: Paths; locations: readonly StorageLocation[]; }): Promise> { const out: Record = {}; const corpus = await getFreeBytes(opts.paths.channelsDir); out[INTERNAL_LOCATION_ID] = Number.isFinite(corpus) ? corpus : undefined; await Promise.all( opts.locations.map(async (loc) => { // NOT ASKED WHILE ITS DRIVE IS NOT ANSWERING: each stat and the statfs // below would hold an I/O thread until it did. "—", like unmounted. The // calls that reach the drive go through `onDrive`'s watchdog; one that // has not answered within the budget (3 s by default) marks the location // stalled, and reads "—" too. if (stalledLocation(loc)) { out[loc.id] = undefined; return; } try { out[loc.id] = await freeOnLocation(loc); } catch (err) { if (!isDriveNotAnswering(err)) throw err; out[loc.id] = undefined; } }), ); return out; } // One location's free space, for volumeFreeBytes. Throws DriveNotAnsweringError // when its drive does not answer. async function freeOnLocation( loc: StorageLocation, ): Promise { const orNull = (p: Promise): Promise => p.catch((err) => { if (isDriveNotAnswering(err)) throw err; return null; }); const st = await orNull(onDrive(loc, () => stat(loc.root))); if (!st?.isDirectory()) return undefined; // THE DIRECTORY EXISTING IS NOT THE DRIVE BEING THERE, and this is the // half the stat above could not catch. An unmounted mountpoint is a real, // empty directory ON ITS PARENT'S FILESYSTEM — so `getFreeBytes` succeeds // and reports the parent volume's free space, which on this machine is // the disk the operator is trying to empty. The location would then read // "233 GB free" about a platter that is not plugged in. // // THE BOUNDARY IS TESTED AT THE MOUNTPOINT, NEVER AT THE ROOT, and the // first version of this got that wrong in the one way that matters on // this machine. A location's root is `join(mountpoint, relPath)` — the // production `platter` is `/run/media///archilyzer-media`, a // SUBDIRECTORY of the mountpoint — so the root and its parent are on the // same filesystem BY CONSTRUCTION whenever `relPath` is non-empty, and // comparing those two devices reported "free space unknown" for a // correctly mounted drive. A bind mount reads the same way. // // Crossing a mount changes the device number, so `stat(mountpoint).dev` // against `stat(dirname(mountpoint)).dev` is the honest question, and it // is two syscalls — TABLES NEVER PROBE (see the header) rules out asking // `probeLocation`, which is up to three subprocesses. // // Only asked of a location that has learned a `volume.uuid`: that field // is the assertion that the root is supposed to be on its own volume. A // location on a plain directory (never probed, a container, a // subdirectory of the system disk by design) shares its parent's device // legitimately, and withholding its free space would be wrong. const mountpoint = loc.volume?.uuid ? (loc.volume.mountpoint ?? "").trim() : ""; // `/` is its own parent, so a volume mounted at the root has no boundary // to test and is trivially there — the process is reading from it. if (mountpoint && mountpoint !== path.dirname(mountpoint)) { // The mountpoint is the drive's own top directory; its parent is on the // filesystem above it, so only the first goes through the watchdog. const [atMount, aboveMount] = await Promise.all([ orNull(onDrive(loc, () => stat(mountpoint))), stat(path.dirname(mountpoint)).catch(() => null), ]); // Gone entirely, or present as an ordinary directory on the parent // filesystem: either way nothing is mounted there. if (!atMount || (aboveMount && aboveMount.dev === atMount.dev)) { return undefined; } } const free = await onDrive(loc, () => getFreeBytes(loc.root)); return Number.isFinite(free) ? free : undefined; } // --------------------------------------------------------------------------- // The probe memo // --------------------------------------------------------------------------- // // IT LIVES IN `lib/storageVolumes.ts` NOW, and is re-exported here because // every caller names it through this module. It had to move: the relocation // movers' root-presence guard (`assertRelocationRootPresent`) probes, and it is // called per channel from the bulk move and the re-point preflight — but THIS // module imports `relocationRootProblem` from `relocateChannelMedia.ts`, so a // memo reached from there through here would be an import cycle. Nothing about // a ten-second probe memo is controller-level; it is volume machinery. export { probeLocationMemo, resetStorageProbeMemo, PROBE_MEMO_MS, type MemoizedProbe, }; export async function probeAllLocations( locations: readonly StorageLocation[], bins: VolumeBins, opts: ProbeOptions & { refresh?: boolean; now?: number } = {}, ): Promise> { const out: Record = {}; await Promise.all( locations.map(async (loc) => { out[loc.id] = await probeLocationMemo(loc, bins, opts); }), ); return out; } function sameVolume(a?: StorageVolume, b?: StorageVolume): boolean { if (!a || !b) return a === b; return ( a.uuid === b.uuid && a.mountpoint === b.mountpoint && a.relPath === b.relPath && (a.fstype ?? "") === (b.fstype ?? "") && (a.label ?? "") === (b.label ?? "") ); } // IDENTITY ONLY, AND ONLY WHEN IT CHANGED. // // A refresh must never write availability — that would rewrite settings.json // (and so bump the pulse revision every reader polls) each time anybody loaded // the page. What a refresh MAY learn is the disk's identity: the uuid, the // mountpoint it is at, and the root's path relative to it. That is what turns // "the platter is gone" into "the platter is at /mnt/platter now", so it is // worth a write — but only on the probe that actually changed it. // // Returns whether it wrote. export async function recordProbedIdentity(opts: { locationId: string; probe: StorageLocationProbe; io?: SettingsIO; }): Promise { const io = opts.io ?? DEFAULT_SETTINGS_IO; const identity = opts.probe.identity; if (!identity.known) return false; const settings = io.read(); const loc = settings.storage.locations.find((l) => l.id === opts.locationId); if (!loc) return false; const volume: StorageVolume = { uuid: identity.uuid, ...(identity.fstype ? { fstype: identity.fstype } : {}), ...(identity.label ? { label: identity.label } : {}), mountpoint: identity.mountpoint, relPath: identity.relPath, }; if (sameVolume(loc.volume, volume)) return false; await io.write({ ...settings, storage: { ...settings.storage, locations: settings.storage.locations.map((l) => l.id === opts.locationId ? { ...l, volume } : l, ), }, }); return true; } // --------------------------------------------------------------------------- // The re-point preflight // --------------------------------------------------------------------------- export type RepointPreflight = { ok: boolean; // Every reason this re-point is refused, each naming the thing to fix. problems: string[]; // The channels that would be re-pointed, in the order the job would do it. channels: string[]; // A SUBSET of `channels`: the ones whose symlink ALREADY points at the new // target while config.json still names the old one. That is the state a crash // in the one-instruction window between the symlink and the config write // leaves behind, and it reads as `inconsistent` — so without this the channel // stays "on" the old root for ever and every later re-point is refused // wholesale, naming a state with no remedy. For these the job writes the // config and does NOT touch the link: the link is already right. resumable: string[]; // A SUBSET of `channels`: the ones still on the RETIRED whole-directory // layout (`legacy`, release 17). They are re-pointed the retired way — their // `data` link and `config.dataDir` to `//data` — so a // location holding migrated and not-yet-migrated channels side by side // still follows its disk, and the tier migration only ever sees // `dataDir = //data`. legacy: string[]; // The identity the new root actually has, when the preflight was able to // probe it (bins passed AND the location has a recorded uuid to compare // against). The job writes THIS rather than deriving a mountpoint from the // old record — it is a measurement, not an inference. newVolume?: StorageVolume; locationId: string; oldRoot: string; newRoot: string; }; // What a re-point rewrites for one channel: the `media` link and `mediaDir`, // or — a legacy channel — the retired `data` link and `dataDir`. type RepointShape = { linkName: string; key: "mediaDir" | "dataDir"; targetFor: (root: string, slug: string) => string; }; const MEDIA_SHAPE: RepointShape = { linkName: MEDIA_LINK_NAME, key: "mediaDir", targetFor: relocatedMediaDir, }; const LEGACY_SHAPE: RepointShape = { linkName: "data", key: "dataDir", targetFor: relocatedDataDir, }; // Does the channel's link (`media`, or a legacy channel's `data`) already point // exactly where a re-point would put it? lstat/readlink, never stat: the target // may not exist yet either, and a stat would call a perfectly good link // missing. async function linkAlreadyAt( paths: Paths, slug: string, newTarget: string, linkName: string = MEDIA_LINK_NAME, ): Promise { try { const linkTarget = await readlink( path.join(paths.channelsDir, slug, linkName), ); return path.resolve(linkTarget) === path.resolve(newTarget); } catch { return false; } } async function isDirectory(p: string): Promise { try { return (await stat(p)).isDirectory(); } catch { return false; } } // NOTHING IS WRITTEN HERE. Every refusal the job can make is made here first, // so the operator reads it in the page rather than in a job log, and so // `maybeAutoRepoint` can decline silently instead of queueing a job that will // throw. export async function preflightRepoint(opts: { paths: Paths; locationId: string; newRoot: string; io?: SettingsIO; bins?: VolumeBins; probeOptions?: ProbeOptions; isBusy?: BusyCheck; }): Promise { const io = opts.io ?? DEFAULT_SETTINGS_IO; const settings = io.read(); const loc = settings.storage.locations.find((l) => l.id === opts.locationId); const newRoot = opts.newRoot.trim().replace(/\/+$/, "") || opts.newRoot.trim(); const base: RepointPreflight = { ok: false, problems: [], channels: [], resumable: [], legacy: [], locationId: opts.locationId, oldRoot: loc?.root ?? "", newRoot, }; if (!loc) { base.problems.push(`There is no storage location "${opts.locationId}".`); return base; } if (!path.isAbsolute(newRoot)) { base.problems.push( `The new root must be an absolute path (got "${opts.newRoot}").`, ); return base; } if (newRoot === loc.root) { base.problems.push( `"${loc.label}" is already at ${loc.root} — there is nothing to re-point.`, ); return base; } if (!(await isDirectory(newRoot))) { base.problems.push( `${newRoot} is not a directory. Mount the volume there first.`, ); return base; } // IDENTITY IS A REFUSAL, NEVER A REQUIREMENT. Two KNOWN uuids that differ // mean the operator is about to point a location at a different disk that // happens to hold directories with the right names, and every channel's // symlink would then silently read someone else's media. Unknown on either // side (no probe yet, a container where block devices are invisible) is not // evidence of anything and does not block: re-point by path is the whole // story there. See RUNNING_IN_DOCKER.md §another drive. if (opts.bins && loc.volume?.uuid) { const probe = await probeLocation( { ...loc, root: newRoot }, opts.bins, opts.probeOptions ?? {}, ); if (probe.identity.known && probe.identity.uuid === loc.volume.uuid) { // The same disk, measured at the new root: uuid, fstype, label, the // mountpoint it is actually on and the root's path relative to it. This // is what the job stores, so `root === join(mountpoint, relPath)` holds // because it was observed and not reconstructed. const { known: _known, ...volume } = probe.identity; base.newVolume = volume; } if (probe.identity.known && probe.identity.uuid !== loc.volume.uuid) { base.problems.push( `${newRoot} is on a different disk: it is on UUID ` + `${probe.identity.uuid}, and "${loc.label}" was last seen on ` + `${loc.volume.uuid}. Re-point is for a volume that moved, not for ` + `a different volume — move the media if that is what you meant.`, ); return base; } } const rollups = await channelsOnLocation({ paths: opts.paths, locations: [loc], }); const slugs = rollups[loc.id]?.slugs ?? []; const missing: string[] = []; for (const slug of slugs) { const config = await readChannelConfig(opts.paths, slug); const media = await inspectChannelMedia(opts.paths, slug, config, { fresh: true, }); if (media.status === "in-transition") { base.problems.push( `${slug}: a media relocation is in flight or was interrupted ` + `(${media.detail ?? "marker present"}). Finish or clear it first.`, ); continue; } // THE CRASH WINDOW, RECOGNISED RATHER THAN REFUSED. Between the symlink and // the config write there is one instant where the link names the new target // and config.json still names the old one. A process killed there leaves an // `inconsistent` channel that is still "on" the old root — so the next // re-point would refuse the whole location over a state whose only remedy // is the job being refused. When the link already points exactly where this // run would point it, the channel is not broken: it is half done, and the // remaining half is the config write this job performs anyway. // THE RETIRED LAYOUT (release 17) is re-pointed the retired way: its // `data` link and `dataDir`. Only a channel whose `data/` IS a link — a // recorded `dataDir` over a real `data/` is a disagreement, refused below. const legacy = media.status === "legacy"; const shape = legacy ? LEGACY_SHAPE : MEDIA_SHAPE; const resumable = await linkAlreadyAt( opts.paths, slug, shape.targetFor(newRoot, slug), shape.linkName, ); if (legacy) { const dataIsLink = await lstat(path.join(opts.paths.channelsDir, slug, "data")) .then((l) => l.isSymbolicLink()) .catch(() => false); if (!dataIsLink) { base.problems.push( `${slug}: config.json records the retired dataDir but its data/ is ` + `not a link — ${media.detail ?? "run archilyzer storage migrate-tier"}.`, ); continue; } } else if (!resumable && media.status !== "ok" && media.status !== "unreachable") { base.problems.push( `${slug}: its media reads as ${media.status} — ` + `${media.detail ?? "disk and config do not agree"}. A re-point ` + `rewrites the link and would bake that disagreement in.`, ); continue; } const busy = opts.isBusy?.(slug, "re-pointing its storage location"); if (busy) { base.problems.push(`${slug}: ${busy}`); continue; } const target = shape.targetFor(newRoot, slug); if (!(await isDirectory(target))) { missing.push(slug); continue; } const rootProblem = await relocationRootProblem({ paths: opts.paths, slug, root: newRoot, }); if (rootProblem) { base.problems.push(`${slug}: ${rootProblem}`); continue; } base.channels.push(slug); if (resumable) base.resumable.push(slug); if (legacy) base.legacy.push(slug); } // THE MISSING TARGETS ARE ONE REFUSAL, NOT n. A root that holds none of the // channels is the ordinary mistake (the wrong mountpoint, a subdirectory too // deep), and n lines each saying the same thing buries the one fact that // matters: which channels are not there. if (missing.length > 0) { base.problems.push( `${missing.length} channel(s) have no media under ${newRoot}: ` + `${missing.join(", ")}. Expected ` + `${relocatedMediaDir(newRoot, missing[0])} (or, for a channel not yet ` + `migrated, ${relocatedDataDir(newRoot, missing[0])}) and friends.`, ); } base.ok = base.problems.length === 0; return base; } // --------------------------------------------------------------------------- // The re-point itself // --------------------------------------------------------------------------- type ChannelLedgerEntry = { slug: string; // Which link and config key this channel's re-point rewrites. shape: RepointShape; oldTarget: string; newTarget: string; unlinked: boolean; relinked: boolean; configWritten: boolean; oldConfig: ChannelConfig | null; }; export type RepointResult = { locationId: string; oldRoot: string; newRoot: string; // Channels whose link and config were rewritten by THIS run. Empty on an // idempotent rerun that only had the settings write left to do. channels: string[]; settingsWritten: boolean; }; // Undo one channel's three steps, in reverse, best-effort. The state it restores // is the one the job found: a link pointing at the OLD target, which with the // disk moved reads as `unreachable`. That is the honest restore — never // `inconsistent`, which is a state move-back refuses and which a half-rolled // channel (link removed, config rewritten) would leave behind. async function rollbackChannel( paths: Paths, entry: ChannelLedgerEntry, ): Promise { if (entry.configWritten && entry.oldConfig) { // Only the field this job changed goes back; anything else edited since // stays. (It used to rewrite the whole config it found at the start.) await patchChannelConfig(paths, entry.slug, { [entry.shape.key]: entry.oldTarget, }).catch(() => {}); } if (entry.relinked || entry.unlinked) { const link = path.join(paths.channelsDir, entry.slug, entry.shape.linkName); await unlink(link).catch(() => {}); await symlink(entry.oldTarget, link).catch(() => {}); } } // THE JOB BODY. Sequential, per channel, on the renameChannel ledger pattern // (`controller/renameChannel.ts:139-201`): each step records that it RAN, and a // throw replays only what ran, in reverse — across every channel already done, // not just the one that failed. A location half re-pointed is a corpus where // some channels read from the new mountpoint and some from the old, which is // exactly the state nobody can reason about at 3am. // // The settings write is LAST and is in the ledger too: if it fails, every // channel goes back, because a location whose root still names the old path // while its channels point at the new one is the same split brain in the other // direction. // // IDEMPOTENT RERUN. A channel that was already re-pointed is no longer ON the // old root (`locationOfDataDir` reads its `mediaDir`), so the preflight does not // list it and this does not touch it. A rerun after a crash finishes the rest; // a rerun after a complete run finds no channels and refuses with "already at". export async function repointStorageLocation(opts: { paths: Paths; locationId: string; newRoot: string; io?: SettingsIO; bins?: VolumeBins; probeOptions?: ProbeOptions; isBusy?: BusyCheck; onLog?: (line: string) => void; signal?: AbortSignal; }): Promise { const io = opts.io ?? DEFAULT_SETTINGS_IO; const log = opts.onLog ?? (() => {}); const pre = await preflightRepoint({ paths: opts.paths, locationId: opts.locationId, newRoot: opts.newRoot, io, bins: opts.bins, probeOptions: opts.probeOptions, isBusy: opts.isBusy, }); if (!pre.ok) throw new Error(pre.problems.join("\n")); const settings = io.read(); const loc = settings.storage.locations.find((l) => l.id === opts.locationId); if (!loc) throw new Error(`There is no storage location "${opts.locationId}".`); log( `Re-pointing "${loc.label}" from ${pre.oldRoot} to ${pre.newRoot} — ` + `${pre.channels.length} channel(s). No bytes move: this rewrites ` + `${pre.channels.length} symlink(s), ${pre.channels.length} config ` + `field(s) and the location's root.`, ); const ledger: ChannelLedgerEntry[] = []; try { for (const slug of pre.channels) { opts.signal?.throwIfAborted(); const fresh = await readChannelConfig(opts.paths, slug); const shape = pre.legacy.includes(slug) ? LEGACY_SHAPE : MEDIA_SHAPE; const oldTarget = fresh?.[shape.key]?.trim() ?? ""; const newTarget = shape.targetFor(pre.newRoot, slug); const entry: ChannelLedgerEntry = { slug, shape, oldTarget, newTarget, unlinked: false, relinked: false, configWritten: false, oldConfig: fresh, }; ledger.push(entry); const link = path.join(opts.paths.channelsDir, slug, shape.linkName); // RESUMING SKIPS THE LINK, and must: it already points at newTarget, so // unlinking and recreating it would be two syscalls to reach the state it // is in — and a crash between them would turn a half-done channel into a // channel with no `media` at all, which is strictly worse than what we // found. The ledger records nothing for the link for the same reason: a // rollback must undo what THIS run did, and this run did not move it. const resuming = pre.resumable.includes(slug); try { if (!resuming) { await unlink(link); entry.unlinked = true; await symlink(newTarget, link); entry.relinked = true; } // A null patch means config.json vanished or became unreadable after // preflight listed the channel: fail this step, so the ledger rolls // the link back, rather than leave a link no mediaDir records. const written = await patchChannelConfig(opts.paths, slug, { [shape.key]: newTarget, }); if (!written) { throw new Error(`channels/${slug}/config.json is missing or unreadable`); } entry.configWritten = true; } catch (err) { const step = resuming ? "writing config.json (resuming a half-finished channel)" : !entry.unlinked ? "removing the old symlink" : !entry.relinked ? "creating the new symlink" : "writing config.json"; throw new Error( `${slug}: failed while ${step} — ${(err as Error).message}`, { cause: err }, ); } log( `${slug}: ${oldTarget} -> ${newTarget}` + (resuming ? " (link was already moved — finished its config)" : ""), ); } // The location last, from a FRESH read: the per-channel loop above wrote no // settings, but a concurrent action (an edit on another tab) may have, and // clobbering it with a snapshot taken before the job started would silently // undo it. const latest = io.read(); await io.write({ ...latest, storage: { ...latest.storage, locations: latest.storage.locations.map((l) => { if (l.id !== opts.locationId) return l; // The identity, re-anchored to the new root. Written here rather than // by a refresh because THIS is the moment // `root === join(mountpoint, relPath)` holds again — a refresh that // wrote it before the re-point would record a mountpoint the root is // not under. // The measurement first, the inference only when there was nothing // to measure (no findmnt, a container, a location with no recorded // uuid to probe against). const volume = pre.newVolume ?? volumeForNewRoot(l, pre.newRoot); return { ...l, root: pre.newRoot, ...(volume ? { volume } : {}), }; }), }, }); } catch (err) { for (const entry of [...ledger].reverse()) { await rollbackChannel(opts.paths, entry); } log( `Rolled back ${ledger.length} channel(s) — nothing on disk moved, and ` + `every link points where it did before this run.`, ); throw err; } resetStorageProbeMemo(); // Every channel on it has a new link and a new mediaDir: the page memo's keys // already differ, and this drops the old answers rather than letting them // age out. forgetChannelMedia(); log( `Done. "${loc.label}" is at ${pre.newRoot}; ${pre.channels.length} ` + `channel(s) re-pointed.`, ); return { locationId: opts.locationId, oldRoot: pre.oldRoot, newRoot: pre.newRoot, channels: pre.channels, settingsWritten: true, }; } // THE FALLBACK, when the preflight had no probe to take. The stored identity, // re-anchored to the new root. The uuid/fstype/label are // the disk's and do not change; the mountpoint and relPath are the halves that // do, and they are recomputed from the new root so the invariant // `root === join(mountpoint, relPath)` survives the re-point. A location with // no recorded identity gets none. function volumeForNewRoot( loc: StorageLocation, newRoot: string, ): StorageVolume | undefined { const vol = loc.volume; if (!vol) return undefined; const rel = vol.relPath; if (!rel) return { ...vol, mountpoint: newRoot, relPath: "" }; const suffix = `/${rel}`; if (newRoot.endsWith(suffix)) { return { ...vol, mountpoint: newRoot.slice(0, newRoot.length - suffix.length) || "/", relPath: rel, }; } // The new root does not end in the recorded relative path — the operator // pointed the location somewhere structurally different. Keep the uuid and // record the root itself as the mountpoint rather than inventing a // mountpoint the root is not under. return { ...vol, mountpoint: newRoot, relPath: "" }; } // --------------------------------------------------------------------------- // The job, auto re-point and the boot pass // --------------------------------------------------------------------------- // ONE ENQUEUE, for the three callers that have one: the /storage button, the // Refresh action's auto path, and the boot pass. Same reason // `editor/app/channels/lib/relocationJob.ts` exists — what must not drift // between them is the job record's shape, not the guards, which are the // caller's. // // The queue key is `relocationQueueKey()`, SHARED WITH THE RELOCATE JOB on // purpose: a re-point rewrites the same links a move is in the middle of // rewriting, and the registry caps a key at concurrency 1. One at a time, // across both kinds. export async function enqueueRepoint(opts: { paths: Paths; locationId: string; newRoot: string; bins?: VolumeBins; isBusy?: BusyCheck; io?: SettingsIO; // Called after the body finishes, inside the job. The editor passes its // revalidatePath calls here; nothing else has pages to invalidate. afterDone?: () => void; }): Promise { return runManagedFunction({ kind: "repoint-storage-location", queueKey: relocationQueueKey(), paths: opts.paths, fn: async (onLog, signal) => { await repointStorageLocation({ paths: opts.paths, locationId: opts.locationId, newRoot: opts.newRoot, bins: opts.bins, isBusy: opts.isBusy, io: opts.io, onLog, signal, }); opts.afterDone?.(); }, }); } export type AutoRepointOutcome = | { started: true; jobId?: string; newRoot: string } | { started: false; reason: string }; // The opt-in. `autoRepoint` is per-location and off by default, because a // re-point rewrites every channel symlink on the location and that is not // something to do silently unless the operator asked for it. // // Refused unless ALL of: the flag is on, the volume is mounted somewhere else, // the probe named a candidate root, and the full preflight passes. The last one // is what makes this safe to call from a boot hook: an auto re-point that // cannot be done cleanly declines with a reason and leaves the manual button // exactly where it was. export async function maybeAutoRepoint(opts: { paths: Paths; location: StorageLocation; probe: StorageLocationProbe; bins?: VolumeBins; isBusy?: BusyCheck; io?: SettingsIO; enqueue?: boolean; afterDone?: () => void; }): Promise { const loc = opts.location; if (!loc.autoRepoint) return { started: false, reason: "auto re-point is off" }; if (opts.probe.status !== "mounted-elsewhere") { return { started: false, reason: `status is ${opts.probe.status}` }; } const candidate = opts.probe.candidateRoot; if (!candidate) { return { started: false, reason: "the probe named no candidate root" }; } const pre = await preflightRepoint({ paths: opts.paths, locationId: loc.id, newRoot: candidate, io: opts.io, bins: opts.bins, isBusy: opts.isBusy, }); if (!pre.ok) return { started: false, reason: pre.problems.join("; ") }; if (opts.enqueue === false) { return { started: false, reason: "enqueue is disabled (idle boot)" }; } const result = await enqueueRepoint({ paths: opts.paths, locationId: loc.id, newRoot: candidate, bins: opts.bins, isBusy: opts.isBusy, io: opts.io, afterDone: opts.afterDone, }); if (!result.ok) return { started: false, reason: result.error }; return { started: true, jobId: result.jobId, newRoot: candidate }; } export type BootPassResult = { probed: number; started: string[]; declined: Array<{ locationId: string; reason: string }>; }; // ONE PASS AT BOOT: probe every location, and for the ones the operator armed, // re-point to where the volume actually came up. // // `enqueue: false` — the idle boot (`ARCHILYZER_IDLE_BOOT`) — still PROBES. The // probe is read-only and its whole cost is a findmnt per location, and the memo // it fills is what makes the operator's first page load fast. What idle boot // refuses is starting work, and a re-point is work. export async function runStorageBootPass(opts: { enqueue: boolean; paths?: Paths; bins?: VolumeBins; isBusy?: BusyCheck; io?: SettingsIO; }): Promise { const io = opts.io ?? DEFAULT_SETTINGS_IO; const settings = io.read(); const locations = settings.storage.locations; const out: BootPassResult = { probed: 0, started: [], declined: [] }; if (locations.length === 0) return out; const paths = opts.paths ?? getPaths(); const bins = opts.bins ?? paths; for (const loc of locations) { const probe = await probeLocationMemo(loc, bins); out.probed += 1; if (probe.identity.known) { await recordProbedIdentity({ locationId: loc.id, probe, io }).catch( () => {}, ); } const outcome = await maybeAutoRepoint({ paths, location: loc, probe, bins, isBusy: opts.isBusy, io, enqueue: opts.enqueue, }); if (outcome.started) out.started.push(loc.id); else out.declined.push({ locationId: loc.id, reason: outcome.reason }); } return out; }