Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 888034d132a6e812a53185a90d009ca909e6fde8
parent ca3d6ea1fd8df84330ca4ae19e092cffa09f617f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Thu, 17 Sep 2026 13:44:21 -0400

storage: the probe answers with stat, and only ever adds identity

probeLocation is five answers — available / mounted-elsewhere / unmounted /
absent / missing — and exactly one of them comes from a subprocess.
AVAILABILITY IS `stat`: a root that is a directory is available even when
every identity probe fails. That is not defensive coding, it is Docker: block
devices are invisible in a container and the media root is an identity bind
mount, so findmnt legitimately knows nothing about a healthy root. A probe
that read "I could not ask" as "your disk is gone" would declare every
containerised corpus broken.

So every subprocess is execa + reject:false + a timeout + a catch, and every
failure lands on identity "unknown". freeBytes is measured only when
available — getFreeBytes walks up on ENOENT, so on an unmounted platter it
would report the free space of the disk holding /run/media.

mountByUuid is the one call that changes the machine, offered only when the
volume is attached-but-unmounted and udisksctl resolves (--version, memoised
per binary). Never retried: a polkit denial under a service session is the
expected failure and a second attempt is just a second denial, so its words
are surfaced verbatim instead.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Acommon/lib/storageLocations.test.ts | 114+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/storageVolumes.ts | 329+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 443 insertions(+), 0 deletions(-)

diff --git a/common/lib/storageLocations.test.ts b/common/lib/storageLocations.test.ts @@ -0,0 +1,114 @@ +import test from "node:test"; +import assert from "node:assert/strict"; +import { + defaultLocationRoot, + locationOfDataDir, + migrateMediaRootToLocations, + type StorageLocation, +} from "./storageLocations"; + +function loc(id: string, root: string): StorageLocation { + return { id, label: id, root, autoRepoint: false }; +} + +const PLATTER = loc("platter", "/mnt/platter"); +const ARCHIVE = loc("archive", "/mnt/platter/archive"); + +test("a channel is on the location its dataDir is under", () => { + assert.equal( + locationOfDataDir("/mnt/platter/alpha/data", [PLATTER])?.id, + "platter", + ); + assert.equal(locationOfDataDir("/corpus/channels/alpha/data", [PLATTER]), null); + assert.equal(locationOfDataDir("", [PLATTER]), null); + // A sibling whose name merely starts the same is not under it. + assert.equal(locationOfDataDir("/mnt/platter-old/alpha/data", [PLATTER]), null); + // "Under" is strict: the root itself is not a channel's dataDir. + assert.equal(locationOfDataDir("/mnt/platter", [PLATTER]), null); +}); + +test("nested roots: the longest match wins, whatever the list order", () => { + const deep = "/mnt/platter/archive/alpha/data"; + assert.equal(locationOfDataDir(deep, [PLATTER, ARCHIVE])?.id, "archive"); + assert.equal(locationOfDataDir(deep, [ARCHIVE, PLATTER])?.id, "archive"); + // Still the outer one for a channel that is not in the nested root. + assert.equal( + locationOfDataDir("/mnt/platter/beta/data", [PLATTER, ARCHIVE])?.id, + "platter", + ); +}); + +test("trailing slashes on either side are one root", () => { + assert.equal( + locationOfDataDir("/mnt/platter/alpha/data/", [loc("p", "/mnt/platter/")]) + ?.id, + "p", + ); +}); + +test("defaultLocationRoot resolves the id, or blanks", () => { + assert.equal( + defaultLocationRoot({ + locations: [PLATTER, ARCHIVE], + defaultLocationId: "archive", + }), + "/mnt/platter/archive", + ); + // A default naming nothing (an empty list, or an id the sanitizer would have + // repaired) is "no default root", which every caller already handles. + assert.equal( + defaultLocationRoot({ locations: [], defaultLocationId: "" }), + "", + ); + assert.equal( + defaultLocationRoot({ locations: [PLATTER], defaultLocationId: "gone" }), + "", + ); +}); + +test("migration rule 1: a file that already spells locations is untouched", () => { + const already = { + locations: [{ id: "cold", label: "Cold", root: "/mnt/cold" }], + defaultLocationId: "cold", + // A stale mediaRoot beside it is NOT merged in as a second location. + mediaRoot: "/mnt/deleted", + }; + assert.deepEqual(migrateMediaRootToLocations(already), already); + // An EMPTY list is also "already spelled" — the operator deleted them all. + const empty = { locations: [], defaultLocationId: "", mediaRoot: "/mnt/x" }; + assert.deepEqual(migrateMediaRootToLocations(empty), empty); +}); + +test("migration rule 2: blank, missing or relative mediaRoot is no location", () => { + const none = { locations: [], defaultLocationId: "" }; + assert.deepEqual(migrateMediaRootToLocations({}), none); + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: "" }), none); + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: " " }), none); + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: 7 }), none); + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: "platter" }), none); + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: "../up" }), none); + // Not an object at all: handed straight back for the sanitizer to default. + assert.equal(migrateMediaRootToLocations(undefined), undefined); + assert.equal(migrateMediaRootToLocations("/mnt/platter"), "/mnt/platter"); +}); + +test("migration rule 3: an absolute mediaRoot becomes the default location", () => { + assert.deepEqual(migrateMediaRootToLocations({ mediaRoot: "/mnt/platter" }), { + locations: [ + { + id: "default", + label: "Default", + root: "/mnt/platter", + // Never armed by a migration: re-point rewrites every channel symlink + // on the location and nobody asked for that. + autoRepoint: false, + }, + ], + defaultLocationId: "default", + }); +}); + +test("migration is idempotent", () => { + const once = migrateMediaRootToLocations({ mediaRoot: "/mnt/platter" }); + assert.deepEqual(migrateMediaRootToLocations(once), once); +}); diff --git a/common/lib/storageVolumes.ts b/common/lib/storageVolumes.ts @@ -0,0 +1,329 @@ +import path from "node:path"; +import { lstat, stat } from "node:fs/promises"; +import { execa } from "execa"; +import type { Paths } from "./paths"; +import { getFreeBytes } from "./diskSpace"; +import type { StorageLocation, StorageVolume } from "./storageLocations"; + +// STORAGE VOLUME PROBES — is this location's disk here, and if not, where? +// +// SERVER-ONLY. It imports execa, so nothing reachable from a `"use client"` +// file may import it (next build fails on `node:child_process`). The pure half +// — types, `locationOfDataDir`, the migration — is `storageLocations.ts`, and +// that is the one a client component may reach for. +// +// EVERY SUBPROCESS HERE FAILS OPEN. `reject: false`, a timeout, and a catch, +// and on any failure the answer is "identity unknown" — never "unreachable". +// The reason is Docker: block devices are invisible inside a container and the +// media root is an identity bind-mount, so `findmnt` can legitimately know +// nothing about a perfectly healthy root (RUNNING_IN_DOCKER.md §another drive). +// A probe that turned "I could not ask" into "your disk is gone" would declare +// every containerised corpus broken. Availability comes from `stat`, which is +// the one thing that is always true; the subprocesses only ever ADD identity. + +export type StorageLocationStatus = + // The root is a directory right now. The only status with freeBytes. + | "available" + // The root is not there, but the recorded UUID is mounted somewhere else — + // `candidateRoot` is where the root would be after a re-point. + | "mounted-elsewhere" + // The recorded UUID has a /dev/disk/by-uuid node but is not mounted. + | "unmounted" + // The recorded UUID is not present on this machine at all. + | "absent" + // The root is not there and we have no identity to look for — nothing to say + // beyond "that path does not exist". + | "missing"; + +// `known: false` is the fail-open answer and is NOT a problem report: it means +// the probe could not ask (no findmnt, a container, a timeout), not that the +// root is in a bad state. +export type StorageIdentity = + | ({ known: true } & StorageVolume) + | { known: false }; + +export type StorageLocationProbe = { + status: StorageLocationStatus; + identity: StorageIdentity; + // Only for "mounted-elsewhere": join(currentMountpoint, volume.relPath). + candidateRoot?: string; + // Only when available and the mount looks like it will not survive a reboot. + warning?: string; + // Only when available — `getFreeBytes` walks up on ENOENT, so on a missing + // root it would cheerfully report the PARENT volume's free space, which for + // an unmounted platter is the free space of the disk holding /run/media. + freeBytes?: number; +}; + +export type VolumeBins = Pick<Paths, "findmntBin" | "udisksctlBin">; + +// The plan's budget: a findmnt that has not answered in 3 s has hit a wedged +// automounter or a hung NFS mount, and the right answer is "unknown", now. +export const FINDMNT_TIMEOUT_MS = 3_000; +// udisksctl talks to a daemon over D-Bus and then waits for a real mount. +export const MOUNT_TIMEOUT_MS = 15_000; + +// Test seam ONLY. Production callers pass nothing and get the constants above; +// the unit tests shorten the findmnt timeout so the "fake binary that sleeps" +// case does not cost the suite three seconds. +export type ProbeOptions = { findmntTimeoutMs?: number }; + +type Run = { ok: boolean; stdout: string; stderr: string }; + +async function run( + bin: string, + args: string[], + timeout: number, +): Promise<Run> { + try { + const res = await execa(bin, args, { + buffer: true, + reject: false, + timeout, + }); + return { + ok: res.exitCode === 0 && !res.timedOut, + stdout: typeof res.stdout === "string" ? res.stdout : "", + stderr: typeof res.stderr === "string" ? res.stderr : "", + }; + } catch { + // ENOENT on the binary itself lands here in some execa paths. Same answer. + return { ok: false, stdout: "", stderr: "" }; + } +} + +// `findmnt -J -T <path>` — the mount the path is ON (-T resolves a path, not +// just a mountpoint), as JSON so a label with a space cannot be misparsed. +async function identityOfPath( + root: string, + bins: VolumeBins, + timeoutMs: number, +): Promise<StorageIdentity> { + const res = await run( + bins.findmntBin, + ["-J", "-T", root, "-o", "TARGET,SOURCE,FSTYPE,LABEL,UUID"], + timeoutMs, + ); + if (!res.ok) return { known: false }; + let parsed: unknown; + try { + parsed = JSON.parse(res.stdout); + } catch { + return { known: false }; + } + const fs0 = (parsed as { filesystems?: unknown[] })?.filesystems?.[0] as + | Record<string, unknown> + | undefined; + if (!fs0) return { known: false }; + const uuid = typeof fs0.uuid === "string" ? fs0.uuid : ""; + const mountpoint = typeof fs0.target === "string" ? fs0.target : ""; + // No mountpoint means we learned nothing usable; no UUID means the volume has + // no stable name to find it by later (tmpfs, overlay, a bind mount in a + // container) — in both cases identity stays unknown rather than half-filled. + if (!uuid || !mountpoint) return { known: false }; + return { + known: true, + uuid, + fstype: typeof fs0.fstype === "string" ? fs0.fstype : undefined, + label: typeof fs0.label === "string" ? fs0.label : undefined, + mountpoint, + relPath: relativeUnder(mountpoint, root), + }; +} + +// The root's path relative to its mountpoint, "" when they are the same dir. +// Never "..": if `root` is somehow not under `mountpoint` we keep "" rather +// than inventing a traversal that a later join would follow off the volume. +function relativeUnder(mountpoint: string, root: string): string { + const rel = path.relative(mountpoint, root); + if (rel === "" || rel.startsWith("..") || path.isAbsolute(rel)) return ""; + return rel; +} + +// `findmnt -rn -S UUID=<u> -o TARGET` — where that volume is mounted now, if +// anywhere. Raw + no headings, one mountpoint per line; we take the first. +async function mountpointOfUuid( + uuid: string, + bins: VolumeBins, + timeoutMs: number, +): Promise<string> { + const res = await run( + bins.findmntBin, + ["-rn", "-S", `UUID=${uuid}`, "-o", "TARGET"], + timeoutMs, + ); + if (!res.ok) return ""; + const first = res.stdout + .split("\n") + .map((l) => l.trim()) + .find((l) => l !== ""); + return first ?? ""; +} + +// `findmnt --fstab -S UUID=<u>` exits 1 when the volume has no fstab entry. +// FAILS OPEN TO "yes, it is in fstab": a missing findmnt must not produce a +// warning on every location on the page. +async function isInFstab( + uuid: string, + bins: VolumeBins, + timeoutMs: number, +): Promise<boolean> { + const res = await run( + bins.findmntBin, + ["--fstab", "-S", `UUID=${uuid}`], + timeoutMs, + ); + return res.ok ? res.stdout.trim() !== "" : true; +} + +function automountWarning( + loc: StorageLocation, + identity: StorageIdentity, +): string { + const uuid = identity.known ? identity.uuid : "<uuid>"; + const fstype = (identity.known && identity.fstype) || "auto"; + return ( + `automount — may not be present at boot; add ` + + `\`UUID=${uuid} /mnt/${loc.id} ${fstype} nofail 0 2\` to /etc/fstab, ` + + `or rely on re-point` + ); +} + +// Probe one location. Read-only: it never mounts, never writes settings, and +// never touches the corpus. +export async function probeLocation( + loc: StorageLocation, + bins: VolumeBins, + opts: ProbeOptions = {}, +): Promise<StorageLocationProbe> { + const timeoutMs = opts.findmntTimeoutMs ?? FINDMNT_TIMEOUT_MS; + const root = loc.root.trim(); + + // AVAILABILITY IS `stat`, AND ONLY `stat`. A root that is a directory is + // available even when every identity probe below fails — see the header. + let isDir = false; + if (root !== "") { + try { + isDir = (await stat(root)).isDirectory(); + } catch { + isDir = false; + } + } + + if (isDir) { + const identity = await identityOfPath(root, bins, timeoutMs); + const freeBytes = await getFreeBytes(root); + // Two ways a mount will not be there after a reboot: udisks put it under + // /run/media (or /media) because a human plugged it in, or there is no + // fstab entry naming its UUID. The fstab call is skipped when the + // mountpoint already says "automount" — same warning, one less subprocess. + let warning: string | undefined; + if (identity.known) { + const automounted = + identity.mountpoint.startsWith("/run/media/") || + identity.mountpoint.startsWith("/media/"); + if (automounted || !(await isInFstab(identity.uuid, bins, timeoutMs))) { + warning = automountWarning(loc, identity); + } + } + return { + status: "available", + identity, + ...(warning ? { warning } : {}), + freeBytes, + }; + } + + // The root is not a directory (absent, or — treated identically — a file). + // Without a recorded UUID there is nothing to look for. + const uuid = loc.volume?.uuid?.trim() ?? ""; + if (!uuid) return { status: "missing", identity: { known: false } }; + + const target = await mountpointOfUuid(uuid, bins, timeoutMs); + if (target) { + const relPath = loc.volume?.relPath ?? ""; + return { + status: "mounted-elsewhere", + // A live sighting of the recorded volume: uuid and mountpoint come from + // findmnt, fstype/label are carried over from the record (this query asks + // for TARGET only). This is what the re-point job writes back as the + // location's `volume`. + identity: { + known: true, + uuid, + fstype: loc.volume?.fstype, + label: loc.volume?.label, + mountpoint: target, + relPath, + }, + candidateRoot: relPath ? path.join(target, relPath) : target, + }; + } + + // Not mounted. Is the disk even attached? `/dev/disk/by-uuid/<u>` is a + // symlink the kernel maintains; lstat it so a dangling link still counts as + // "the udev entry is there". + try { + await lstat(path.join("/dev/disk/by-uuid", uuid)); + return { status: "unmounted", identity: { known: false } }; + } catch { + return { status: "absent", identity: { known: false } }; + } +} + +// udisksctl availability, memoised per binary path. Same shape as a digest +// app's probe (`digestApps.ts` claudeCode.probe): `--version`, reject:false, +// short timeout. Memoised because /storage asks once per render and the answer +// does not change while the process lives. +const udisksctlProbes = new Map<string, Promise<boolean>>(); + +export function udisksctlAvailable(bins: VolumeBins): Promise<boolean> { + const bin = bins.udisksctlBin; + const hit = udisksctlProbes.get(bin); + if (hit) return hit; + const probe = run(bin, ["--version"], FINDMNT_TIMEOUT_MS).then((r) => r.ok); + udisksctlProbes.set(bin, probe); + return probe; +} + +// Test seam: the memo above outlives a test's fake bin otherwise. +export function resetUdisksctlProbeCache(): void { + udisksctlProbes.clear(); +} + +export type MountResult = { + ok: boolean; + // The mountpoint udisksctl reported, when it said one. + mountpoint?: string; + // udisksctl's own words. Surfaced verbatim — a polkit denial under a service + // session is the expected failure and the operator needs to read it. + error?: string; +}; + +// Mount a volume by UUID. Offered only for an `unmounted` location, and only +// when the binary resolves. NEVER RETRIED: a failure here is a policy or +// hardware answer, and a second attempt just produces a second denial. +export async function mountByUuid( + uuid: string, + bins: VolumeBins, +): Promise<MountResult> { + if (!(await udisksctlAvailable(bins))) { + return { + ok: false, + error: `udisksctl not found (set UDISKSCTL_BIN) — mount ${uuid} by hand`, + }; + } + const res = await run( + bins.udisksctlBin, + ["mount", "-b", path.join("/dev/disk/by-uuid", uuid)], + MOUNT_TIMEOUT_MS, + ); + if (!res.ok) { + const said = res.stderr.trim() || res.stdout.trim(); + return { ok: false, error: said || `udisksctl mount failed for ${uuid}` }; + } + // "Mounted /dev/sdb1 at /run/media/user/<uuid>." — the trailing period is + // part of the message and not part of the path. + const m = /\bat\s+(.+?)\.?\s*$/m.exec(res.stdout.trim()); + return { ok: true, mountpoint: m ? m[1] : undefined }; +}