commit b245b936312a4ecb56a0851334cbb815a0986d9c
parent c10c87a8b613d072a1ccfb2fe900fb13b1555120
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Sat, 26 Sep 2026 03:26:28 -0400
jobs: say what a hung mount does to the boot pass's re-queues (review fix)
The comment and the timeout line promised that a job re-queued for a
channel the storage pass had not reached is refused at once and /jobs says
so. True for an unmounted drive (the media guard's stat gets ENOENT); not
for the hung mount the 60 s bound exists for: the guard's own stat
(lib/channelMedia.ts) hangs on the same syscall, and since re-queues run
one at a time, the ones after it stay `queued` until the mount answers.
Cancels have all run by then and re-queues are few, so it is recorded as
found and left. No behaviour change.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
1 file changed, 17 insertions(+), 9 deletions(-)
diff --git a/common/jobs/bootQueuedJobs.ts b/common/jobs/bootQueuedJobs.ts
@@ -110,13 +110,21 @@ export type SettleQueuedOpts = {
// 60 s covers several locations at that worst case; past it the pass is stuck on
// a syscall that will not answer, and waiting longer buys nothing.
//
-// On a timeout the settle runs anyway, and the cost is bounded: a job re-queued
-// for a channel whose drive the pass has not re-pointed yet meets the media
-// guard (jobs/jobKinds.ts `needsMedia`) — refused at submission, which closes
-// the old meta `cancelled` with the error, or failed when it starts, with the
-// reason in its log. Either way /jobs says so, and Retry is one click. The
-// storage pass is NOT cancelled (it has no signal, and a re-point it already
-// enqueued must finish): it runs on, and a line says when it finally ends.
+// On a timeout the settle runs anyway. What a job re-queued for a channel the
+// pass has not reached then meets depends on why the pass is slow:
+// - an UNMOUNTED drive (the root is simply absent): the media guard
+// (jobs/jobKinds.ts `needsMedia`, lib/channelMedia.ts) gets ENOENT at
+// once, so the job is refused — at submission, which closes the old meta
+// `cancelled` with the error, or when it starts, with the reason in its
+// log. /jobs says so, and Retry is one click.
+// - a HUNG mount, the case this bound exists for: the guard's own `stat`
+// (lib/channelMedia.ts) hangs on the same syscall, so that job waits with
+// it. Re-queues run one at a time (below), so every re-queue after it stays
+// `queued` until the mount answers. Every cancel has run by then — cancels
+// come first — and re-queues are few (one at the first live boot), so this
+// is left, not fixed: see the release 10 record, L2 "found and left".
+// The storage pass is NOT cancelled (it has no signal, and a re-point it
+// already enqueued must finish): it runs on, and a line says when it ends.
export const STORAGE_PASS_WAIT_MS = 60_000;
export async function waitForStoragePass(
@@ -147,8 +155,8 @@ export async function waitForStoragePass(
const secs = (ms: number) => `${Math.round(ms / 1000)} s`;
log(
`[boot] storage pass still running after ${secs(timeoutMs)}; settling queued jobs without it ` +
- `(a hung mount? a job re-queued for a channel it has not re-pointed yet ` +
- `will be refused as unreachable, and /jobs will say so)`,
+ `(a hung mount? a job re-queued for a channel on it will wait on the same mount, ` +
+ `and the re-queues after it with it)`,
);
void finished.then(() =>
log(