Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit ed9c10c00e135db5579540babdeaee60fcb9626e
parent aaaf0919c9f81766944094e69070f73061fe36c3
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon,  7 Sep 2026 20:43:44 -0400

common: a runner's run is its whole lifetime, and the spend cap has to know that

Three fixes to the long-lived half of slice 1.2, all found by reading the run
context back against what "per-RUN state" means when the run never ends.

**The metered spend cap was resettable by a clock.** The runner rebuilds its
operation-run context when the digest/backfill/diarization/attribution settings
change or on a sixty-second TTL — and `openOperationRun` starts a fresh context
with `costUsd: 0`. So a `$5` per-run cap would have become `$5 a minute` on the
metered lane, silently, with the log still printing a cap it was no longer
enforcing. The accumulator and the disk-floor latch now survive the refresh;
they are per-RUN, and a runner's run is its whole lifetime.

**A refresh could swap the context out from under an in-flight unit.** Its llm
and unit-executor leases decrement counters on the object it was dispatched
with, so replacing that object mid-flight leaks a slot from `laneLimit`'s point
of view — permanently, because nothing ever decrements it again. The refresh
now waits for the lane to be idle.

**A dead engine was re-probed every three seconds.** The TTL covered a
successful open and not a failed one, so a lane idling at `engine-unreachable`
re-asked the probe on every poll. It covers both now; the cost is up to a minute
of idling after the engine comes back, and the job log already says which state
it is in.

Two smaller ones beside them. The batch's resume line said "Transcription
finished; resuming the digest lane" whatever the hold had been — so lifting an
operator pause reported something that never happened, which is the same class
of mistake the hold messages exist to prevent; it names the lane and only claims
transcription when the hold was the yield. And `countOperationWork` resolved
each operation's freshness target inside the per-video loop rather than once per
operation, which is the exact shape its own header warns about.

tsc clean in six packages, common 896, mcp 205, numbers diff empty.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/autoRunner.ts | 31++++++++++++++++++++++---------
Mcommon/controller/operationBatch.ts | 39+++++++++++++++++++++++++++------------
2 files changed, 49 insertions(+), 21 deletions(-)

diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -818,18 +818,23 @@ async function runLoop( settings.attribution, ]); - const refreshLaneRun = async (force = false): Promise<void> => { + const refreshLaneRun = async (): Promise<void> => { if (!isOperationLane(kind) || laneRun.refreshing) return; const settings = getSettings(); const key = laneRunKey(settings); - if ( - !force && - laneRun.run && - laneRun.key === key && - Date.now() - laneRun.at < LANE_RUN_TTL_MS - ) { - return; - } + // The TTL covers a FAILED probe too, or a dead ollama would be re-asked on + // every three-second poll. The log line already says the lane is idle and + // why; the cost of the TTL is up to a minute of idling after the engine + // comes back. + const fresh = + laneRun.key === key && Date.now() - laneRun.at < LANE_RUN_TTL_MS; + if (fresh && (laneRun.run !== null || laneRun.error !== null)) return; + // NEVER SWAP THE CONTEXT OUT FROM UNDER AN IN-FLIGHT UNIT. Its llm and + // unit-executor leases decrement counters on the object it was dispatched + // with, so replacing that object mid-flight would leak a slot from the + // limit's point of view — permanently, since nothing ever decrements it + // again. + if (laneRun.run && live.inFlight.size > 0) return; laneRun.refreshing = true; try { const opened = await openOperationRun({ @@ -839,6 +844,14 @@ async function runLoop( ...(clusterPlan ? { clusterPlan } : {}), }); clusterPlan = opened.digest?.clusterPlan ?? clusterPlan; + // THE PER-RUN ACCUMULATORS SURVIVE THE REFRESH, and they have to: the + // metered spend cap is a per-RUN ceiling, and a runner's run is its whole + // lifetime — resetting the total every sixty seconds would turn a $5 cap + // into $5 a minute. The disk-floor latch is the same shape. + if (laneRun.run) { + opened.live.costUsd = laneRun.run.live.costUsd; + opened.live.diskFloorHit = laneRun.run.live.diskFloorHit; + } // The engine fail-fast, asked ONCE per context rather than per video: an // unreachable ollama would otherwise produce one failure per candidate // over 55,956 of them. A refusal is an IDLE, not a stop — it is a diff --git a/common/controller/operationBatch.ts b/common/controller/operationBatch.ts @@ -1384,6 +1384,22 @@ export async function countOperationWork( // Folded through the SHARED counters rather than a private if-chain, so this // path and the snapshot path can no longer disagree about what a state means, // and a state added later is counted here without anyone remembering. + // Resolved ONCE per operation, not per video: the identity is a settings read + // plus some string work, and deriving it per item is how a counter and a + // runner end up disagreeing about what is stale. + const targets = new Map<string, unknown>(); + for (const op of run.operations) { + if (run.digest) { + const resolved = await resolveDigestFor(run, channelSlug); + targets.set(op.id, { + target: resolved.target, + sections: resolved.sections, + }); + } else { + targets.set(op.id, await resolveTargetFor(run, op, channelSlug)); + } + } + const counts = emptyOperationCounts(); for (const id of dirs) { const videoDir = path.join(dataDir, id); @@ -1391,19 +1407,13 @@ export async function countOperationWork( checkUntranscribable: true, }); for (const op of run.operations) { - const target = run.digest - ? await (async () => { - const r = await resolveDigestFor(run, channelSlug); - return { target: r.target, sections: r.sections }; - })() - : await resolveTargetFor(run, op, channelSlug); addOperationState( counts, await op.state({ videoDir, videoId: id, files, - target, + target: targets.get(op.id), settings: run.settings, }), ); @@ -1777,6 +1787,7 @@ export async function runOperationBatch( // a lane that goes from yielding to paused has changed state and should say // so, where a boolean would stay `true` and stay silent. let heldReason: string | null = null; + const laneWord = run.digest ? "digest" : "backfill"; const NEVER = new AbortController().signal; // Seed the bar before the first video finishes, so a job that spends its @@ -1804,12 +1815,16 @@ export async function runOperationBatch( return verdict.limit; } if (heldReason !== null) { + // NAMES WHAT ENDED, not just that something did. "Transcription + // finished" after an operator lifted a pause would send them to look at + // the transcription queue for something that never happened, which is + // the same class of mistake the hold messages exist to prevent. + const resumed = + heldReason === "yield" + ? `Transcription finished; resuming the ${laneWord} lane.` + : `Resuming the ${laneWord} lane.`; heldReason = null; - log( - run.digest - ? "Transcription finished; resuming the digest lane." - : "Resuming the backfill lane.", - ); + log(resumed); } return verdict.limit; },