commit 848723b88c51913f389ff343d12345fae0a56260
parent a79b3c4e8cf8be94549edc127af6572d4b335573
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 24 Aug 2026 10:22:59 -0400
sweep console: each operation says where it can run
A read-only "also runs on …" line per scope row (and on the digest panel,
which renders no scope control): the delegate workers — LLM endpoints and
tagged unit executors — whose tags match the operation, folded to one row per
configured worker with slot counts, derived through the same workerMatches
rule the pool grants by so the display and the routing cannot disagree. Never
persisted into the scope; delegation stays a per-item dispatch decision. The
loud state is capacity configured with none of it available (all degraded or
disabled), which otherwise reads as "distributed" from the arm button.
sweepPlan/sweepPreview and the arm actions are untouched.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Diffstat:
4 files changed, 108 insertions(+), 1 deletion(-)
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **The sweep console now says where each operation will run.** Each row of the scope control gains a read-only line naming the delegate machines that can take that operation — LLM endpoints for the call-bound kinds, tagged executors for units, with their slot counts — and the digest lane, which has no operation list, gets the same line on its panel. It is derived through the same matching rule the scheduler grants by, so the display and the routing cannot disagree, and it is deliberately not part of the scope: delegation is decided per item at dispatch, not persisted at arm time. The one loud state: remote capacity is configured and none of it can currently take work (everything degraded or disabled) — which from the arm button looks identical to "distributed" and would otherwise be discovered from a week of single-machine throughput.
- **The backfill lane now dispatches units to those executors, and the lane's ceiling rises to match.** For every candidate the runner first tries a free slot on a *tagged* remote worker whose tags cover the operation (tagging is the opt-in: an untagged remote from before this protocol keeps doing transcription only, rather than being shipped envelopes an older build answers with errors). A shipped unit comes back as records, applied through the same guarded writers a local run uses — so a unit's text-only speaker record still loses to a diarized one that landed while it was in flight. A network failure is charged to the *worker*, never the work: the item retries on another executor or falls back to running locally, and a machine that stops answering is health-checked and benched, exactly as remote transcription already does. The lane's concurrency becomes local + remote slots — with the remote share counted at the *minimum* across the run's operations, so on a mixed run a slot justified by one operation's remote capacity can never push a different operation onto this box while it is meant to be standing aside. Digest deliberately does not ship as units: its distribution is the LLM-endpoint fan-out, which reaches the same ollama with less machinery.
- **A machine with no corpus can now run whole backfill units for this one.** The remote-transcription protocol grew a general sibling: `/api/worker/unit` accepts one unit of any *backfill kind* — speaker attribution, diarization — as a small envelope of input files, runs it against a throwaway scratch corpus, and hands the produced sidecar back. Each kind now declares its own contract: what a unit needs (attribution ships the cue sidecar, metadata and raw transcript — the freshness gate compares their mtimes, so the executor writes the cues file *last* or the unit would silently do nothing), what it produces, and how the result lands back on the primary — always through the guarded writers, never a raw copy, so a unit's text-only record still cannot overwrite a diarized one and applying one digest section still preserves the other. Two refusals are load-bearing: the endpoint takes only backfill kinds (downloads and transcription are refused at the door, so download politeness stays one machine's promise), and a unit that reports "disabled" or "not configured" on the executor is an *error* — on a bare box that answer means the primary's injected model/prompt identity was dropped and the executor's default settings leaked in, which is exactly how a second machine writes permanently-stale records. Deploying an executor needs no new software: this app, `ARCHILYZER_IDLE_BOOT=1`, `WORKER_TOKEN`, and an empty transcripts dir — see RUNNING_IN_DOCKER.md.
- **A second machine can now carry the AI backlog by running nothing but `ollama serve`.** The digest and speaker-attribution sweeps — about 194,000 and 78,000 model calls on this archive — bottom out in exactly one HTTP call per chunk; everything around that call (chunking, prompts, parsing, the guarded sidecar writes) is cheap and stays on this box. So the new **LLM endpoint** worker kind is just a URL: no repo, no editor, no token, no copy of the corpus on the other machine. The digest and attribution runners fan their calls across every free endpoint slot alongside the local one, and the lanes' limits rise to match — including while the digest lane is yielding the GPU to transcription, when the *local* term goes to zero and remote endpoints keep the lane moving (an operator pause and the metered spend cap still stop everything; intent and money are global). With no endpoints configured, nothing changes at all.
diff --git a/editor/app/auto-queue/components/SweepLane.tsx b/editor/app/auto-queue/components/SweepLane.tsx
@@ -24,7 +24,7 @@ import type { SweepLaneStatus } from "../lanes";
import type { OperationBand } from "../../components/pipelines/band";
import { OrderReach } from "./OrderReach";
import { SweepPlan } from "./SweepPlan";
-import { SweepScope } from "./SweepScope";
+import { SweepScope, WhereItRuns } from "./SweepScope";
import { formatElapsed } from "./dispatch";
// A sweep-fed lane: digest or backfill.
@@ -233,6 +233,14 @@ export function SweepLane({
)}
<LaneFigures lane={lane} band={band} />
+ {/* A single-operation lane (digest) renders no scope control, so its
+ "where it runs" line — the LLM endpoints its calls fan out across —
+ lives here instead. Multi-operation lanes carry it per row inside
+ SweepScope. */}
+ {lane.operations.length < 2 &&
+ lane.operations.map((op) => (
+ <WhereItRuns key={op.id} op={op} className="block" />
+ ))}
<InFlight lane={lane} />
<OrderReach
diff --git a/editor/app/auto-queue/components/SweepScope.tsx b/editor/app/auto-queue/components/SweepScope.tsx
@@ -90,6 +90,11 @@ export function SweepScope({
<span className="text-xs text-muted-foreground">
{op.costBasis}
</span>
+ {/* WHERE IT RUNS — read-only, never part of the scope.
+ Delegation is decided per item at dispatch by the worker
+ pool; this row only reports which configured endpoints and
+ executors match, through the same rule the pool grants by. */}
+ <WhereItRuns op={op} className="w-full pl-6" />
</label>
</li>
);
@@ -111,3 +116,39 @@ export function SweepScope({
</div>
);
}
+
+// Which delegate workers can take an operation's items, beside this machine.
+// Rendered only when delegates are configured — the lane always runs here, and
+// stating that on every row would be noise. The one loud case: capacity is
+// configured and NONE of it can currently take work (degraded/disabled), which
+// looks identical to "distributed" from the arm button.
+export function WhereItRuns({
+ op,
+ className = "",
+}: {
+ op: SweepOperation;
+ className?: string;
+}) {
+ if (op.runsOn.length === 0) return null;
+ const noneAvailable = op.runsOn.every((w) => w.available === 0);
+ return (
+ <span className={`text-xs text-muted-foreground ${className}`}>
+ also runs on{" "}
+ {op.runsOn
+ .map(
+ (w) =>
+ `${w.name} (${w.kind === "llm" ? "LLM endpoint" : "executor"}${
+ w.slots > 1 ? ` ×${w.slots}` : ""
+ })`,
+ )
+ .join(", ")}
+ {noneAvailable && (
+ <span className="text-warning">
+ {" "}
+ — none of it is available right now (degraded or disabled); this
+ machine carries the whole operation until one is re-enabled.
+ </span>
+ )}
+ </span>
+ );
+}
diff --git a/editor/app/auto-queue/lanes.ts b/editor/app/auto-queue/lanes.ts
@@ -25,6 +25,8 @@ import {
arbiterBlockedReason,
getArbiterJobId,
} from "yt-dlp-transcript-common/controller/arbiter";
+import { getWorkerPool } from "yt-dlp-transcript-common/jobs/workerPool";
+import { workerMatches } from "yt-dlp-transcript-common/lib/workers";
import { flattenLeaves } from "yt-dlp-transcript-common/jobs/autoQueuePolicy";
import { getChannelBriefs } from "../lib/requestCache";
import {
@@ -106,6 +108,25 @@ export type SweepOperation = {
// Corpus-wide reachable work for this operation alone, so the checkbox says
// what ticking it costs before the plan below re-folds.
reachable: number;
+ // WHERE THE WORK CAN RUN besides this machine: the delegate workers (LLM
+ // endpoints, tagged unit executors) whose tags match this operation, folded
+ // to one row per configured worker with its slot count. READ-ONLY and never
+ // persisted into the scope — delegation is a worker-pool concern, decided
+ // per item at dispatch, and the console only reports it. Derived through the
+ // SAME workerMatches rule the dispatchers grant by, so this row and the
+ // actual routing cannot disagree.
+ runsOn: SweepOperationWorker[];
+};
+
+export type SweepOperationWorker = {
+ // The configured worker's id (slot-expansion suffixes folded back together).
+ id: string;
+ name: string;
+ kind: "remote" | "llm";
+ slots: number;
+ // Slots currently enabled and not degraded — 0 across every delegate is the
+ // "configured but nothing can take it right now" warning state.
+ available: number;
};
// The unified dispatcher's state. Not a lane: it does not do work, it decides
@@ -209,6 +230,41 @@ export async function buildAutoQueueLanes(): Promise<AutoQueueLanesPayload> {
ids.filter((id) => row.byKind[id]).map((id) => [id, row.byKind[id]]),
),
}));
+ // The delegate workers per operation, matched the way the dispatchers match:
+ // an llm worker serves an operation its tags (or the llm default set) name;
+ // a remote worker takes units only when TAGGED (the unit-dispatch opt-in)
+ // and its tags intersect [operation, contended resource]. Local workers are
+ // not listed — the lane always runs here, and saying so on every row would
+ // be noise.
+ const workerSummary = getWorkerPool().summary();
+ const contendsForOf = new Map(
+ kinds.map((k) => [k.id, (k.laneFor?.(settings) ?? k.lane).contendsFor]),
+ );
+ const runsOnFor = (id: string): SweepOperationWorker[] => {
+ const unitRequires = [id, contendsForOf.get(id) ?? "cpu"];
+ const byBase = new Map<string, SweepOperationWorker>();
+ for (const w of workerSummary) {
+ const takesIt =
+ w.kind === "llm"
+ ? workerMatches({ kind: "llm", tags: w.tags }, [id])
+ : w.kind === "remote" &&
+ (w.tags?.length ?? 0) > 0 &&
+ workerMatches({ kind: "remote", tags: w.tags }, unitRequires);
+ if (!takesIt) continue;
+ const base = w.id.split("#")[0];
+ const entry = byBase.get(base) ?? {
+ id: base,
+ name: w.name.replace(/ #\d+$/, ""),
+ kind: w.kind as "remote" | "llm",
+ slots: 0,
+ available: 0,
+ };
+ entry.slots++;
+ if (w.state === "enabled" && !w.degraded) entry.available++;
+ byBase.set(base, entry);
+ }
+ return [...byBase.values()];
+ };
const operationsFor = (ids: ReadonlyArray<string>): SweepOperation[] =>
ids.map((id) => ({
id,
@@ -218,6 +274,7 @@ export async function buildAutoQueueLanes(): Promise<AutoQueueLanesPayload> {
(n, row) => n + (row.byKind[id]?.reachable ?? 0),
0,
),
+ runsOn: runsOnFor(id),
}));
return {