commit 0930b12baa86c072ebd26614d4b95ce8b4497585
parent 6e001f2b97a0e6476041c190f812435bb9141928
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 21 Aug 2026 23:33:08 -0400
auto-queue: a policy leaf can name an operation
Expressible only — nothing dispatches an operation yet. This is the
shape the arbiter needs: one tree, one claiming path, four pipelines.
`operation` sits on AutoQueueMatch beside `bucket`, not on the node.
A node field would need group inheritance ("this group is the digest
subtree"), and inheritance is resolution logic buildPendingByLeaf does
not have; on the match it needs one sanitizer and one claiming path.
ChannelWork gains a separate `operations` id space rather than more
entries in `buckets`. `buckets` is walked as a priority-ordered union
by every leaf naming no bucket, so an operation folded in there would
be claimed by a transcription catch-all that never asked for it — and
selectableBucketsForKind feeds the editor's bucket dropdown, where an
operation must not appear as a bucket.
THE CLAIM KEY BECOMES `${operation}\0${id}`. The same video is
legitimately pending for digest AND diarization — different work on
the same input — so an id-keyed `claimed` set would let whichever leaf
ran first steal the other's work and silently drop it. Every tree in
existence names no operation, so every key is "\0"+id and the dedup is
byte-identical.
A leaf naming both: operation wins and the sanitizer DROPS bucket, so
the stored tree cannot express the ambiguity (coerce-to-legal, this
file's existing style).
The runner projects only the operations the tree actually names. A
snapshot's backfill[op].ids is up to 11,329 strings on one channel, so
projecting the whole catalog would put ~78,000 strings a tick behind a
feature that is off.
AutoQueueKind is deliberately NOT widened — it threads through ~8
branches in autoRunner, the control route, status.ts and two e2e
suites.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
3 files changed, 220 insertions(+), 9 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -15,6 +15,7 @@ import { makeTaskTracker } from "../jobs/taskHooks";
import type { JobRunContext } from "../jobs/streamCommand";
import {
type ActiveCounts,
+ type AutoQueueGroup,
type AutoQueueOrder,
type AutoQueuePolicy,
type ChannelWork,
@@ -204,6 +205,13 @@ async function buildChannelWork(
paths: Paths,
kind: AutoQueueKind,
meta: ReadonlyArray<ChannelMeta>,
+ // Operation id spaces to project alongside the buckets — exactly the ones the
+ // policy tree actually names, never the whole catalog. A snapshot's
+ // backfill[op].ids is up to 11,329 strings on one channel, so projecting an
+ // operation nothing asks for would put ~78,000 strings a tick behind a
+ // feature that is off. Every tree written before operations existed names
+ // none, so this is empty and the projection is byte-identical to today's.
+ operations: ReadonlyArray<string> = [],
): Promise<{ channels: ChannelWork[]; owner: Map<string, string> }> {
const channels: ChannelWork[] = [];
const owner = new Map<string, string>();
@@ -232,11 +240,34 @@ async function buildChannelWork(
buckets[name] = ids;
for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
}
- channels.push({ slug, platform, buckets });
+ let ops: Record<string, string[]> | undefined;
+ if (operations.length > 0) {
+ ops = {};
+ for (const op of operations) {
+ // `ids` IS the reachable set — missing + stale + partial, never
+ // missing-input, deferred or blocked (see BackfillSnapshotEntry). A
+ // leaf pointed at an operation therefore claims only work the lane can
+ // actually do, which is the same contract a bucket carries.
+ const ids = snap.backfill?.[op]?.ids ?? [];
+ ops[op] = ids;
+ for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
+ }
+ }
+ channels.push({ slug, platform, buckets, operations: ops });
}
return { channels, owner };
}
+// Every operation id named by a leaf anywhere in the tree, deduped. Drives the
+// projection above, so an operation costs nothing until a rule asks for it.
+export function operationsNamedBy(root: AutoQueueGroup): string[] {
+ const out = new Set<string>();
+ for (const leaf of flattenLeaves(root)) {
+ if (leaf.match.operation) out.add(leaf.match.operation);
+ }
+ return [...out];
+}
+
// The policy's `order`, defaulted for settings files written before the field
// existed. One place, so the runner and the status page can never disagree about
// what an absent field means.
@@ -269,7 +300,10 @@ async function recencyOrdering(
const candidateIds = new Set<string>();
const owner = new Map<string, string>();
for (const ch of channels) {
- for (const ids of Object.values(ch.buckets)) {
+ for (const ids of [
+ ...Object.values(ch.buckets),
+ ...Object.values(ch.operations ?? {}),
+ ]) {
for (const id of ids) {
candidateIds.add(id);
if (!owner.has(id)) owner.set(id, ch.slug);
@@ -336,7 +370,12 @@ export async function computeLeafPending(
): Promise<LeafPending> {
const policy = getSettings().autoQueue[kind];
const meta = await listChannelMeta(paths);
- const { channels, owner } = await buildChannelWork(paths, kind, meta);
+ const { channels, owner } = await buildChannelWork(
+ paths,
+ kind,
+ meta,
+ operationsNamedBy(policy.root),
+ );
const { compare, keys } = await recencyOrdering(
kind,
paths,
@@ -630,7 +669,12 @@ async function runLoop(
const slugToPlatform = new Map(
metaCache.map((m) => [m.slug, m.platform ?? "unknown"]),
);
- const { channels, owner } = await buildChannelWork(paths, kind, metaCache);
+ const { channels, owner } = await buildChannelWork(
+ paths,
+ kind,
+ metaCache,
+ operationsNamedBy(policy.root),
+ );
// Re-derived each iteration from the freshly-read policy, like `enabled`, so
// toggling the replace-auto-captions lane (or the ordering) takes effect
// without a restart. The recency index behind the comparator has its own
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -711,3 +711,116 @@ test("sanitize: a lapsed snooze normalizes to null, a future one survives", () =
assert.equal(sanitizeAutoQueue({}).transcription.snoozeUntil, null);
assert.equal(sanitizeAutoQueue({ download: { snoozeUntil: "soon" } }).download.snoozeUntil, null);
});
+
+// --- Operations: a leaf can name one -----------------------------------------
+
+const OP_CHANNELS: ChannelWork[] = [
+ {
+ slug: "ch",
+ platform: "youtube",
+ buckets: { downloadedNoTranscript: ["v1", "v2"] },
+ // The SAME videos are pending for two different operations, which is the
+ // normal case: a transcript can want a digest and its audio a diarization.
+ operations: { digest: ["v1", "v3"], diarization: ["v1", "v4"] },
+ },
+];
+
+test("a leaf naming an operation draws from operations, not buckets", () => {
+ const pending = buildPendingByLeaf(
+ {
+ id: "root",
+ mode: "strict",
+ children: [
+ { id: "dig", match: { type: "all", operation: "digest" } },
+ { id: "rest", match: { type: "all" } },
+ ],
+ },
+ OP_CHANNELS,
+ ["downloadedNoTranscript"],
+ );
+ assert.deepEqual(pending.dig, ["v1", "v3"]);
+ // And the bucket leaf is UNAFFECTED: v1 being claimed for digest must not
+ // remove it from the transcription queue, because they are different work.
+ assert.deepEqual(pending.rest, ["v1", "v2"]);
+});
+
+test("two operations claim the same video independently", () => {
+ // The whole reason the claim key carries the operation. An id-keyed set would
+ // let whichever leaf ran first take v1 and silently drop it from the other
+ // operation's queue.
+ const pending = buildPendingByLeaf(
+ {
+ id: "root",
+ mode: "strict",
+ children: [
+ { id: "dig", match: { type: "all", operation: "digest" } },
+ { id: "dia", match: { type: "all", operation: "diarization" } },
+ ],
+ },
+ OP_CHANNELS,
+ ["downloadedNoTranscript"],
+ );
+ assert.deepEqual(pending.dig, ["v1", "v3"]);
+ assert.deepEqual(pending.dia, ["v1", "v4"]);
+});
+
+test("two leaves on the SAME operation still dedup, first-match-wins", () => {
+ const pending = buildPendingByLeaf(
+ {
+ id: "root",
+ mode: "strict",
+ children: [
+ { id: "first", match: { type: "channel", value: "ch", operation: "digest" } },
+ { id: "second", match: { type: "all", operation: "digest" } },
+ ],
+ },
+ OP_CHANNELS,
+ ["downloadedNoTranscript"],
+ );
+ assert.deepEqual(pending.first, ["v1", "v3"]);
+ assert.deepEqual(pending.second, []);
+});
+
+test("a projection with no operations map leaves an operation leaf empty", () => {
+ // Every ChannelWork written before operations existed omits the map. The leaf
+ // must find nothing rather than throw or fall back to the buckets.
+ const pending = buildPendingByLeaf(
+ { id: "root", mode: "strict", children: [{ id: "dig", match: { type: "all", operation: "digest" } }] },
+ [{ slug: "ch", platform: "youtube", buckets: { downloadedNoTranscript: ["v1"] } }],
+ ["downloadedNoTranscript"],
+ );
+ assert.deepEqual(pending.dig, []);
+});
+
+test("a tree with no operation is byte-for-byte what it was", () => {
+ // The compatibility claim: every settings.json in existence names no
+ // operation, so every claim key is "\0"+id and the dedup is unchanged.
+ const before = buildPendingByLeaf(ORDER_ROOT, ORDER_CHANNELS, [
+ "downloadedNoTranscript",
+ "failedListed",
+ ]);
+ assert.deepEqual(before.all, ["c_old", "c_new", "d_mid"]);
+});
+
+test("sanitize: operation wins and DROPS bucket, so the tree cannot be ambiguous", () => {
+ const s = sanitizeAutoQueue({
+ transcription: {
+ root: {
+ id: "root",
+ mode: "strict",
+ children: [
+ { id: "a", match: { type: "all", operation: " digest ", bucket: "failedListed" } },
+ { id: "b", match: { type: "all", bucket: "failedListed" } },
+ { id: "c", match: { type: "all", operation: " " } },
+ ],
+ },
+ },
+ download: { root: {} },
+ });
+ const leaves = flattenLeaves(s.transcription.root);
+ assert.deepEqual(leaves[0].match, { type: "all", operation: "digest" });
+ // An untouched bucket leaf keeps working exactly as before.
+ assert.deepEqual(leaves[1].match, { type: "all", bucket: "failedListed" });
+ // A blank operation is not an operation.
+ assert.deepEqual(leaves[2].match, { type: "all" });
+});
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -90,6 +90,23 @@ export type AutoQueueMatch = {
// runner kind (transcription → downloadedNoTranscript, download →
// undownloadedIds). E.g. bucket="failedListed" prioritizes retries.
bucket?: string;
+ // Optional OPERATION this leaf draws from — a registered backfill kind id, or
+ // "digest". Same meaning as `bucket` one level up: it narrows what the leaf
+ // claims, and it draws from ChannelWork.operations rather than
+ // ChannelWork.buckets.
+ //
+ // It lives on the MATCH, beside `bucket`, and not on the node. A field on the
+ // node would need group inheritance — "this group is the digest subtree" —
+ // and inheritance is resolution logic buildPendingByLeaf does not have. Here
+ // it needs exactly one sanitizer and exactly one claiming path.
+ //
+ // It is a SEPARATE id space from `bucket`, and the sanitizer enforces that a
+ // leaf names at most one of the two (operation wins): `defaultBuckets` is a
+ // priority-ordered union, so a name that meant a bucket to one leaf and an
+ // operation to another would silently mix two id spaces, and
+ // selectableBucketsForKind feeds the editor's bucket dropdown, where an
+ // operation must not appear as a bucket.
+ operation?: string;
};
export type AutoQueueLeaf = {
@@ -236,6 +253,15 @@ export type ChannelWork = {
platform: Platform | null;
// bucket name -> available video ids in that bucket
buckets: Record<string, string[]>;
+ // operation id -> reachable video ids for that operation, from
+ // snapshot.backfill[op].ids. A SEPARATE map rather than more entries in
+ // `buckets`, because `buckets` is walked as a priority-ordered union by every
+ // leaf that names no bucket — an operation folded in there would be claimed
+ // by a transcription catch-all that never meant to ask for it.
+ //
+ // Optional: every projection written before operations existed omits it, and
+ // a leaf naming an operation simply finds nothing.
+ operations?: Record<string, string[]>;
};
// Pre-order (priority-order) flatten of every leaf in the tree.
@@ -281,17 +307,33 @@ export function buildPendingByLeaf(
const leaves = flattenLeaves(root);
const pending: Record<string, string[]> = {};
for (const leaf of leaves) pending[leaf.id] = [];
+ // Claims are keyed `${operation}\0${id}`, NOT by id.
+ //
+ // The same video is legitimately pending for digest AND for diarization —
+ // they are different work on the same input — so an id-keyed set would let
+ // whichever leaf happened to run first steal the other operation's work and
+ // silently drop it from the queue. Every tree written before operations
+ // existed has no operation anywhere, so every key is "\0" + id and the
+ // dedup behaves exactly as it always has.
const claimed = new Set<string>();
for (const leaf of leaves) {
- const bucketNames = leaf.match.bucket ? [leaf.match.bucket] : defaultBuckets;
- for (const bucket of bucketNames) {
+ const operation = leaf.match.operation ?? "";
+ // An operation leaf draws from its own id space and ignores buckets
+ // entirely; sanitizeMatch guarantees a leaf never names both.
+ const lists: ReadonlyArray<string> = operation
+ ? [operation]
+ : leaf.match.bucket
+ ? [leaf.match.bucket]
+ : defaultBuckets;
+ for (const list of lists) {
for (const ch of channels) {
if (!matchesChannel(leaf.match, ch)) continue;
- const ids = ch.buckets[bucket];
+ const ids = operation ? ch.operations?.[list] : ch.buckets[list];
if (!ids) continue;
for (const id of ids) {
- if (claimed.has(id)) continue;
- claimed.add(id);
+ const claim = `${operation}\u0000${id}`;
+ if (claimed.has(claim)) continue;
+ claimed.add(claim);
pending[leaf.id].push(id);
}
}
@@ -421,6 +463,18 @@ function sanitizeMatch(value: unknown): AutoQueueMatch {
: "all";
const out: AutoQueueMatch = { type };
if (typeof r.value === "string" && r.value.trim()) out.value = r.value.trim();
+ const operation =
+ typeof r.operation === "string" && r.operation.trim()
+ ? r.operation.trim()
+ : "";
+ if (operation) {
+ // Coerce-to-legal, this file's existing style: a leaf naming BOTH an
+ // operation and a bucket is ambiguous, so the stored tree is not allowed to
+ // express it. Operation wins and the bucket is dropped, rather than the
+ // pair being kept and resolved differently by whichever reader looks first.
+ out.operation = operation;
+ return out;
+ }
if (typeof r.bucket === "string" && r.bucket.trim()) {
out.bucket = r.bucket.trim();
}