Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a7e1e20f0a2b8ea9baeefe7f864774d12503c9dd
parent cde4f0521f02dc8e1e42737a908c3d09ef260cc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 26 Aug 2026 10:36:51 -0400

operations: the registry says "operation", not "backfill kind"

Mechanical rename, zero behaviour change. lib/backfillKinds.ts is
lib/operations.ts (with its two test files), and the ~25 Backfill* exports
take the noun the backend already used (operationCatalog, OperationDescriptor):
BackfillKind → Operation, BackfillLane → Lane, BACKFILL_KINDS → OPERATIONS,
getBackfillKind → getOperation, BackfillState → OperationState,
BackfillCostTier → OperationTier, *_KIND_ID → *_OPERATION_ID, and so on.

The rule that decided each one: "backfill" survives where it names the
LANE and its persisted contracts, and goes where it named the kind concept.
So laneBackfillKinds is backfillLaneOperations (it filters on
BACKFILL_QUEUE), laneEntriesOf is backfillLaneEntriesOf, and
resolveBackfillKinds is resolveBackfillLaneOperations — once Lane is generic,
"lane entries" of WHICH lane has to be said. BACKFILL_QUEUE, the
backfill-channel/backfill-sweep job kinds, settings.backfill, the snapshot's
`backfill` key, kindIds/sweepKinds and every "kind" outside the registry's
export table are unchanged, each on purpose.

The header is rewritten around the new noun and states the lane rule; the
costBasis comment states the "unit" rule (a unit is dispatchable work; what
a video of an operation costs is its cost basis) that commit 2 applies.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/arbiter.test.ts | 10+++++-----
Mcommon/controller/arbiter.ts | 20++++++++++----------
Mcommon/controller/arbiterWork.ts | 2+-
Mcommon/controller/attributionTarget.ts | 2+-
Mcommon/controller/autoRunner.ts | 2+-
Mcommon/controller/backfillBatch.test.ts | 6+++---
Mcommon/controller/backfillBatch.ts | 48++++++++++++++++++++++++------------------------
Mcommon/controller/backfillSweep.ts | 24++++++++++++------------
Mcommon/controller/channelSnapshot.test.ts | 16++++++++--------
Mcommon/controller/channelSnapshot.ts | 84++++++++++++++++++++++++++++++++++++++++----------------------------------------
Mcommon/controller/digestBatch.ts | 2+-
Mcommon/controller/laneForOperation.test.ts | 6+++---
Mcommon/controller/laneGuards.test.ts | 2+-
Mcommon/controller/laneGuards.ts | 8++++----
Mcommon/controller/normalizeAll.test.ts | 2+-
Mcommon/controller/operationJobs.ts | 8++++----
Mcommon/controller/operationLane.ts | 10+++++-----
Mcommon/controller/remoteUnit.ts | 2+-
Mcommon/controller/sweepPreview.ts | 16++++++++--------
Mcommon/controller/sweepRecency.ts | 0
Mcommon/controller/workerServer.ts | 16++++++++--------
Mcommon/lib/attribution.ts | 2+-
Dcommon/lib/backfillKinds.test.ts | 1494-------------------------------------------------------------------------------
Dcommon/lib/backfillKinds.ts | 1704-------------------------------------------------------------------------------
Dcommon/lib/backfillUnit.test.ts | 228-------------------------------------------------------------------------------
Mcommon/lib/diarization.ts | 2+-
Acommon/lib/operationUnit.test.ts | 228+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/operations.test.ts | 1494+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/lib/operations.ts | 1728+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/settings.ts | 2+-
Mcommon/lib/sweepPlan.ts | 4++--
Mcommon/lib/videoStatus.ts | 2+-
Mcommon/lib/workers.ts | 4++--
Meditor/app/actionable/lib/loadActionable.ts | 14+++++++-------
Meditor/app/actionable/page.tsx | 6+++---
Meditor/app/api/widget/sync/route.ts | 30+++++++++++++++---------------
Meditor/app/api/worker/unit/route.ts | 4++--
Meditor/app/channels/[slug]/components/flow/SidingList.tsx | 2+-
Meditor/app/channels/[slug]/components/stages/BackfillStage.tsx | 2+-
Meditor/app/channels/[slug]/components/stages/DigestStage.tsx | 2+-
Meditor/app/channels/[slug]/lib/channelFlow.test.ts | 34+++++++++++++++++-----------------
Meditor/app/channels/[slug]/lib/channelFlow.ts | 34+++++++++++++++++-----------------
Meditor/app/channels/[slug]/lib/stageOrder.test.ts | 2+-
Meditor/app/channels/[slug]/lib/stageStatus.ts | 12++++++------
Meditor/app/channels/[slug]/page.tsx | 28++++++++++++++--------------
Meditor/app/channels/components/ChannelsTable.tsx | 2+-
Meditor/app/channels/groupActions.ts | 4++--
Meditor/app/channels/lib/channelGroupSections.test.ts | 8++++----
Meditor/app/channels/lib/channelGroupSections.ts | 22+++++++++++-----------
Meditor/app/channels/page.tsx | 6+++---
Meditor/app/components/lanes/LaneDeck.tsx | 14+++++++-------
Meditor/app/components/pipelines/band.ts | 2+-
Meditor/app/components/pipelines/buildBands.test.ts | 6+++---
Meditor/app/components/pipelines/buildBands.ts | 18+++++++++---------
Meditor/app/jobs/actions.ts | 6+++---
Meditor/app/jobs/active/buildActiveJobs.ts | 4++--
Meditor/app/operations/[id]/page.tsx | 2+-
Meditor/app/operations/components/SweepLane.tsx | 2+-
Meditor/app/operations/components/SweepPlan.tsx | 2+-
Meditor/app/operations/components/SweepScope.tsx | 2+-
Meditor/app/operations/components/railStates.ts | 2+-
Meditor/app/operations/lanes.ts | 16++++++++--------
Meditor/app/settings/components/SettingsForm.tsx | 2+-
Meditor/app/settings/components/WorkersField.tsx | 2+-
Meditor/app/settings/page.tsx | 2+-
Meditor/e2e/channel-line.spec.ts | 2+-
66 files changed, 3749 insertions(+), 3725 deletions(-)

diff --git a/common/controller/arbiter.test.ts b/common/controller/arbiter.test.ts @@ -4,7 +4,7 @@ import type { AutoQueueGroup, ChannelWork, } from "../jobs/autoQueuePolicy"; -import type { BackfillLane } from "../lib/backfillKinds"; +import type { Lane } from "../lib/operations"; import { planArbiterUnits, type ArbiterUnit } from "./arbiter"; // Run with: node_modules/.bin/tsx --test common/controller/arbiter.test.ts @@ -13,15 +13,15 @@ import { planArbiterUnits, type ArbiterUnit } from "./arbiter"; // policy tree's PRIORITY becomes a DISPATCH ORDER. The loop around it is // job-starting and sleeping, which an integration run covers. -const GPU: BackfillLane = { queueKey: "digest:local", contendsFor: "gpu" }; -const CPU: BackfillLane = { queueKey: "backfill", contendsFor: "cpu" }; +const GPU: Lane = { queueKey: "digest:local", contendsFor: "gpu" }; +const CPU: Lane = { queueKey: "backfill", contendsFor: "cpu" }; -const LANES: Record<string, BackfillLane> = { +const LANES: Record<string, Lane> = { digest: GPU, diarization: CPU, "attribution-text": CPU, }; -const laneOf = (op: string): BackfillLane | null => LANES[op] ?? null; +const laneOf = (op: string): Lane | null => LANES[op] ?? null; function channels(): ChannelWork[] { return [ diff --git a/common/controller/arbiter.ts b/common/controller/arbiter.ts @@ -46,11 +46,11 @@ import { type ChannelWork, } from "../jobs/autoQueuePolicy"; import { - DIGEST_KIND_ID, - allBackfillKinds, + DIGEST_OPERATION_ID, + allOperations, operationLabel, - type BackfillLane, -} from "../lib/backfillKinds"; + type Lane, +} from "../lib/operations"; import { laneForOperation } from "./operationLane"; import { runOperationChannelJob } from "./operationJobs"; import { @@ -101,7 +101,7 @@ export type ArbiterUnit = { operation: string; channelSlug: string; ids: string[]; - lane: BackfillLane; + lane: Lane; // The leaf that claimed it, for the log. An operator asking "why that // channel" gets the rule back rather than a shrug. leafId: string; @@ -121,7 +121,7 @@ export type ArbiterUnit = { export function planArbiterUnits( roots: ReadonlyArray<AutoQueueGroup>, channels: ReadonlyArray<ChannelWork>, - laneOf: (operation: string) => BackfillLane | null, + laneOf: (operation: string) => Lane | null, ): ArbiterUnit[] { const units: ArbiterUnit[] = []; for (const root of roots) { @@ -146,7 +146,7 @@ export function planArbiterUnits( // (digest.dependsOn needs them to be) but dispatched by their own runners. // // A switched-off kind does not reach here at all, and not through this - // branch: `enabled` below projects ids only for kinds allBackfillKinds + // branch: `enabled` below projects ids only for kinds allOperations // returns, so a disabled feature arrives with an empty id list and is // skipped one line up. laneForOperation itself does not consult // `enabled` — see controller/laneForOperation.test.ts. @@ -217,7 +217,7 @@ async function runOneUnit( // mega-job — and so this loop is a scheduler rather than a second runtime. // // ONE CALL, whichever operation it is. This used to be a ternary on - // DIGEST_KIND_ID with two hand-written option sets, which meant the arbiter + // DIGEST_OPERATION_ID with two hand-written option sets, which meant the arbiter // had to be edited in step with the registry and re-derived the digest lane // from settings a second time — while `unit.lane` beside it was already the // answer. It is now: laneForOperation asks the kind's own laneFor, so @@ -281,8 +281,8 @@ async function runArbiterLoop( // may name one an operator later disabled, and projecting it would put tens // of thousands of ids behind a feature that is off. const enabled = new Set([ - ...allBackfillKinds(settings).map((k) => k.id), - DIGEST_KIND_ID, + ...allOperations(settings).map((k) => k.id), + DIGEST_OPERATION_ID, ]); const operations = new Set<string>(); for (const root of roots) { diff --git a/common/controller/arbiterWork.ts b/common/controller/arbiterWork.ts @@ -16,7 +16,7 @@ import type { Platform } from "../lib/platform"; // not have on the arbiter's call. // // `ids` IS the reachable set (missing + stale + partial), never missing-input, -// deferred or blocked; see BackfillSnapshotEntry. So a leaf pointed at an +// deferred or blocked; see OperationSnapshotEntry. So a leaf pointed at an // operation claims only work the lane can actually do, which is exactly the // contract a bucket carries. diff --git a/common/controller/attributionTarget.ts b/common/controller/attributionTarget.ts @@ -3,7 +3,7 @@ // reason that module exists separately from digestVideo.ts: the COUNTERS need // the identity without needing the runner. // -// lib/backfillKinds.ts is imported by controller/channelSnapshot.ts, which +// lib/operations.ts is imported by controller/channelSnapshot.ts, which // classifies every video of every channel and sits on the editor's hot path. // Reaching the runner from there drags its whole import graph in with it — the // transcript normalizer, the markdown renderer, the digest prompt module and the diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -246,7 +246,7 @@ async function buildChannelWork( ops = {}; for (const op of operations) { // `ids` IS the reachable set — missing + stale + partial, never - // missing-input, deferred or blocked (see BackfillSnapshotEntry). A + // missing-input, deferred or blocked (see OperationSnapshotEntry). A // leaf pointed at an operation therefore claims only work the lane can // actually do, which is the same contract a bucket carries. const ids = snap.backfill?.[op]?.ids ?? []; diff --git a/common/controller/backfillBatch.test.ts b/common/controller/backfillBatch.test.ts @@ -7,7 +7,7 @@ import { sanitizeBackfill, sanitizeDiarization, } from "../lib/settings"; -import type { BackfillClassification } from "../lib/backfillKinds"; +import type { OperationClassification } from "../lib/operations"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/backfillBatch.test.ts @@ -151,7 +151,7 @@ test("every classification has an explicit decision", () => { // union without deciding what the pull does with it fails HERE as well as at // the compile step — a runtime backstop for the `never` check, since the // hazard this replaces was precisely a silent fall-through. - const ALL: BackfillClassification[] = [ + const ALL: OperationClassification[] = [ "present", "stale", // Part-done. Dispatched like stale — the runner regenerates only the @@ -170,7 +170,7 @@ test("every classification has an explicit decision", () => { assert.doesNotThrow(() => candidateAction(state, DISPATCH), state); } assert.throws( - () => candidateAction("invented" as BackfillClassification, DISPATCH), + () => candidateAction("invented" as OperationClassification, DISPATCH), /unhandled backfill state/, ); }); diff --git a/common/controller/backfillBatch.ts b/common/controller/backfillBatch.ts @@ -39,15 +39,15 @@ import type { TaskTracker } from "../jobs/taskHooks"; import type { JobProgress } from "../jobs/registry"; import { readVideoFiles } from "../lib/videoStatus"; import { - addBackfillState, - emptyBackfillCounts, + addOperationState, + emptyOperationCounts, laneYieldsToTranscription, - reachableBackfillWork, - resolveBackfillKinds, - type BackfillClassification, - type BackfillKind, - type BackfillRunOutcome, -} from "../lib/backfillKinds"; + reachableOperationWork, + resolveBackfillLaneOperations, + type OperationClassification, + type Operation, + type OperationRunOutcome, +} from "../lib/operations"; import { transcriptionActivity } from "./digestYield"; import { acquireLlmSlot, freeLlmSlots } from "./llmWorkers"; import { getWorkerPool } from "../jobs/workerPool"; @@ -157,8 +157,8 @@ export type BackfillBatchResult = { // // THIS USED TO BE AN IF-CHAIN INSIDE THE PULL, AND THAT WAS THE HAZARD. It // handled the states it knew about and FELL THROUGH TO DISPATCHING everything -// else — so a state added to BackfillState was not skipped by default, it was -// RUN by default. Nothing in the repo checked BackfillState exhaustively, so +// else — so a state added to OperationState was not skipped by default, it was +// RUN by default. Nothing in the repo checked OperationState exhaustively, so // widening the union raised zero TypeScript errors, and the first sign of a // missing branch would have been the six-hour video the new state existed to // avoid being handed to the engine anyway. @@ -182,7 +182,7 @@ export type CandidateAction = | "blocked"; export function candidateAction( - state: BackfillClassification, + state: OperationClassification, opts: { force: boolean; allowRedownload: boolean }, ): CandidateAction { switch (state) { @@ -204,7 +204,7 @@ export function candidateAction( // `force`. Forcing a video whose prerequisite has not been produced does // not make the prerequisite appear; it just hands the runner an input it // does not have. Running the kind this one dependsOn is how you unblock - // it, and the ordering in resolveBackfillKinds tries to do that for you + // it, and the ordering in resolveBackfillLaneOperations tries to do that for you // within the same pass. // // Note what is NOT here: re-acquiring media. That is what separating this @@ -229,7 +229,7 @@ export function candidateAction( } } -type Candidate = { id: string; kind: BackfillKind; target: unknown }; +type Candidate = { id: string; kind: Operation; target: unknown }; // The lane kinds whose work bottoms out in model calls an "llm" endpoint // worker can serve. Diarization is deliberately absent — its work is an audio @@ -249,7 +249,7 @@ export async function runBackfillBatch( ): Promise<BackfillBatchResult> { const log = opts.onLog ?? ((m: string) => console.log(m)); const settings = getSettings(); - const kinds = resolveBackfillKinds(settings, opts.kindIds); + const kinds = resolveBackfillLaneOperations(settings, opts.kindIds); const result: BackfillBatchResult = { attempted: 0, @@ -348,7 +348,7 @@ export async function runBackfillBatch( : {}), }; let unitActive = 0; - const unitRequires = (kind: BackfillKind): string[] => [ + const unitRequires = (kind: Operation): string[] => [ kind.id, (kind.laneFor?.(settings) ?? kind.lane).contendsFor, ]; @@ -367,7 +367,7 @@ export async function runBackfillBatch( videoDir: string, itemLog: (m: string) => void, runSignal: AbortSignal, - ): Promise<BackfillRunOutcome | null> => { + ): Promise<OperationRunOutcome | null> => { const pool = getWorkerPool(); for (let attempt = 0; attempt < MAX_UNIT_ATTEMPTS; attempt++) { const lease = pool.tryAcquire(unitRequires(candidate.kind), { @@ -616,7 +616,7 @@ export async function runBackfillBatch( // free; else the llm fan-out for the call-bound kinds; else the local // run exactly as today. All claims are non-parking on purpose: a parked // acquire inside a runPool slot would deadlock the batch. - let outcome: BackfillRunOutcome | null = await runBackfillUnit( + let outcome: OperationRunOutcome | null = await runBackfillUnit( candidate, videoDir, itemLog, @@ -757,10 +757,10 @@ export async function runBackfillBatch( // // DIGEST NOW DECLARES A laneFor TOO, and is still not in this decision — // but not because of this guard. `kinds` here came through - // resolveBackfillKinds, which filters via laneBackfillKinds and so admits + // resolveBackfillLaneOperations, which filters via backfillLaneOperations and so admits // BACKFILL_QUEUE kinds only; digest is on its own queue and can never // reach this array, even when asked for by name. That is the invariant - // backfillKinds.test.ts pins. + // operations.test.ts pins. // // Which is also why laneFor MUST STAY OPTIONAL. This line keys off its // PRESENCE, so "give every kind a laneFor defaulting to lane" would @@ -850,7 +850,7 @@ export async function countBackfillWork( blocked: number; }> { const settings = getSettings(); - const kinds = resolveBackfillKinds(settings, kindIds); + const kinds = resolveBackfillLaneOperations(settings, kindIds); if (kinds.length === 0) return { reachable: 0, missingInput: 0, deferred: 0, blocked: 0 }; const dataDir = path.join(paths.channelsDir, channelSlug, "data"); @@ -868,18 +868,18 @@ export async function countBackfillWork( // does; it used to re-spell the rule instead, and had already drifted — // `blocked` was classified by state() and then counted by nothing here, so a // channel of prerequisite-waiting videos reported three zeroes and looked - // finished. addBackfillState + reachableBackfillWork is the same pair the + // finished. addOperationState + reachableOperationWork is the same pair the // channel snapshot folds with, so this path and the snapshot path can no // longer disagree about what a state means, and a state added later is // counted here without anyone remembering to come back. - const counts = emptyBackfillCounts(); + const counts = emptyOperationCounts(); for (const id of dirs) { const videoDir = path.join(dataDir, id); const files = await readVideoFiles(videoDir, { checkUntranscribable: true, }); for (const kind of kinds) { - addBackfillState( + addOperationState( counts, await kind.state({ videoDir, @@ -892,7 +892,7 @@ export async function countBackfillWork( } } return { - reachable: reachableBackfillWork(counts), + reachable: reachableOperationWork(counts), missingInput: counts.missingInput, deferred: counts.deferred, blocked: counts.blocked, diff --git a/common/controller/backfillSweep.ts b/common/controller/backfillSweep.ts @@ -33,10 +33,10 @@ import { runManagedFunction } from "../jobs/streamCommand"; import { drainStream } from "../jobs/drainStream"; import { BACKFILL_QUEUE } from "../lib/queueKeys"; import { - reachableBackfillWork, - resolveBackfillKinds, - type BackfillSnapshotEntry, -} from "../lib/backfillKinds"; + reachableOperationWork, + resolveBackfillLaneOperations, + type OperationSnapshotEntry, +} from "../lib/operations"; import { listChannelStatsFromDisk, readChannelSnapshot } from "./channels"; import { countBackfillWork } from "./backfillBatch"; import { runBackfillChannelJob } from "./operationJobs"; @@ -145,17 +145,17 @@ export async function buildBackfillSweepPlan(opts: { const channels = (await listChannelStatsFromDisk(opts.paths)).filter( (ch) => wanted.size === 0 || wanted.has(ch.slug), ); - // Resolved once: the same filtering resolveBackfillKinds does for the batch, + // Resolved once: the same filtering resolveBackfillLaneOperations does for the batch, // so the plan counts exactly the kinds the run will act on. A snapshot holds // an entry per kind that was enabled when it was WRITTEN, which may be more // than are in scope now. - const kinds = resolveBackfillKinds(getSettings(), opts.kindIds).map( + const kinds = resolveBackfillLaneOperations(getSettings(), opts.kindIds).map( (k) => k.id, ); const order = opts.order ?? "listed"; const plan: BackfillPlanEntry[] = []; // Reachable ids per channel, collected only when they will be used. The - // snapshot's `ids` IS the reachable set (see BackfillSnapshotEntry), so this + // snapshot's `ids` IS the reachable set (see OperationSnapshotEntry), so this // is a read of something already in hand — but on this corpus it is ~78,000 // strings, so it is not built when nothing will sort by it. const idsBySlug = order === "listed" ? null : new Map<string, string[]>(); @@ -239,14 +239,14 @@ export async function buildBackfillSweepPlan(opts: { // The snapshot's per-kind counts, summed over the kinds in scope. `missing` and // `stale` are the reachable half; `missingInput` stays its own number and is -// never added to them — see lib/backfillKinds.ts's header. +// never added to them — see lib/operations.ts's header. // // EXPORTED so the console's preview counts with the RUN'S OWN ARITHMETIC rather // than a second copy of it. A preview that disagrees with the run about how much // work a channel holds is worse than no preview: it is the number an operator // arms a multi-day commitment against. See ./sweepPreview.ts. export function countFromSnapshot( - backfill: Record<string, BackfillSnapshotEntry>, + backfill: Record<string, OperationSnapshotEntry>, kinds: string[], ): { reachable: number; missingInput: number } { let reachable = 0; @@ -254,7 +254,7 @@ export function countFromSnapshot( for (const id of kinds) { const entry = backfill[id]; if (!entry) continue; - reachable += reachableBackfillWork(entry); + reachable += reachableOperationWork(entry); missingInput += entry.missingInput; } return { reachable, missingInput }; @@ -285,7 +285,7 @@ async function runSweepLoop( // The feature itself may have been switched off underneath us. Same clean // stop: a sweep for a disabled backfill has nothing to do and should not // sit there looking wedged. - if (resolveBackfillKinds(getSettings(), kindIds).length === 0) { + if (resolveBackfillLaneOperations(getSettings(), kindIds).length === 0) { onLog( "No enabled backfill kind is in scope — stopping. (Enable the feature in Settings, then re-arm.)", ); @@ -318,7 +318,7 @@ async function runSweepLoop( ); return; } - // BOTH NUMBERS, never their sum. See lib/backfillKinds.ts's header. + // BOTH NUMBERS, never their sum. See lib/operations.ts's header. onLog( `Pass ${pass}: ${work.length} channel(s), ${reachable.toLocaleString()} video(s) ` + `reachable now, ${missingInput.toLocaleString()} needing their media re-acquired` + diff --git a/common/controller/channelSnapshot.test.ts b/common/controller/channelSnapshot.test.ts @@ -8,10 +8,10 @@ import { foldBackfillEntry, } from "./channelSnapshot"; import { - emptyBackfillCounts, - reachableBackfillWork, - type BackfillClassification, -} from "../lib/backfillKinds"; + emptyOperationCounts, + reachableOperationWork, + type OperationClassification, +} from "../lib/operations"; // The snapshot's accounting, tested where it can actually be reached. // @@ -21,7 +21,7 @@ import { // looks exactly like success: a work list that disagrees with its own count, and // a coverage number that reads "all done" over a channel nothing has looked at. -function fold(states: BackfillClassification[]) { +function fold(states: OperationClassification[]) { return foldBackfillEntry(states.map((state, i) => ({ id: `v${i}`, state }))); } @@ -40,7 +40,7 @@ test("ids are EXACTLY the reachable set — the invariant nothing else checks", "missing-input", "not-applicable", ]); - assert.equal(entry.ids.length, reachableBackfillWork(entry)); + assert.equal(entry.ids.length, reachableOperationWork(entry)); assert.deepEqual(entry.ids, ["v0", "v1", "v2"]); // And the three that must never be reachable are still counted, just not there. assert.equal(entry.blocked, 1); @@ -51,7 +51,7 @@ test("ids are EXACTLY the reachable set — the invariant nothing else checks", test("eligible counts every video the operation had an opinion about", () => { // not-applicable is the ONLY exclusion: an untranscribable video is out of // scope, everything else is either done or outstanding. Without this the - // `present` half is unrecoverable, because addBackfillState discards it. + // `present` half is unrecoverable, because addOperationState discards it. const entry = fold([ "present", "present", @@ -93,7 +93,7 @@ test("digestWorkOf prefers the registry entry over the legacy bucket", () => { const work = digestWorkOf({ backfill: { digest: { - ...emptyBackfillCounts(), + ...emptyOperationCounts(), missing: 2, partial: 1, blocked: 4, diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -27,16 +27,16 @@ import { isDoNotClean } from "../lib/doNotClean-server"; import { loadDigest } from "../lib/digest-server"; import { isSectionFresh } from "../lib/digest"; import { - addBackfillState, - allBackfillKinds, - emptyBackfillCounts, - presentBackfillWork, - reachableBackfillWork, - DIGEST_KIND_ID, - DIARIZATION_KIND_ID, - type BackfillClassification, - type BackfillSnapshotEntry, -} from "../lib/backfillKinds"; + addOperationState, + allOperations, + emptyOperationCounts, + presentOperationWork, + reachableOperationWork, + DIGEST_OPERATION_ID, + DIARIZATION_OPERATION_ID, + type OperationClassification, + type OperationSnapshotEntry, +} from "../lib/operations"; import { resolveDigestTarget } from "./digestTarget"; import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; @@ -83,7 +83,7 @@ export type ChannelSnapshot = { // local lane is actually carrying the corpus. Optional: older snapshots lack // it; readers default to {}. digestEngines?: Record<string, number>; - // Per-OPERATION work counts, keyed by kind id (see lib/backfillKinds.ts). + // Per-OPERATION work counts, keyed by kind id (see lib/operations.ts). // Beside `totals` and NOT in `buckets`, following the digestEngines precedent // above for the same reason: buckets is a closed literal of `string[]` id // lists, and this is per-kind counts. @@ -95,17 +95,17 @@ export type ChannelSnapshot = { // reason /api/widget/actionable refuses to filter on `noDigest`. // // THIS MAP IS NO LONGER THE BACKFILL LANE, and anything reading it generically - // must say which lane it means. It is written from allBackfillKinds — every + // must say which lane it means. It is written from allOperations — every // enabled operation in the catalog — because a work list the snapshot does not // carry is a work list nothing can select, count or schedule. The digest // operation runs on its own queue key and is the first entry here that the - // backfill lane must not touch. Read it with laneEntriesOf() to get the lane, + // backfill lane must not touch. Read it with backfillLaneEntriesOf() to get the lane, // or by id to get one operation; a bare Object.values() over this map now // means "every operation", which on this corpus is a ~75,000-video difference. // // Optional: snapshots written before this existed lack it, and readers default // to {}. - backfill?: Record<string, BackfillSnapshotEntry>; + backfill?: Record<string, OperationSnapshotEntry>; buckets: { noTranscript: string[]; downloadedNoTranscript: string[]; @@ -337,7 +337,7 @@ export function attributeAudioHold(v: { } // Will the diarize lane ever produce this video's sidecar? `undefined` means the -// kind isn't enabled at all (allBackfillKinds dropped it — models unconfigured), +// kind isn't enabled at all (allOperations dropped it — models unconfigured), // which is the case the guard cannot see: the sweep holds the audio on // settings.diarization.enabled alone, so the hold is permanent and silent. // @@ -346,7 +346,7 @@ export function attributeAudioHold(v: { // `not-applicable` (an untranscribable transcript) are all holds nothing queued // will release. export function diarizationWillNeverClear( - state: BackfillClassification | undefined, + state: OperationClassification | undefined, ): boolean { if (state === undefined) return true; return !(state === "missing" || state === "stale" || state === "partial"); @@ -416,26 +416,26 @@ export function normalizeMaybeMissing( // archive reader and a corpus on disk — which is to say, never. The rule that // matters and that nothing else checks: // -// ids.length === reachableBackfillWork(entry) +// ids.length === reachableOperationWork(entry) // // A policy leaf hands `ids` out as the work to do while the stage cards render // the count, so the two drifting apart is a progress bar that stalls one short // of complete forever. That is not hypothetical — countBackfillWork re-spelled // this same rule and had already lost `blocked` from it. export function foldBackfillEntry( - videos: Iterable<{ id: string; state: BackfillClassification | undefined }>, -): BackfillSnapshotEntry { - const counts = emptyBackfillCounts(); + videos: Iterable<{ id: string; state: OperationClassification | undefined }>, +): OperationSnapshotEntry { + const counts = emptyOperationCounts(); const ids: string[] = []; // The denominator: every video this kind had an opinion about. `present` is - // the classification addBackfillState deliberately discards, so without this + // the classification addOperationState deliberately discards, so without this // no reader can tell "0 outstanding because it is all done" from "0 // outstanding because there was nothing here". let eligible = 0; for (const { id, state } of videos) { if (!state) continue; if (state !== "not-applicable") eligible++; - addBackfillState(counts, state); + addOperationState(counts, state); if (state === "missing" || state === "stale" || state === "partial") { ids.push(id); } @@ -485,15 +485,15 @@ export function digestWorkOf( | null | undefined, ): DigestWork { - const entry = snapshot?.backfill?.[DIGEST_KIND_ID]; + const entry = snapshot?.backfill?.[DIGEST_OPERATION_ID]; if (entry) { return { ids: entry.ids ?? [], - reachable: reachableBackfillWork(entry), + reachable: reachableOperationWork(entry), blocked: entry.blocked ?? 0, deferred: entry.deferred ?? 0, partial: entry.partial ?? 0, - present: presentBackfillWork(entry), + present: presentOperationWork(entry), eligible: entry.eligible ?? null, source: "registry", }; @@ -607,12 +607,12 @@ export async function generateChannelSnapshot( // per-video probe below is a comparison rather than a derivation, exactly as // digestTarget is above. // - // allBackfillKinds, NOT laneBackfillKinds. The snapshot's job is to carry a + // allOperations, NOT backfillLaneOperations. The snapshot's job is to carry a // work list for every operation in the catalog, not for one lane: nothing can // count, select or schedule work the snapshot does not record, which is why // registering the digest kind in Phase C changed no number anywhere — its // state() was called by nothing. The lane filter belongs on the READ side, - // where a surface says which lane it means (laneEntriesOf). + // where a surface says which lane it means (backfillLaneEntriesOf). // // COST, measured rather than assumed, because this runs per video over ~79,000 // of them: the digest classification is 0.34 ms/video on a 125-video channel @@ -623,9 +623,9 @@ export async function generateChannelSnapshot( // deliberately NOT done: at this cost it would be an optimization with no // measurement behind it. const backfillSettings = getSettings(); - const backfillKinds = allBackfillKinds(backfillSettings); + const liveOperations = allOperations(backfillSettings); const backfillTargets: Record<string, unknown> = {}; - for (const kind of backfillKinds) { + for (const kind of liveOperations) { backfillTargets[kind.id] = await kind.resolveTarget({ settings: backfillSettings, paths, @@ -695,16 +695,16 @@ export async function generateChannelSnapshot( // Work state, per registered operation. // // This USED to cost nothing in the default configuration, because every - // backfill feature ships off and `backfillKinds` was therefore empty. + // backfill feature ships off and `liveOperations` was therefore empty. // That is no longer true and the change is deliberate: the digest // operation is enabled whenever an app is configured (its pause lives at // dispatch, not here, so a paused lane still reports its outstanding // work), so this loop now always runs at least once per video. See the - // measured per-video cost where backfillKinds is resolved above — it is + // measured per-video cost where liveOperations is resolved above — it is // roughly the readVideoFiles call this classification reuses rather than // repeats. - const backfill: Record<string, BackfillClassification> = {}; - for (const kind of backfillKinds) { + const backfill: Record<string, OperationClassification> = {}; + for (const kind of liveOperations) { backfill[kind.id] = await kind.state({ videoDir: dir, videoId: id, @@ -841,12 +841,12 @@ export async function generateChannelSnapshot( const heldAudioCounts = emptyHeldAudio(); let reclaimableAtRiskBytes = 0; // Is the lane that would release an awaiting-diarization video even running? - // allBackfillKinds drops a kind whose enabled() is false, so an ABSENT entry + // allOperations drops a kind whose enabled() is false, so an ABSENT entry // means "no diarize job will ever be produced" — while the sweep's guard holds // the audio on settings.diarization.enabled alone. That gap is a permanent, // silent hold, and this is the only place in the app that can see it. - const diarizationKindEnabled = backfillKinds.some( - (k) => k.id === DIARIZATION_KIND_ID, + const diarizationKindEnabled = liveOperations.some( + (k) => k.id === DIARIZATION_OPERATION_ID, ); for (const { id, @@ -905,7 +905,7 @@ export async function generateChannelSnapshot( if ( gate === "awaitingDiarization" && diarizationWillNeverClear( - diarizationKindEnabled ? backfill[DIARIZATION_KIND_ID] : undefined, + diarizationKindEnabled ? backfill[DIARIZATION_OPERATION_ID] : undefined, ) ) { heldAudioBytes.diarizationNeverClears += audioBytes; @@ -1168,19 +1168,19 @@ export async function generateChannelSnapshot( : undefined; // Fold the per-video classifications into per-operation counts. Only the - // reachable half carries ids — see BackfillSnapshotEntry for why missing-input + // reachable half carries ids — see OperationSnapshotEntry for why missing-input // does not. // - // `ids` MUST stay exactly the set reachableBackfillWork counts, because a + // `ids` MUST stay exactly the set reachableOperationWork counts, because a // policy leaf consumes this list as the work to hand out while the stage cards // render the number: a list and a total that disagree is a progress bar that // stalls one short of complete forever. `partial` is reachable, so it is in - // both. channelSnapshot.test.ts pins `ids.length === reachableBackfillWork()` + // both. channelSnapshot.test.ts pins `ids.length === reachableOperationWork()` // for exactly this reason: the two derivations have drifted apart once already // (countBackfillWork silently dropped `blocked`), and nothing about the shape // here would have caught it. - const backfillCounts: Record<string, BackfillSnapshotEntry> = {}; - for (const kind of backfillKinds) { + const backfillCounts: Record<string, OperationSnapshotEntry> = {}; + for (const kind of liveOperations) { backfillCounts[kind.id] = foldBackfillEntry( perVideo.map((v) => ({ id: v.id, state: v.backfill[kind.id] })), ); diff --git a/common/controller/digestBatch.ts b/common/controller/digestBatch.ts @@ -27,7 +27,7 @@ import type { JobProgress } from "../jobs/registry"; import { isSectionFresh, type DigestSectionKind } from "../lib/digest"; import { digestLaneFor, -} from "../lib/backfillKinds"; +} from "../lib/operations"; import { loadDigest } from "../lib/digest-server"; import { resolveDigestTarget, diff --git a/common/controller/laneForOperation.test.ts b/common/controller/laneForOperation.test.ts @@ -80,7 +80,7 @@ test("diarization resolves through its laneFor too", () => { test("a KNOWN external operation still has no lane", () => { // download and transcription are in operationCatalog() — they have to be, or // digest.dependsOn = ["transcription"] names nothing — but they are dispatched - // by their OWN runners. laneForOperation asks getBackfillKind, not the + // by their OWN runners. laneForOperation asks getOperation, not the // catalog, precisely so they come back null and planArbiterUnits skips them. // // A catalog lookup would hand them a lane and the arbiter would start a @@ -96,9 +96,9 @@ test("a KNOWN external operation still has no lane", () => { test("a switched-off kind still resolves a lane — the off-switch is elsewhere", () => { // Pinning the BOUNDARY, because it is the one an obvious-looking "fix" would // move. This resolver answers "which lane would this run on", not "should it - // run": getBackfillKind is a registry lookup and has never consulted + // run": getOperation is a registry lookup and has never consulted // `enabled`. What keeps a disabled feature out of the arbiter is the `enabled` - // set in runArbiterPass, which projects ids only for kinds allBackfillKinds + // set in runArbiterPass, which projects ids only for kinds allOperations // returns — so a switched-off kind arrives with an EMPTY id list and produces // no unit, well before a lane is asked for. // diff --git a/common/controller/laneGuards.test.ts b/common/controller/laneGuards.test.ts @@ -1,7 +1,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { defaultDigest, type SiteSettings } from "../lib/settings"; -import { digestLaneFor } from "../lib/backfillKinds"; +import { digestLaneFor } from "../lib/operations"; import { BACKFILL_QUEUE } from "../lib/queueKeys"; import { digestGate, diff --git a/common/controller/laneGuards.ts b/common/controller/laneGuards.ts @@ -2,14 +2,14 @@ import type { SiteSettings } from "../lib/settings"; import { digestLaneFor, laneYieldsToTranscription, - type BackfillLane, -} from "../lib/backfillKinds"; + type Lane, +} from "../lib/operations"; import type { DigestLane } from "../lib/digest"; import { transcriptionActivity } from "./digestYield"; // THE LANE'S RULES, declared once, in the two shapes a dispatcher can act on. // -// This exists because of a sentence in laneBackfillKinds' own header: +// This exists because of a sentence in backfillLaneOperations' own header: // // "digest carries digestsPaused, the yield-to-transcription carve-out (with // its CPU-worker exemption), spendCapUsd on the metered lane, the @@ -188,7 +188,7 @@ export function digestGate(input: DigestGateInput): LaneGate { // Declared here rather than inferred from an id, so a dispatcher can ask the // question without knowing which operation it is holding. `useClusters: false` // on a run still overrides it; this is the DEFAULT, not a lock. -export function laneSharesDuplicates(lane: BackfillLane): boolean { +export function laneSharesDuplicates(lane: Lane): boolean { return ( lane.queueKey === digestLaneFor("local-gpu").queueKey || lane.queueKey === digestLaneFor("remote-api").queueKey diff --git a/common/controller/normalizeAll.test.ts b/common/controller/normalizeAll.test.ts @@ -27,7 +27,7 @@ import type { Paths } from "../lib/paths"; // // The distinction is about COPY and about whether an operator has anything to // do, not about dispatch — both still classify `deferred` — so it lives on -// isCuesJsonFresh's reason rather than in a new BackfillState. +// isCuesJsonFresh's reason rather than in a new OperationState. const CUES_BODY = JSON.stringify({ version: CUES_FILE_VERSION, diff --git a/common/controller/operationJobs.ts b/common/controller/operationJobs.ts @@ -39,7 +39,7 @@ import { DIGEST_REMOTE_QUEUE, resolveQueueKey, } from "../lib/queueKeys"; -import { DIGEST_KIND_ID } from "../lib/backfillKinds"; +import { DIGEST_OPERATION_ID } from "../lib/operations"; import { countBackfillWork, runBackfillBatch } from "./backfillBatch"; import { countMissingDigests, @@ -115,7 +115,7 @@ export async function runBackfillChannelJob( // reading a card's could not see that 3 were mid-rewrite. // // `missingInput` is printed SEPARATELY and never added to anything. See - // lib/backfillKinds.ts's header: on this corpus it is ~91x the reachable + // lib/operations.ts's header: on this corpus it is ~91x the reachable // figure, and one summed "remaining" would be noise. onLog( `Backfill ${channelSlug}: ${batch.succeeded} done, ${batch.fresh} already current, ` + @@ -293,14 +293,14 @@ export type OperationChannelJobOptions = { // This is the seam the arbiter and the /channels group buttons wanted: a caller // that has an operation id and a channel slug should not also have to know which // of two runners that id belongs to, nor which queue key to reserve. Both were -// open-coded before — the arbiter with a ternary on DIGEST_KIND_ID, the group +// open-coded before — the arbiter with a ternary on DIGEST_OPERATION_ID, the group // buttons with a per-station map — and both had to be edited in step with the // registry. export function runOperationChannelJob( opts: OperationChannelJobOptions, ): Promise<StreamActionResult> { const { paths, channelSlug, operation } = opts; - if (operation === DIGEST_KIND_ID) { + if (operation === DIGEST_OPERATION_ID) { return runDigestChannelJob({ paths, channelSlug, diff --git a/common/controller/operationLane.ts b/common/controller/operationLane.ts @@ -6,7 +6,7 @@ // plans. Left in arbiter.ts, that was an import cycle. import { getSettings } from "../lib/settings"; -import { getBackfillKind, type BackfillLane } from "../lib/backfillKinds"; +import { getOperation, type Lane } from "../lib/operations"; // The lane an operation runs on, or null when the registry does not know it. // @@ -18,14 +18,14 @@ import { getBackfillKind, type BackfillLane } from "../lib/backfillKinds"; // places and only digest's copy was live: diarization's laneFor was invisible // to the arbiter. // -// getBackfillKind, NOT operationCatalog(), and that is deliberate. Today -// getBackfillKind("download") is undefined, so download and transcription get +// getOperation, NOT operationCatalog(), and that is deliberate. Today +// getOperation("download") is undefined, so download and transcription get // no lane and planArbiterUnits skips them — which is correct, because they are // dispatched by their own runners. A // catalog lookup would hand them a lane and the arbiter would start a backfill // channel job for work no backfill kind can do. -export function laneForOperation(operation: string): BackfillLane | null { - const kind = getBackfillKind(operation); +export function laneForOperation(operation: string): Lane | null { + const kind = getOperation(operation); if (!kind) return null; return kind.laneFor?.(getSettings()) ?? kind.lane; } diff --git a/common/controller/remoteUnit.ts b/common/controller/remoteUnit.ts @@ -38,7 +38,7 @@ export type RemoteUnitOptions = { }; export type RemoteUnitOutcome = { - // The executor-side BackfillRunOutcome ("done", "already-present", …). + // The executor-side OperationRunOutcome ("done", "already-present", …). outcome: string; // The kind's declared output files as UTF-8 JSON text, to be applied on the // primary through kind.applyResult — never written raw. diff --git a/common/controller/sweepPreview.ts b/common/controller/sweepPreview.ts @@ -18,9 +18,9 @@ import type { AutoQueueOrder } from "../jobs/autoQueuePolicy"; import { - DIGEST_KIND_ID, - presentBackfillWork, -} from "../lib/backfillKinds"; + DIGEST_OPERATION_ID, + presentOperationWork, +} from "../lib/operations"; import { foldSweepPlan, type SweepChannelCounts, @@ -71,8 +71,8 @@ function emptyKindCounts(): SweepKindCounts { // The reachable ids for one operation on one channel. // // snapshot.backfill[id].ids IS the reachable set and nothing else — never -// missing-input, deferred or blocked (a test in lib/backfillKinds keeps it equal -// to reachableBackfillWork). So dating these ids dates exactly the work the +// missing-input, deferred or blocked (a test in lib/operations keeps it equal +// to reachableOperationWork). So dating these ids dates exactly the work the // sweep would do, which is the only population the order is about. function reachableIdsFor( snapshot: ChannelSnapshot, @@ -85,7 +85,7 @@ function reachableIdsFor( // the buckets with the transcript and cues-staleness gates applied — the same // one components/pipelines/buildBands.ts uses, for the same reason: without it // a stale channel reads "all digested". - if (id === DIGEST_KIND_ID) return digestWorkOf(snapshot).ids; + if (id === DIGEST_OPERATION_ID) return digestWorkOf(snapshot).ids; return []; } @@ -107,10 +107,10 @@ function countsFor(snapshot: ChannelSnapshot, id: string): WorkCounts | null { blocked: entry.blocked ?? 0, deferred: entry.deferred ?? 0, eligible: entry.eligible ?? null, - present: presentBackfillWork(entry), + present: presentOperationWork(entry), }; } - if (id === DIGEST_KIND_ID) { + if (id === DIGEST_OPERATION_ID) { const work = digestWorkOf(snapshot); return { reachable: work.reachable, diff --git a/common/controller/sweepRecency.ts b/common/controller/sweepRecency.ts Binary files differ. diff --git a/common/controller/workerServer.ts b/common/controller/workerServer.ts @@ -14,9 +14,9 @@ import { runManagedFunction } from "../jobs/streamCommand"; import { makeTaskTracker } from "../jobs/taskHooks"; import { transcribeWithWorker } from "./transcribeOne"; import { - getBackfillKind, - type BackfillRunOutcome, -} from "../lib/backfillKinds"; + getOperation, + type OperationRunOutcome, +} from "../lib/operations"; import { CUES_JSON_FILENAME } from "../lib/videoStatus"; import type { AttributionSettings, @@ -197,7 +197,7 @@ const UNIT_OUTCOME_FILENAME = "unit-outcome.json"; export type StartWorkerUnitInput = { // A backfill kind id. `download`/`transcription` are ExternalOperations, not - // kinds, so getBackfillKind() refusing them is the door guard that keeps + // kinds, so getOperation() refusing them is the door guard that keeps // download politeness (and the transcription pool) single-scheduler. op: string; channelSlug: string; @@ -226,7 +226,7 @@ export type StartWorkerUnitInput = { export async function startWorkerUnit( input: StartWorkerUnitInput, ): Promise<{ remoteJobId: string }> { - const kind = getBackfillKind(input.op); + const kind = getOperation(input.op); if (!kind) { throw new Error( `operation "${input.op}" is not a backfill kind this executor can run`, @@ -271,7 +271,7 @@ export async function startWorkerUnit( // snapshot for a channel this box does not have. videoId: input.videoId, fn: async (onLog, signal) => { - const outcome: BackfillRunOutcome = await kind.run({ + const outcome: OperationRunOutcome = await kind.run({ paths: scratchPaths, videoDir, videoId: input.videoId, @@ -317,7 +317,7 @@ export async function startWorkerUnit( } export type WorkerUnitResult = { - // The kind's BackfillRunOutcome ("done", "already-present", "skipped", …). + // The kind's OperationRunOutcome ("done", "already-present", "skipped", …). outcome: string; // The kind's declared output files, as UTF-8 JSON text by name. May be empty // for a no-op outcome. @@ -343,7 +343,7 @@ export async function readWorkerUnitResult( } catch { // keep "done" — the marker only exists after a successful run } - const kind = getBackfillKind(job.unit.op); + const kind = getOperation(job.unit.op); const files: Record<string, string> = {}; for (const name of kind?.outputs ?? []) { const p = path.join(job.unit.videoDir, name); diff --git a/common/lib/attribution.ts b/common/lib/attribution.ts @@ -231,7 +231,7 @@ function sameVersion(a: number | undefined, b: number | undefined): boolean { // but would be wrong for the text kind, where a diarized record means there is // nothing to do. The text kind therefore checks isAttributionDowngrade FIRST and // only asks this question when it is genuinely its own record it is looking at. -// See lib/backfillKinds.ts. +// See lib/operations.ts. export function isAttributionFresh( record: AttributionRecord | null, target: AttributionFreshnessTarget, diff --git a/common/lib/backfillKinds.test.ts b/common/lib/backfillKinds.test.ts @@ -1,1494 +0,0 @@ -import { test } from "node:test"; -import assert from "node:assert/strict"; -import path from "node:path"; -import os from "node:os"; -import { mkdtemp, mkdir, writeFile, rm, utimes } from "node:fs/promises"; -import { - getBackfillKind, - allBackfillKinds, - laneBackfillKinds, - resolveBackfillKinds, - orderByDependencies, - operationCatalog, - operationCostBasis, - operationGroup, - operationLabel, - operationsActionLabel, - operationsGroupLabel, - digestLaneFor, - diarizationLaneFor, - laneYieldsToTranscription, - addBackfillState, - emptyBackfillCounts, - reachableBackfillWork, - type BackfillClassification, - type BackfillKind, - laneEntriesOf, - laneKindEntriesOf, - presentBackfillWork, -} from "./backfillKinds"; -import { candidateAction } from "../controller/backfillBatch"; -import { - readVideoFiles, - CUES_JSON_FILENAME, - META_FILENAME, - SOURCE_MEDIA_BASENAME, -} from "./videoStatus"; -import { CUES_FILE_VERSION } from "../controller/normalizeTranscript"; -import { attributeOneVideo } from "../controller/attributeOne"; -import { DIGEST_FILENAME, OLLAMA_DIGEST_APP_ID } from "./digest"; -import { BACKFILL_QUEUE, DIGEST_LOCAL_QUEUE } from "./queueKeys"; -import { - ATTRIBUTION_FILENAME, - ATTRIBUTION_PROMPT_VERSION, - type AttributionRecord, -} from "./attribution"; -import { - DEFAULT_DIARIZATION_THRESHOLD, - DIARIZATION_FILENAME, - SORTFORMER_DIARIZATION_ENGINE, - diarizationTarget, - isDiarizationFresh, - type DiarizationRecord, -} from "./diarization"; -import { defaultSiteSettings, type SiteSettings } from "./settings"; -import { SAVED_VIDEO_POINTER_FILENAME } from "./savedVideo"; - -// Run with: -// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/backfillKinds.test.ts -// -// Every case here is a FIXTURE DIRECTORY, on purpose: `state()` is defined as -// "read from disk, never stored", and a test that hands it a hand-built record -// would not be testing the thing the indicators actually call. - -const diarization = getBackfillKind("diarization")!; - -// resolveTarget is channel-scoped since the digest entry (a digest's identity -// includes the hash of the channel's context note). None of the kinds tested -// here reads paths/channelSlug, so a stub is honest as well as convenient. -function targetCtx(settings: SiteSettings) { - return { - settings, - paths: { channelsDir: "/nonexistent" } as unknown as Parameters< - typeof diarization.resolveTarget - >[0]["paths"], - channelSlug: "test-channel", - }; -} - -function settingsWithDiarization( - over: Partial<SiteSettings["diarization"]> = {}, -): SiteSettings { - const s = defaultSiteSettings(); - return { - ...s, - diarization: { - ...s.diarization, - enabled: true, - segModel: "/opt/models/seg-1.onnx", - embModel: "/opt/models/emb-1.onnx", - threshold: DEFAULT_DIARIZATION_THRESHOLD, - ...over, - }, - }; -} - -// The duration cap is OFF by default now that windowing exists, so a test of the -// cap has to ask for one. Stated here once rather than inline, so it is obvious -// that every OTHER test in this file runs with the cap disabled — which is the -// shipped configuration. -function settingsWithCap(hours: number): SiteSettings { - return settingsWithDiarization({ maxAudioHours: hours }); -} - -// A video dir built from a description of what is on disk. `sidecar` is written -// verbatim so a MALFORMED file can be tested — that case is not hypothetical, -// it is what a crash mid-write leaves behind. -async function fixture(opts: { - transcript?: boolean; - audio?: boolean; - container?: boolean; - savedPointer?: boolean; - sidecar?: string; - // Written into metadata.info.json. Absent means NO metadata file at all, - // which is the "duration unknown" case the cap has to get right. - durationSec?: number; -}): Promise<{ dir: string; cleanup: () => Promise<void> }> { - const root = await mkdtemp(path.join(os.tmpdir(), "backfill-kinds-")); - const dir = path.join(root, "vid1"); - await mkdir(dir, { recursive: true }); - if (opts.transcript !== false) { - await writeFile( - path.join(dir, "transcript.json"), - JSON.stringify({ transcription: [{ text: "hi" }] }), - ); - } - if (opts.audio) await writeFile(path.join(dir, "audio.mp3"), "x"); - // `source-media.<ext>` specifically — isSourceMediaFile keys off the basename, - // and an arbitrarily-named mp4 in a video dir is not a persisted container. - if (opts.container) { - await writeFile(path.join(dir, `${SOURCE_MEDIA_BASENAME}.mp4`), "x"); - } - if (opts.savedPointer) { - await writeFile( - path.join(dir, SAVED_VIDEO_POINTER_FILENAME), - JSON.stringify({ storedAt: "now", dir: "/nowhere", file: "video.mp4" }), - ); - } - if (opts.sidecar !== undefined) { - await writeFile(path.join(dir, DIARIZATION_FILENAME), opts.sidecar); - } - if (opts.durationSec !== undefined) { - await writeFile( - path.join(dir, META_FILENAME), - JSON.stringify({ id: "vid1", duration: opts.durationSec }), - ); - } - return { dir, cleanup: () => rm(root, { recursive: true, force: true }) }; -} - -function sidecar(engine: DiarizationRecord["engine"]): string { - return JSON.stringify({ - videoId: "vid1", - generatedAt: "2026-08-07T00:00:00.000Z", - speakers: 2, - turns: [{ start: 0, end: 4, speaker: 0 }], - engine, - } satisfies DiarizationRecord); -} - -// The current identity, as scripts/diarize.mjs would record it: BASENAMES, not -// the configured full paths. -const CURRENT = { - engine: "sherpa-onnx", - segmentationModel: "seg-1.onnx", - embeddingModel: "emb-1.onnx", - threshold: DEFAULT_DIARIZATION_THRESHOLD, -}; - -async function classify( - dirOpts: Parameters<typeof fixture>[0], - settings: SiteSettings = settingsWithDiarization(), -): Promise<BackfillClassification> { - const { dir, cleanup } = await fixture(dirOpts); - try { - const files = await readVideoFiles(dir, { checkUntranscribable: true }); - return await diarization.state({ - videoDir: dir, - videoId: "vid1", - files, - target: await diarization.resolveTarget(targetCtx(settings)), - settings, - }); - } finally { - await cleanup(); - } -} - -test("present: the sidecar matches the identity we would produce now", async () => { - assert.equal( - await classify({ audio: true, sidecar: sidecar(CURRENT) }), - "present", - ); - // Still present with the audio already cleaned away — the whole point of - // capturing while the audio exists is that the result outlives it. - assert.equal(await classify({ sidecar: sidecar(CURRENT) }), "present"); -}); - -test("stale: a different threshold or model is work, not coverage", async () => { - // The threshold is the single most consequential knob (it decides how many - // speakers come out), so a change to it MUST show as work. Before the - // comparator, this read as done. - assert.equal( - await classify({ - audio: true, - sidecar: sidecar({ ...CURRENT, threshold: 0.5 }), - }), - "stale", - ); - assert.equal( - await classify({ - audio: true, - sidecar: sidecar({ ...CURRENT, segmentationModel: "seg-OLD.onnx" }), - }), - "stale", - ); - assert.equal( - await classify({ - audio: true, - sidecar: sidecar({ ...CURRENT, embeddingModel: "emb-OLD.onnx" }), - }), - "stale", - ); -}); - -// THE ENGINE IS NOT COMPARED, and that is a fix rather than an omission. -// scripts/diarize.mjs records whichever binary actually ran, and nothing in -// DiarizationSettings can predict that — so a hardcoded engine in the target -// would mark every sidecar from any other wrapper permanently stale, which at -// ~500-680 s/audio-hour is an infinite regeneration loop. (The e2e fake engine -// records "fake-diarize" and would have tripped it on the first run.) -test("a different engine binary is not, by itself, stale", async () => { - assert.equal( - await classify({ - audio: true, - sidecar: sidecar({ ...CURRENT, engine: "some-other-engine" }), - }), - "present", - ); - // The comparison is still WRITTEN, so adding an engine setting later needs no - // new logic: a target that does declare one still rejects a mismatch. - assert.equal( - isDiarizationFresh( - { - videoId: "v", - generatedAt: "now", - speakers: 1, - turns: [], - engine: { ...CURRENT, engine: "some-other-engine" }, - }, - { ...CURRENT, engine: "sherpa-onnx" }, - ), - false, - ); -}); - -// The engine-as-a-setting case the test above says needs "no new logic". It is -// asserted for sortformer and NOT for the default, and that asymmetry is the -// whole design: the default engine still cannot predict which binary a -// `--engine` override runs, so it must keep asserting nothing. -test("selecting sortformer asserts the engine; selecting the default still does not", () => { - const rec = (engine: DiarizationRecord["engine"]): DiarizationRecord => ({ - videoId: "v", - generatedAt: "now", - speakers: 1, - turns: [], - engine, - }); - const sherpaSidecar = rec({ - engine: "sherpa-onnx", - segmentationModel: "seg-1.onnx", - embeddingModel: "emb-1.onnx", - threshold: DEFAULT_DIARIZATION_THRESHOLD, - }); - const sortformerSidecar = rec({ - engine: SORTFORMER_DIARIZATION_ENGINE, - model: "sortformer-4spk.gguf", - }); - - const sortformerTarget = diarizationTarget({ - engine: SORTFORMER_DIARIZATION_ENGINE, - sortformerModel: "/abs/path/sortformer-4spk.gguf", - // Still configured, because settings carry one set of fields for both - // engines. None of it may leak into the sortformer identity. - segModel: "/abs/seg-1.onnx", - embModel: "/abs/emb-1.onnx", - threshold: DEFAULT_DIARIZATION_THRESHOLD, - }); - - // Basename only, as everywhere else: the corpus is rsynced between shards. - assert.equal(sortformerTarget.model, "sortformer-4spk.gguf"); - assert.equal(sortformerTarget.segmentationModel, undefined); - assert.equal(sortformerTarget.embeddingModel, undefined); - - assert.equal(isDiarizationFresh(sortformerSidecar, sortformerTarget), true); - // The point of switching: every sherpa sidecar becomes work the backfill lane - // will offer to redo, rather than silently staying half a corpus. - assert.equal(isDiarizationFresh(sherpaSidecar, sortformerTarget), false); - - // And back the other way, with no special case needed — a sortformer record - // carries neither segmentation nor embedding model, so it fails the sherpa - // comparison on the models alone. - const sherpaTarget = diarizationTarget({ - engine: "sherpa-onnx", - segModel: "/abs/seg-1.onnx", - embModel: "/abs/emb-1.onnx", - threshold: DEFAULT_DIARIZATION_THRESHOLD, - }); - assert.equal(sherpaTarget.engine, undefined); - assert.equal(isDiarizationFresh(sherpaSidecar, sherpaTarget), true); - assert.equal(isDiarizationFresh(sortformerSidecar, sherpaTarget), false); -}); - -// Sortformer has no clustering step, so the threshold cannot have changed any of -// its turns. Letting it into the identity would regenerate the whole corpus for -// an edit that provably could not affect it. -test("the clustering threshold does not stale a sortformer sidecar", () => { - const sidecar: DiarizationRecord = { - videoId: "v", - generatedAt: "now", - speakers: 1, - turns: [], - engine: { engine: SORTFORMER_DIARIZATION_ENGINE, model: "m.gguf" }, - }; - const at = (threshold: number) => - diarizationTarget({ - engine: SORTFORMER_DIARIZATION_ENGINE, - sortformerModel: "m.gguf", - threshold, - }); - assert.equal(isDiarizationFresh(sidecar, at(DEFAULT_DIARIZATION_THRESHOLD)), true); - assert.equal(isDiarizationFresh(sidecar, at(0.4)), true); - // The model itself IS the identity, though — a different one is a redo. - assert.equal( - isDiarizationFresh( - sidecar, - diarizationTarget({ - engine: SORTFORMER_DIARIZATION_ENGINE, - sortformerModel: "other.gguf", - }), - ), - false, - ); -}); - -// Which resource diarization competes for is a CONFIGURATION outcome, and the -// backfill lane reads it to decide whether a guaranteed share is even available. -// Getting this wrong in the permissive direction puts ~4.4 GB of sortformer next -// to parakeet on an 8 GB card. -test("diarization contends for the GPU only as sortformer on vulkan", () => { - const lane = (engine: "sherpa-onnx" | "sortformer", backend: "vulkan" | "cpu") => - diarizationLaneFor({ engine, backend }); - - assert.equal(lane("sortformer", "vulkan").contendsFor, "gpu"); - assert.equal(laneYieldsToTranscription(lane("sortformer", "vulkan")), true); - - // The same engine on the CPU backend is only after cores. - assert.equal(lane("sortformer", "cpu").contendsFor, "cpu"); - assert.equal(laneYieldsToTranscription(lane("sortformer", "cpu")), false); - - // sherpa-onnx is ONNX/CPU, so a stale `backend: vulkan` left in settings must - // NOT make it claim the card — that would park the lane behind transcription - // for work using no shaders at all, which is the bug digestYield.ts already - // records having hit once. - assert.equal(lane("sherpa-onnx", "vulkan").contendsFor, "cpu"); - assert.equal(laneYieldsToTranscription(lane("sherpa-onnx", "vulkan")), false); - - // The queue key never changes: two diarizations must not run at once whichever - // engine is selected. - assert.equal(lane("sortformer", "vulkan").queueKey, BACKFILL_QUEUE); - assert.equal(lane("sherpa-onnx", "cpu").queueKey, BACKFILL_QUEUE); -}); - -// THE COMPATIBILITY RULE, and the reason it is written down: without it, adding -// a field to the provenance would mark all 77,000 videos stale at once. -test("an absent recorded field compares equal to today's default", async () => { - assert.equal( - await classify({ - audio: true, - // A sidecar from before the threshold was recorded at all. - sidecar: sidecar({ - engine: "sherpa-onnx", - segmentationModel: "seg-1.onnx", - embeddingModel: "emb-1.onnx", - }), - }), - "present", - ); - // ...and it is equal to the DEFAULT specifically, not to anything: with a - // non-default threshold configured, the same record is stale. - assert.equal( - await classify( - { - audio: true, - sidecar: sidecar({ - engine: "sherpa-onnx", - segmentationModel: "seg-1.onnx", - embeddingModel: "emb-1.onnx", - }), - }, - settingsWithDiarization({ threshold: 0.5 }), - ), - "stale", - ); -}); - -test("a version bump alone does not invalidate captured work", async () => { - // ~500-680 s/audio-hour of CPU says an engine point-release is not a reason to - // redo everything when the models and the threshold are unchanged. - assert.equal( - await classify({ - audio: true, - sidecar: sidecar({ ...CURRENT, version: "9.9.9" }), - }), - "present", - ); -}); - -test("missing: no sidecar, and the input is still here", async () => { - assert.equal(await classify({ audio: true }), "missing"); - // A persisted source container counts — ffmpeg reads it directly, which is - // what resolveDiarizableMedia does. - assert.equal(await classify({ container: true }), "missing"); -}); - -test("missing-input: no sidecar and nothing to diarize from", async () => { - assert.equal(await classify({}), "missing-input"); - // A saved-video POINTER whose stored file has gone (an unmounted backup disk) - // is not an input either — the pointer is not the media. - assert.equal(await classify({ savedPointer: true }), "missing-input"); -}); - -// --------------------------------------------------------------------------- -// The duration cap. A stopgap for an OOM that kills 6 of 10 videos over 6 hours -// on this box, burning ~40 minutes each and producing nothing. - -test("deferred: over the cap, with the input right there", async () => { - // 5 hours against the 4-hour default. The input EXISTS — that is the whole - // point of a third state: this is not missing-input (nothing to work from) and - // not missing (work to do); it is work deliberately not attempted. - assert.equal( - await classify({ audio: true, durationSec: 5 * 3600 }, settingsWithCap(4)), - "deferred", - ); -}); - -test("under the cap is ordinary missing work", async () => { - assert.equal( - await classify({ audio: true, durationSec: 3 * 3600 }, settingsWithCap(4)), - "missing", - ); - // Exactly at the cap is not over it. - assert.equal( - await classify({ audio: true, durationSec: 4 * 3600 }, settingsWithCap(4)), - "missing", - ); -}); - -test("unknown duration is NOT deferred", async () => { - // No metadata.info.json at all. Deferring here would quietly remove a video - // from the work list on the strength of a file that could not be read, and it - // would break every fixture in this file that predates the cap. - assert.equal(await classify({ audio: true }, settingsWithCap(4)), "missing"); - // Present but useless — the same answer, for the same reason. - assert.equal( - await classify({ audio: true, durationSec: 0 }, settingsWithCap(4)), - "missing", - ); -}); - -test("maxAudioHours 0 turns the cap off — and 0 is the shipped default", async () => { - // Where this setting went once windowed diarization landed. - assert.equal( - await classify( - { audio: true, durationSec: 12 * 3600 }, - settingsWithDiarization({ maxAudioHours: 0 }), - ), - "missing", - ); -}); - -test("the cap never overrules a sidecar that is already there", async () => { - // A long video ALREADY diarized stays `present`: the cap decides what to - // attempt, not what counts as done. Otherwise raising the cap would look like - // work appearing and lowering it would look like work being undone. - assert.equal( - await classify( - { audio: true, durationSec: 12 * 3600, sidecar: sidecar(CURRENT) }, - settingsWithCap(4), - ), - "present", - ); -}); - -test("the cap does not resurrect a video whose input is gone", async () => { - // missing-input is checked FIRST. A 12-hour video with nothing to diarize from - // is unreachable, not deferred — deferred promises "we could do this if you - // raised the cap", and that would be a lie here. - assert.equal( - await classify({ durationSec: 12 * 3600 }, settingsWithCap(4)), - "missing-input", - ); -}); - -test("a malformed sidecar reads as absent, never as done", async () => { - // The same rule diarization-server.ts's hasDiarization() encodes: a - // half-written file must not be what convinces anything the work is captured - // — that is what would let the cleanup sweep delete the only copy of the audio. - assert.equal( - await classify({ audio: true, sidecar: "{ not json" }), - "missing", - ); - assert.equal( - await classify({ audio: true, sidecar: JSON.stringify({ videoId: "x" }) }), - "missing", - ); -}); - -test("not-applicable: an untranscribed video is not this backfill's business", async () => { - assert.equal( - await classify({ transcript: false, audio: true }), - "not-applicable", - ); -}); - -test("a disabled or unconfigured feature reports no backfill at all", () => { - const s = defaultSiteSettings(); - // Off by default, so nothing advertises catch-up work for it. - assert.equal(diarization.enabled(s), false); - assert.equal(laneBackfillKinds(s).length, 0); - // Enabled but with no models is "not set up", which must also report nothing - // rather than a corpus-sized work list nobody can act on. - assert.equal( - diarization.enabled(settingsWithDiarization({ segModel: "" })), - false, - ); - assert.equal(diarization.enabled(settingsWithDiarization()), true); - assert.equal(laneBackfillKinds(settingsWithDiarization()).length, 1); -}); - -test("an unknown or stale kind id in the sweep scope is dropped, not fatal", () => { - const s = settingsWithDiarization(); - // Empty scope = every enabled lane kind. - assert.deepEqual( - resolveBackfillKinds(s, []).map((k) => k.id), - ["diarization"], - ); - assert.deepEqual( - resolveBackfillKinds(s, undefined).map((k) => k.id), - ["diarization"], - ); - // A settings file naming a kind from another build must not wedge the lane. - assert.deepEqual(resolveBackfillKinds(s, ["from-the-future"]), []); - assert.deepEqual( - resolveBackfillKinds(s, ["diarization", "from-the-future"]).map((k) => k.id), - ["diarization"], - ); -}); - -// --------------------------------------------------------------------------- -// Attribution — the second and third kinds, and the pair that shares one file -// --------------------------------------------------------------------------- - -const attrText = getBackfillKind("attribution-text")!; -const attrDiarized = getBackfillKind("attribution-diarized")!; - -function settingsWithAttribution( - over: Partial<SiteSettings["attribution"]> = {}, -): SiteSettings { - const s = defaultSiteSettings(); - return { - ...s, - attribution: { - ...s.attribution, - enabled: true, - diarizedEnabled: true, - textOnlyEnabled: true, - appId: OLLAMA_DIGEST_APP_ID, - model: "qwen2.5:7b", - ...over, - }, - }; -} - -function attrSidecar(over: Partial<AttributionRecord["provenance"]> = {}): string { - return JSON.stringify({ - videoId: "vid1", - generatedAt: "2026-08-07T00:00:00.000Z", - speakers: [{ index: 0, label: "Host" }], - segments: [{ start: 0, end: 10, speaker: 0 }], - provenance: { - method: "diarized", - appId: OLLAMA_DIGEST_APP_ID, - model: "qwen2.5:7b", - modelRequested: "qwen2.5:7b", - promptVersion: ATTRIBUTION_PROMPT_VERSION, - generatedAt: "2026-08-07T00:00:00.000Z", - ...over, - }, - } satisfies AttributionRecord); -} - -// Same fixture discipline as above — a real directory, because state() is -// defined as "read from disk, never stored". -async function attrFixture(opts: { - transcript?: boolean; - diarization?: string; - attribution?: string; -}): Promise<{ dir: string; cleanup: () => Promise<void> }> { - const root = await mkdtemp(path.join(os.tmpdir(), "backfill-attr-")); - const dir = path.join(root, "vid1"); - await mkdir(dir, { recursive: true }); - if (opts.transcript !== false) { - await writeFile( - path.join(dir, "transcript.json"), - JSON.stringify({ transcription: [{ text: "hi" }] }), - ); - } - if (opts.diarization !== undefined) { - await writeFile(path.join(dir, DIARIZATION_FILENAME), opts.diarization); - } - if (opts.attribution !== undefined) { - await writeFile(path.join(dir, ATTRIBUTION_FILENAME), opts.attribution); - } - return { dir, cleanup: () => rm(root, { recursive: true, force: true }) }; -} - -async function classifyAttr( - kind: typeof attrText, - dirOpts: Parameters<typeof attrFixture>[0], - settings: SiteSettings = settingsWithAttribution(), -): Promise<BackfillClassification> { - const { dir, cleanup } = await attrFixture(dirOpts); - try { - const files = await readVideoFiles(dir, { checkUntranscribable: true }); - return await kind.state({ - videoDir: dir, - videoId: "vid1", - files, - target: await kind.resolveTarget(targetCtx(settings)), - settings, - }); - } finally { - await cleanup(); - } -} - -// A diarization sidecar at a known generatedAt, so the "re-diarizing invalidates -// the naming" case has something to compare against. -const DIARIZED_AT = "2026-08-06T00:00:00.000Z"; -function diarizationSidecar(generatedAt = DIARIZED_AT): string { - return JSON.stringify({ - videoId: "vid1", - generatedAt, - speakers: 2, - turns: [{ start: 0, end: 10, speaker: 0 }], - engine: { engine: "sherpa-onnx" }, - } satisfies DiarizationRecord); -} - -test("attribution-text reaches every transcribed video and never reports missing-input", async () => { - // The claim that prices this lane: its input is the cue stream, which every - // transcribed video has. That is why it can reach the whole corpus, and why - // running it over the whole corpus costs ~194,000 model calls. - assert.equal(await classifyAttr(attrText, {}), "missing"); - assert.equal( - await classifyAttr(attrText, { transcript: false }), - "not-applicable", - ); -}); - -test("attribution-text: fresh, stale, and a malformed file that reads as absent", async () => { - assert.equal( - await classifyAttr(attrText, { - attribution: attrSidecar({ method: "text-only" }), - }), - "present", - ); - assert.equal( - await classifyAttr(attrText, { - attribution: attrSidecar({ method: "text-only", model: "llama3:8b", modelRequested: "llama3:8b" }), - }), - "stale", - ); - // A half-written sidecar must read as absent — and here that matters twice - // over, because a malformed file that read as "present" could masquerade as a - // diarized record and block this lane forever. - assert.equal( - await classifyAttr(attrText, { attribution: "{ not json" }), - "missing", - ); -}); - -// THE DOWNGRADE RULE, in the counter. Reporting a diarized record as work would -// put this lane in a loop of "attempt, refuse, still outstanding" on every pass -// of a multi-day sweep. -test("attribution-text has nothing to do where a diarized record exists", async () => { - assert.equal( - await classifyAttr(attrText, { attribution: attrSidecar() }), - "present", - ); - // Not even when that diarized record is itself stale — it is not this lane's - // record to redo. - assert.equal( - await classifyAttr(attrText, { - attribution: attrSidecar({ model: "llama3:8b", modelRequested: "llama3:8b" }), - }), - "present", - ); -}); - -test("attribution-diarized: no diarization.json is BLOCKED, not missing-input", async () => { - // ~73,000 videos on this corpus, against a handful reachable — so this state - // has to be right or every surface is wrong. - // - // THIS ASSERTION USED TO SAY missing-input, AND THAT WAS THE DEFECT. - // missing-input means one thing to the rest of the system: the media is gone, - // re-acquire it. So these videos were counted as needing media re-fetched, - // and with allowRedownload on the lane would have spent a download per video - // fetching AUDIO — which cannot satisfy a wait for diarization.json — and - // then deleted it again. What they are waiting for is the `diarization` kind, - // which this table produces. - assert.equal(await classifyAttr(attrDiarized, {}), "blocked"); - assert.equal( - await classifyAttr(attrDiarized, { attribution: attrSidecar() }), - "blocked", - ); -}); - -test("blocked is never dispatched, and re-download cannot change that", () => { - // The dispatch decision is the consequential one: `blocked` must not reach a - // runner whatever the flags say. Note allowRedownload — the flag that DOES - // turn missing-input into a dispatch — is deliberately inert here. - for (const force of [false, true]) { - for (const allowRedownload of [false, true]) { - assert.equal( - candidateAction("blocked", { force, allowRedownload }), - "blocked", - `force=${force} allowRedownload=${allowRedownload}`, - ); - } - } - // The contrast, so this test fails if the two ever get conflated again. - assert.equal( - candidateAction("missing-input", { force: false, allowRedownload: true }), - "dispatch", - ); -}); - -test("blocked is counted, and is NOT reachable work", () => { - const counts = emptyBackfillCounts(); - addBackfillState(counts, "blocked"); - addBackfillState(counts, "blocked"); - addBackfillState(counts, "missing"); - assert.equal(counts.blocked, 2); - assert.equal(counts.missing, 1); - // The load-bearing line. A corpus with one diarization and 73,000 waiting - // attributions must not report 73,000 jobs ready to run. - assert.equal(reachableBackfillWork(counts), 1); - // And it is its own number, not folded into the re-acquire population. - assert.equal(counts.missingInput, 0); - assert.equal(counts.deferred, 0); -}); - -test("a prerequisite is ordered before the kind that declares it", () => { - const base = settingsWithDiarization(); - const kinds = resolveBackfillKinds( - { - ...base, - attribution: { - ...base.attribution, - enabled: true, - diarizedEnabled: true, - textOnlyEnabled: true, - }, - }, - undefined, - ); - const ids = kinds.map((k) => k.id); - const diarizationAt = ids.indexOf("diarization"); - const diarizedAttrAt = ids.indexOf("attribution-diarized"); - assert.ok(diarizationAt >= 0 && diarizedAttrAt >= 0, ids.join(",")); - // Without this, a video diarized during a pass only becomes attributable on - // whatever LATER pass happens to find the sidecar on disk. - assert.ok( - diarizationAt < diarizedAttrAt, - `diarization must precede attribution-diarized, got ${ids.join(", ")}`, - ); - // AND the sort is STABLE: attribution-text stays last, which is a separate - // deliberate decision (the better lane must reach a diarized video first, or - // the text lane spends ~30 model calls to produce a record it must not write). - assert.equal(ids[ids.length - 1], "attribution-text", ids.join(",")); -}); - -test("ordering degrades safely when a prerequisite is absent or cyclic", () => { - const a = { id: "a", dependsOn: ["b"] } as unknown as BackfillKind; - const b = { id: "b", dependsOn: ["a"] } as unknown as BackfillKind; - // A cycle must not wedge the lane or silently drop a kind: every entry comes - // back, in declaration order. - assert.deepEqual( - orderByDependencies([a, b]).map((k) => k.id), - ["a", "b"], - ); - // A dependency on something not in the selection imposes no ordering, and - // does not remove the dependant from the run. - const lonely = { id: "lonely", dependsOn: ["not-here"] } as unknown as - BackfillKind; - assert.deepEqual( - orderByDependencies([lonely]).map((k) => k.id), - ["lonely"], - ); -}); - -test("attribution-diarized treats a text-only record as the upgrade queue", async () => { - // `missing`, not `stale`: the diarized record this kind is responsible for - // genuinely was never made. This is the whole of PLAN.md's bespoke "upgrade - // job", and it falls out of the registry rather than needing new machinery. - assert.equal( - await classifyAttr(attrDiarized, { - diarization: diarizationSidecar(), - attribution: attrSidecar({ method: "text-only" }), - }), - "missing", - ); - assert.equal( - await classifyAttr(attrDiarized, { diarization: diarizationSidecar() }), - "missing", - ); -}); - -test("attribution-diarized: present, and stale when the clusters underneath change", async () => { - assert.equal( - await classifyAttr(attrDiarized, { - diarization: diarizationSidecar(), - attribution: attrSidecar({ diarizationGeneratedAt: DIARIZED_AT }), - }), - "present", - ); - // Re-diarized since. Cluster 3 is now a different person, or nobody, so the - // names that pointed at it are work again. - assert.equal( - await classifyAttr(attrDiarized, { - diarization: diarizationSidecar("2026-08-09T00:00:00.000Z"), - attribution: attrSidecar({ diarizationGeneratedAt: DIARIZED_AT }), - }), - "stale", - ); - // A model change is stale the ordinary way too. - assert.equal( - await classifyAttr(attrDiarized, { - diarization: diarizationSidecar(), - attribution: attrSidecar({ - diarizationGeneratedAt: DIARIZED_AT, - model: "llama3:8b", - modelRequested: "llama3:8b", - }), - }), - "stale", - ); -}); - -// Capture may legitimately be switched off after a run: a sidecar on disk is a -// perfectly good input, and refusing to name it would strand exactly the work -// the capture lane exists to protect. -test("attribution-diarized does not require diarization CAPTURE to still be on", async () => { - const s = settingsWithAttribution(); - assert.equal(s.diarization.enabled, false); - assert.equal(attrDiarized.enabled(s), true); - assert.equal( - await classifyAttr( - attrDiarized, - { diarization: diarizationSidecar() }, - s, - ), - "missing", - ); -}); - -test("each attribution lane is gated separately, under one master switch", () => { - const off = defaultSiteSettings(); - assert.equal(attrText.enabled(off), false); - assert.equal(attrDiarized.enabled(off), false); - // Turning the feature on must not by itself arm a ~194,000-call sweep. - const onlyMaster = settingsWithAttribution({ - diarizedEnabled: false, - textOnlyEnabled: false, - }); - assert.equal(attrText.enabled(onlyMaster), false); - assert.equal(attrDiarized.enabled(onlyMaster), false); - assert.equal( - attrText.enabled(settingsWithAttribution({ textOnlyEnabled: false })), - false, - ); - assert.equal( - attrDiarized.enabled(settingsWithAttribution({ diarizedEnabled: false })), - false, - ); - // Both lanes on. Diarization CAPTURE is still off in this fixture, so two — - // which is also the point: an attribution backfill does not need the capture - // lane armed to have work. - assert.equal(laneBackfillKinds(settingsWithAttribution()).length, 2); -}); - -// The cheap, better lane must get to a video first: on a video that has -// diarization, running the text lane first would spend ~30 model calls producing -// a record the diarized lane then replaces. -test("the diarized lane is ordered ahead of the text-only lane", () => { - assert.deepEqual( - resolveBackfillKinds(settingsWithAttribution(), []).map((k) => k.id), - ["attribution-diarized", "attribution-text"], - ); -}); - -// THE RUNNER'S OWN GUARD, and it is not the same test as the counter's. state() -// runs at pull time; the pool can hold a candidate for minutes afterwards, and -// the diarized lane can land a better record in that window. This asserts the -// refusal happens with NO engine call at all — no settings resolved, no model -// chosen, nothing that could fail for an unrelated reason. -test("the text lane refuses to downgrade a diarized record, without calling an engine", async () => { - const { dir, cleanup } = await attrFixture({ - attribution: attrSidecar(), - }); - try { - // A full transcript fixture, so the run gets PAST the cues guard and the - // refusal is genuinely the downgrade rule rather than a missing transcript. - await writeTranscriptFixture(dir); - const outcome = await attributeOneVideo({ - // A bogus engine: reaching it at all is the failure this test is looking - // for. The guard fires before anything resolves an app. - paths: { channelsDir: dir } as never, - videoDir: dir, - videoId: "vid1", - channelSlug: "chan", - method: "text-only", - settings: { - ...settingsWithAttribution().attribution, - appId: "no-such-engine", - }, - // Even FORCED. Forcing a regeneration is not the same as asking for a - // worse record, and nothing in the UI should be able to request the second - // by accident. - force: true, - onLog: () => {}, - }); - assert.equal(outcome, "outranked"); - } finally { - await cleanup(); - } -}); - -// mtimes have to ascend: isCuesJsonFresh compares cues.json against the metadata -// and the raw transcript, and a cues.json older than either means the transcript -// changed underneath and must not be attributed. -async function writeTranscriptFixture(dir: string): Promise<void> { - await writeFile( - path.join(dir, META_FILENAME), - JSON.stringify({ id: "vid1", title: "A video" }), - ); - await writeFile( - path.join(dir, CUES_JSON_FILENAME), - JSON.stringify({ - version: CUES_FILE_VERSION, - id: "vid1", - title: "A video", - cues: [{ start: 0, end: 5, text: "hello" }], - }), - ); - const base = Date.now() / 1000; - await utimes(path.join(dir, META_FILENAME), base, base); - await utimes(path.join(dir, "transcript.json"), base, base); - await utimes(path.join(dir, CUES_JSON_FILENAME), base + 10, base + 10); -} - -// --------------------------------------------------------------------------- -// The digest entry. Registered but NOT dispatched by the backfill lane — see -// laneBackfillKinds — so these tests pin the classification, which is the half -// that is now shared, and the lane declaration, which is what keeps the GPU and -// CPU lanes from serializing. -// --------------------------------------------------------------------------- - -const digestKind = getBackfillKind("digest")!; - -// The digest freshness target, hand-built rather than resolved: resolveTarget -// reads the channel's context note through paths, and these cases are about -// what state() does with a record, not about how the identity is derived. -const DIGEST_TARGET = { - target: { - appId: OLLAMA_DIGEST_APP_ID, - model: "llama3:8b", - promptVersion: 2, - contextHash: "ctx-1", - }, - sections: ["chapters"] as const, -}; - -function digestSidecar(over: Record<string, unknown> = {}): string { - return JSON.stringify({ - videoId: "vid1", - digestSchemaVersion: 1, - sections: { - chapters: { - items: [{ start: 0, title: "Intro" }], - provenance: { - appId: OLLAMA_DIGEST_APP_ID, - model: "llama3:8b", - modelRequested: "llama3:8b", - promptVersion: 2, - contextHash: "ctx-1", - }, - }, - }, - ...over, - }); -} - -async function classifyDigest(opts: { - transcript?: boolean; - cues?: boolean; - staleCues?: boolean; - sidecar?: string; - // Which sections must ALL be fresh. Defaults to the single-section shape the - // live corpus is configured with; the two-section form is what makes a - // part-done digest possible at all. - sections?: readonly string[]; -}): Promise<BackfillClassification> { - const { dir, cleanup } = await fixture({ transcript: opts.transcript }); - try { - if (opts.cues !== false && opts.transcript !== false) { - await writeTranscriptFixture(dir); - if (opts.staleCues) { - // cues.json OLDER than the raw transcript: the transcript changed - // underneath and the normalize pass owes this video a rewrite. - const base = Date.now() / 1000; - await utimes(path.join(dir, CUES_JSON_FILENAME), base - 100, base - 100); - } - } - if (opts.sidecar) { - await writeFile(path.join(dir, DIGEST_FILENAME), opts.sidecar); - } - const files = await readVideoFiles(dir, { checkUntranscribable: true }); - return await digestKind.state({ - videoDir: dir, - videoId: "vid1", - files, - target: opts.sections - ? { ...DIGEST_TARGET, sections: opts.sections } - : DIGEST_TARGET, - settings: defaultSiteSettings(), - }); - } finally { - await cleanup(); - } -} - -test("digest: a video with no transcript is BLOCKED on transcription", async () => { - // PLAN.md states "a video cannot be digested until it has a transcript" in - // prose and nothing enforced it. Now it is declared (dependsOn) and reported. - assert.equal(await classifyDigest({ transcript: false }), "blocked"); - assert.deepEqual(digestKind.dependsOn, ["transcription"]); - // And the dependency resolves to something nameable, which is the whole - // reason the externally-dispatched operations are in the catalog at all. - assert.equal(operationLabel("transcription"), "Transcription"); -}); - -test("digest: a fresh sidecar is present, a mismatched one is stale", async () => { - assert.equal( - await classifyDigest({ sidecar: digestSidecar() }), - "present", - ); - // A model change is work. This is the case the snapshot's old - // "does an ai-digest.json exist?" test got wrong, reading "All digested" - // while the batch reported the whole channel as stale. - assert.equal( - await classifyDigest({ - sidecar: digestSidecar({ - sections: { - chapters: { - items: [{ start: 0, title: "Intro" }], - provenance: { - appId: OLLAMA_DIGEST_APP_ID, - model: "llama3:70b", - modelRequested: "llama3:70b", - promptVersion: 2, - contextHash: "ctx-1", - }, - }, - }, - }), - }), - "stale", - ); -}); - -test("digest: no sidecar at all is missing, not stale", async () => { - assert.equal(await classifyDigest({}), "missing"); -}); - -test("digest: a stale cues.json defers rather than digesting superseded text", async () => { - // Digesting now would describe text that is about to be rewritten, and would - // then look fresh forever. Not a failure — the normalize pass fixes it — so - // it must not be counted as reachable work or as a broken engine. - assert.equal(await classifyDigest({ staleCues: true }), "deferred"); -}); - -test("digest: a transcript with NO cues.json defers too — the case that is 97.6% of them", async () => { - // The population this branch actually catches. Measured over 79,219 video - // dirs: 1,942 have a raw transcript and no cues.json at all, against 47 with - // a superseded one — and 1,683 of the 1,942 are a single `handling: "youtube"` - // channel, which downloads subtitles with --skip-download and so never runs - // transcribeOne, the only automatic caller of normalizeTranscript. - // - // Same classification as the stale case on purpose: one normalize pass fixes - // both, so the distinction is about COPY (see CuesFreshReason), not dispatch. - assert.equal( - await classifyDigest({ transcript: true, cues: false }), - "deferred", - ); -}); - -test("a kind that can defer says WHY, and digest's reason is not 'it clears itself'", () => { - // BackfillStage used to hardcode one sentence about the diarization duration - // cap for every kind's deferred videos at once. Correct only while diarization - // was the sole kind that could defer. - assert.match( - getBackfillKind("diarization")!.deferredHint!, - /Max audio hours/, - ); - // The claim this whole change exists to retract: the old copy said the - // normalize pass clears these on its own. Nothing runs it on its own, so the - // hint has to name the action. - const digestHint = digestKind.deferredHint!; - assert.match(digestHint, /Normalize/); - assert.doesNotMatch(digestHint, /clears? (itself|them|it)/i); -}); - -test("digest: a digest shared from a duplicate cluster counts as done", async () => { - // Worth ~11% of the sweep. If this entry disagreed with the snapshot's - // noDigest bucket here, every mirror would be regenerated. - assert.equal( - await classifyDigest({ - sidecar: digestSidecar({ derivedFrom: { videoId: "canonical" } }), - }), - "present", - ); -}); - -test("digest declares its own lane, and it is NOT the backfill queue", () => { - // THE PROPERTY THAT KEEPS THEM CONCURRENT. registry.ts runs every non-empty - // queueKey at concurrency 1, so sharing a key with diarization would make the - // GPU lane wait on the CPU lane and vice versa. - assert.notEqual(digestKind.lane.queueKey, BACKFILL_QUEUE); - assert.equal(digestKind.lane.queueKey, DIGEST_LOCAL_QUEUE); - assert.equal(getBackfillKind("diarization")!.lane.queueKey, BACKFILL_QUEUE); - // The two digest lanes are distinct from each other for the same reason. - assert.notEqual( - digestLaneFor("local-gpu").queueKey, - digestLaneFor("remote-api").queueKey, - ); - // Only the GPU lane stands aside for transcription. Yielding the metered lane - // would park something that costs nothing to keep running. - assert.equal(laneYieldsToTranscription(digestLaneFor("local-gpu")), true); - assert.equal(laneYieldsToTranscription(digestLaneFor("remote-api")), false); -}); - -test("the backfill lane never dispatches digest", () => { - // laneBackfillKinds feeds backfillBatch, the channel Backfill card and the - // dashboard instrument. Digest must be in the CATALOG and out of THAT list, - // or it both serializes behind diarization and gets double-counted. - const settings = settingsWithDiarization(); - const laneIds = laneBackfillKinds(settings).map((k) => k.id); - const allIds = allBackfillKinds(settings).map((k) => k.id); - assert.ok(allIds.includes("digest"), allIds.join(",")); - assert.ok(!laneIds.includes("digest"), laneIds.join(",")); - // Which means backfillBatch's kind resolution cannot reach it either, even - // when it is asked for by name. - assert.deepEqual(resolveBackfillKinds(settings, ["digest"]), []); -}); - -test("the catalog covers every operation, dispatched here or not", () => { - const ids = operationCatalog().map((o) => o.id); - for (const id of [ - "download", - "transcription", - "diarization", - "attribution-diarized", - "attribution-text", - "digest", - ]) { - assert.ok(ids.includes(id), `${id} missing from ${ids.join(",")}`); - } - // Every declared dependency resolves to a catalogued operation. This is the - // check that would have caught digest.dependsOn naming something that did not - // exist. - for (const op of operationCatalog()) { - for (const dep of op.dependsOn ?? []) { - assert.ok(ids.includes(dep), `${op.id} depends on unknown ${dep}`); - } - } -}); - -test("`runner` names the auto-queue runner, and only for the two that have one", () => { - // The console reads this instead of asking whether the id happens to be - // "download" or "transcription". `dispatch` cannot answer it: both of those - // are `external` WITH a runner, and a future `transcode` would be `external` - // with none — so an id-shaped guess would hand transcode the transcription - // runner's controls. - const runners = new Map(operationCatalog().map((o) => [o.id, o.runner])); - assert.equal(runners.get("download"), "download"); - assert.equal(runners.get("transcription"), "transcription"); - // Everything the sweep dispatches must leave it unset — a backfill kind with - // a runner would render a runner console over a lane no runner feeds. - for (const op of operationCatalog()) { - if (op.id === "download" || op.id === "transcription") continue; - assert.equal(op.runner, undefined, `${op.id} declares a runner`); - } -}); - -test("every catalogued operation declares a group and a cost basis", () => { - // Both are read unconditionally by the UI — the transit line groups stations - // by `group`, and every armed operation prints `costBasis` beside its - // backlog. An entry missing either renders a blank where a fact should be, - // which is the failure mode that let attribution-text sit armed at ~194,000 - // model calls while every screen called it "Backfill". - for (const op of operationCatalog()) { - assert.ok(op.group, `${op.id} has no group`); - assert.ok(op.shortLabel.length > 0, `${op.id} has no shortLabel`); - assert.ok( - op.shortLabel.length <= 12, - `${op.id} shortLabel "${op.shortLabel}" is too long for a column header`, - ); - assert.ok(op.costBasis.length > 0, `${op.id} has no costBasis`); - } -}); - -test("the three speaker operations share one group; digest does not", () => { - assert.equal(operationGroup("diarization"), "speakers"); - assert.equal(operationGroup("attribution-diarized"), "speakers"); - assert.equal(operationGroup("attribution-text"), "speakers"); - assert.equal(operationGroup("digest"), "digest"); - assert.equal(operationGroup("download"), "media"); - assert.equal(operationGroup("transcription"), "transcript"); - // An id the catalog does not know is null, NOT filed under the first group. - assert.equal(operationGroup("no-such-operation"), null); -}); - -test("a set's label is derived, so a mixed lane cannot claim one member's name", () => { - // The whole reason this is derived: the backfill lane holds three speaker - // operations today and its station can honestly say "Speakers". Add a kind - // from another group and it degrades to the generic name rather than - // continuing to advertise a label that now describes two thirds of it. - assert.equal( - operationsGroupLabel([ - "diarization", - "attribution-diarized", - "attribution-text", - ]), - "Speakers", - ); - assert.equal( - operationsGroupLabel(["diarization", "digest"]), - "Derived data", - ); - assert.equal(operationsGroupLabel([]), "Derived data"); - // Unknown ids contribute nothing rather than poisoning a single-group set. - assert.equal(operationsGroupLabel(["digest", "no-such-op"]), "Digest"); -}); - -test("a group has a station name AND a name you can put a verb in front of", () => { - // "Run speakers work" is why these are two declarations rather than one - // lower-cased derivation. The station eyebrow needs a noun; the button needs - // an object. - assert.equal( - operationsActionLabel([ - "diarization", - "attribution-diarized", - "attribution-text", - ]), - "speaker work", - ); - assert.equal(operationsActionLabel(["digest"]), "digests"); - assert.equal(operationsActionLabel([]), "derived data"); - assert.equal(operationsActionLabel(["diarization", "digest"]), "derived data"); -}); - -test("attribution-text's cost basis states the CHUNK unit, not the video", () => { - // The measured fact that hid behind the word "backfill": this lane's unit is - // the transcript chunk, so its 11,337 reachable videos are on the order of - // 194,000 model calls. Every other lane in the table is per-video, and a - // reader who assumes that of this one is wrong by ~17x. - assert.match(operationCostBasis("attribution-text"), /chunk/); - assert.match(operationCostBasis("attribution-diarized"), /per video/); - assert.match(operationCostBasis("diarization"), /per video/); - assert.equal(operationCostBasis("no-such-operation"), ""); -}); - -// The accounting rule the whole feature turns on: reachable work and -// needs-re-acquiring are never added together. -test("counts keep reachable work and needs-re-acquiring apart", () => { - const counts = emptyBackfillCounts(); - for (const state of [ - "missing", - "missing", - "stale", - "missing-input", - "missing-input", - "missing-input", - "present", - "not-applicable", - ] as BackfillClassification[]) { - addBackfillState(counts, state); - } - assert.deepEqual(counts, { - missing: 2, - stale: 1, - partial: 0, - missingInput: 3, - deferred: 0, - blocked: 0, - }); - // 3, not 6. Measured on the real corpus the difference is 835 vs 77,105, and - // reporting the larger number is what would make every surface useless. - assert.equal(reachableBackfillWork(counts), 3); -}); - -test("deferred is counted, and is NOT reachable work", async () => { - const counts = emptyBackfillCounts(); - for (const state of [ - "missing", - "deferred", - "deferred", - "stale", - ] as BackfillClassification[]) { - addBackfillState(counts, state); - } - assert.deepEqual(counts, { - missing: 1, - stale: 1, - partial: 0, - missingInput: 0, - deferred: 2, - blocked: 0, - }); - // 2, not 4. This is the assertion that keeps a capped corpus from ever reading - // as finished, and the one that fails if someone "tidies up" by folding - // deferred into the total. - assert.equal(reachableBackfillWork(counts), 2); -}); - -// --------------------------------------------------------------------------- -// PARTIAL: some sections at the current identity, some not. - -test("digest: some sections fresh and some not is PARTIAL, not stale", async () => { - // THE CASE THIS EXISTS FOR. `sections` is a setting and the pre-sweep decision - // is to turn tags on; the moment that happens every already-digested video in - // the corpus is part-done at once. Without the split all ~77,000 would read as - // `stale` — indistinguishable on a stage card from a PROMPT_VERSION bump that - // really did invalidate everything, when in fact digestVideo would regenerate - // only the tags. - assert.equal( - await classifyDigest({ - sidecar: digestSidecar(), - sections: ["chapters", "tags"], - }), - "partial", - ); - // The same sidecar against the section list it was made for is DONE, which is - // what makes the line above about the sections and not about the sidecar. - assert.equal( - await classifyDigest({ sidecar: digestSidecar(), sections: ["chapters"] }), - "present", - ); -}); - -test("digest: partial is REACHABLE work, unlike deferred and blocked", () => { - // The distinction is about what the work COSTS, never about whether the lane - // can do it. A part-done video is dispatched exactly like a stale one, so the - // sum must be unchanged by splitting them — otherwise a corpus mid-tags- - // backfill reports less work than it has. - const split = emptyBackfillCounts(); - for (const st of ["missing", "partial", "stale"] as BackfillClassification[]) { - addBackfillState(split, st); - } - assert.equal(reachableBackfillWork(split), 3); - const blockedAndDeferred = emptyBackfillCounts(); - for (const st of ["blocked", "deferred"] as BackfillClassification[]) { - addBackfillState(blockedAndDeferred, st); - } - assert.equal(reachableBackfillWork(blockedAndDeferred), 0); -}); - -test("reachableBackfillWork survives a snapshot written before `partial`", () => { - // Every snapshot on disk predates the field. Reading it as undefined and - // adding it would produce NaN, which renders as "NaN" and sorts unpredictably - // — strictly worse than under-reporting. - const old = { - missing: 2, - stale: 1, - missingInput: 5, - deferred: 0, - blocked: 0, - } as unknown as ReturnType<typeof emptyBackfillCounts>; - assert.equal(reachableBackfillWork(old), 3); -}); - -// --------------------------------------------------------------------------- -// laneEntriesOf: the read-side twin of laneBackfillKinds. - -test("laneEntriesOf keeps a digest entry OUT of the lane's sums", () => { - // THE REGRESSION THIS WHOLE DESIGN EXISTS TO PREVENT. Four surfaces summed - // Object.values(snapshot.backfill) on the assumption that the map WAS the - // backfill lane. Now that the snapshot carries an entry per catalog operation, - // that assumption would fold ~75,000 digest videos into the dashboard's - // backfill instrument, /actionable's backfill rows and the widget. - const backfill = { - diarization: { ...emptyBackfillCounts(), missing: 3, ids: [], eligible: 3 }, - digest: { - ...emptyBackfillCounts(), - missing: 75_000, - ids: [], - eligible: 75_000, - }, - }; - const lane = laneEntriesOf(backfill); - assert.equal(lane.length, 1); - assert.equal( - lane.reduce((n, e) => n + reachableBackfillWork(e), 0), - 3, - ); -}); - -test("laneEntriesOf filters by the DECLARATION, not by a hardcoded id", () => { - // `key !== "digest"` would pass the test above and leave the identical trap - // armed for the next operation registered on a lane of its own. The rule is - // the queue key, asked of the registry — so every lane kind is in, and an id - // the catalog does not know is out. - const laneIds = laneBackfillKinds(settingsWithDiarization()).map((k) => k.id); - const backfill: Record<string, ReturnType<typeof emptyBackfillCounts> & { ids: string[] }> = {}; - for (const id of [...laneIds, "digest", "some-kind-from-a-newer-build"]) { - backfill[id] = { ...emptyBackfillCounts(), missing: 1, ids: [] }; - } - assert.equal(laneEntriesOf(backfill).length, laneIds.length); - assert.ok(laneIds.length > 0, "expected at least one lane kind enabled"); - // Empty and absent are both simply nothing, never a throw: a snapshot may - // predate the field entirely. - assert.deepEqual(laneEntriesOf(undefined), []); - assert.deepEqual(laneEntriesOf({}), []); -}); - -test("laneKindEntriesOf applies the SAME filter, keyed by kind", () => { - // The keyed form exists so the corpus-wide backfill card can say WHICH kind a - // number came from: summed, this lane reads "77,952 reachable · 77,134 need - // media", where 99.5% of the first is attribution-text at ~1 model call per - // transcript CHUNK and all of the second is diarization at a few hundred audio - // passes. A breakdown that re-derived its own filter is how a "Digest" row - // ends up on the backfill card contradicting the figure above it — so the two - // are one function, and this asserts they cannot drift. - const laneIds = laneBackfillKinds(settingsWithDiarization()).map((k) => k.id); - const backfill: Record< - string, - ReturnType<typeof emptyBackfillCounts> & { ids: string[] } - > = {}; - for (const id of [...laneIds, "digest", "some-kind-from-a-newer-build"]) { - backfill[id] = { ...emptyBackfillCounts(), missing: 1, ids: [] }; - } - assert.deepEqual( - laneKindEntriesOf(backfill).map(([id]) => id).sort(), - [...laneIds].sort(), - ); - // Exactly the entries laneEntriesOf returns, in the same order. - assert.deepEqual( - laneKindEntriesOf(backfill).map(([, e]) => e), - laneEntriesOf(backfill), - ); - assert.deepEqual(laneKindEntriesOf(undefined), []); - assert.deepEqual(laneKindEntriesOf({}), []); -}); - -test("presentBackfillWork says UNKNOWN rather than zero on an old snapshot", () => { - // A 0 here would render as "nothing digested" on a fully digested channel, - // which is the most dangerous direction for a coverage number to be wrong. - assert.equal( - presentBackfillWork({ ...emptyBackfillCounts(), ids: [] }), - null, - ); - assert.equal( - presentBackfillWork({ - ...emptyBackfillCounts(), - missing: 2, - blocked: 1, - ids: [], - eligible: 10, - }), - 7, - ); -}); diff --git a/common/lib/backfillKinds.ts b/common/lib/backfillKinds.ts @@ -1,1704 +0,0 @@ -// The single source of truth for what a BACKFILL is. -// -// The problem this exists for repeats: a derived-data feature lands, and the -// corpus that already exists does not have what it needs. For diarization that -// input is AUDIO, which cleanAudioFromTranscribed deletes once a video is -// transcribed. Before this table the only catch-up was a per-channel button on -// a controller written for that one feature (controller/diarizeAll.ts) — there -// was no way to ask "how much of the corpus is missing this?", and no way to run -// catch-up alongside new-video work without one starving the other. -// -// So, in the shape jobKinds.ts already uses: adding a backfill should mean -// adding ONE entry here, and the system supplies the lane, the resource share -// and the indicator. -// -// FOUR STATES, NOT TWO, and the split is the load-bearing part. Measured on this -// corpus at the time of writing: 77,106 videos, 836 with media still on disk, 1 -// diarized. A single "remaining" number would therefore read 77,105 — and 91x of -// that is unreachable without re-downloading. The repo has already been burned by -// exactly this once: editor/app/api/widget/actionable/route.ts deliberately -// refuses to filter on `noDigest` because during the backfill that is 99.87% of -// the corpus and counting it would put every channel in the list forever. So -// `missing` (reachable now) and `missing-input` (needs re-acquiring) are -// SEPARATE numbers, everywhere, and no surface is allowed to add them together. -// -// STALENESS IS PROVENANCE, NOT AGE. `state` is derived from disk on every read -// and never stored, and a kind that records what produced its output compares -// that against what we would produce now — lib/digest.ts's isSectionFresh, whose -// absent-field-equals-today's-default trick is what stops adding a field from -// invalidating the whole corpus. Diarization writes that provenance and, before -// this, had no comparator at all: diarizeOne short-circuited on mere existence, -// so the two most likely reasons to re-run (a threshold or model change) left -// everything looking done. -// -// SERVER-ONLY, despite living in lib/. It reads the filesystem and calls a -// controller, so it is `-server.ts` in everything but name; the path is the one -// the plan named. No client component imports it — the UI is handed plain -// numbers off the channel snapshot, and labels as props. -// -// THREE ENTRIES, AND THE SECOND PAIR IS WHAT MAKES THIS AN ABSTRACTION. A -// registry with one entry is a wrapper: nothing proved that "a feature declares -// what it needs and the system supplies the lane, the share and the indicator" -// was true. Attribution is the test of it, and it passed — registering -// `attribution-diarized` and `attribution-text` lit the channel stage card, -// /actionable, the dashboard instrument and the widget strip with ZERO UI -// changes, because all four iterate snapshot.backfill[kindId]. -// -// It also exercised the parts of the shape that one entry could not: -// `missing-input` for something other than audio (diarization.json, of which -// this corpus has one), and two kinds writing the SAME FILE at different quality -// tiers — see the ordering rule above the attribution entries. -// -// WHAT IS STILL NOT HERE. The other catch-up mechanisms in the repo do not fit -// this per-video probe, and forcing them in would make the table lie: -// controller/backfillAvailability.ts is CHANNEL-scoped (one JSON map, folded -// into sync at runYtdlp.ts), and controller/normalizeAll.ts has no recorded -// provenance to compare, so its "stale" is undefined. `tier` still exists -// because it is what keeps them apart if they are ever added: `inline` folds -// into an existing pass and `lane` gets the concurrent queue and the share. - -import type { Paths } from "./paths"; -import type { - AttributionSettings, - DiarizationSettings, - SiteSettings, -} from "./settings"; -import { - BACKFILL_QUEUE, - DIGEST_LOCAL_QUEUE, - DIGEST_REMOTE_QUEUE, - TRANSCRIPTION_QUEUE, -} from "./queueKeys"; -// TYPE-ONLY, and that is what keeps this safe: the import is erased at compile -// time, so naming the runner union here cannot create a cycle no matter what -// jobs/autoQueueState.ts imports. -import type { AutoQueueKind } from "../jobs/autoQueueState"; -import { - DIGEST_FILENAME, - isDigestSectionKind, - isSectionFresh, - type DigestAppConfig, - type DigestFreshnessTarget, - type DigestItem, - type DigestLane, - type DigestProvenance, - type DigestRecord, - type DigestSectionKind, - type DigestWarning, -} from "./digest"; -import { loadDigest, writeDigestSection } from "./digest-server"; -import type { DigestContext } from "./digestContext-server"; -import { - SORTFORMER_DIARIZATION_ENGINE, - diarizationTarget, - isDiarizationFresh, - type DiarizationBackend, - type DiarizationEngineId, - type DiarizationFreshnessTarget, - type DiarizationRecord, -} from "./diarization"; -import { loadDiarization, writeDiarization } from "./diarization-server"; -import { - ATTRIBUTION_FILENAME, - isAttributionDowngrade, - isAttributionFresh, - type AttributionFreshnessTarget, - type AttributionMethod, - type AttributionRecord, -} from "./attribution"; -import { loadAttribution, writeAttribution } from "./attribution-server"; -import { - CUES_JSON_FILENAME, - DIARIZATION_FILENAME, - META_FILENAME, - WHISPER_FILENAME, - findSourceMedia, - isVideoTranscribed, - readVideoDurationSec, - type VideoFiles, -} from "./videoStatus"; -import { pickPreferredAudio } from "./mediaFiles"; -import { SAVED_VIDEO_POINTER_FILENAME } from "./savedVideo"; -import { resolveSavedVideo } from "./savedVideo-server"; -import { diarizeOneVideo } from "../controller/diarizeOne"; -// The TARGET resolver only — a settings read plus the digest app registry. -// attributeOne is loaded LAZILY inside run() below: this module is imported by -// controller/channelSnapshot.ts, which classifies every video of every channel, -// so its eager import graph sits on the editor's hot path, and a classification -// needs none of the runner's (the transcript normalizer, the markdown renderer, -// the digest prompt module, the channel-context reader). See -// controller/attributionTarget.ts for what this is and is not worth — it is a -// structural argument, not a measured speedup. -import { resolveAttributionTarget } from "../controller/attributionTarget"; -import type { AttributeOneOutcome } from "../controller/attributeOne"; -import { isCuesJsonFresh } from "../controller/normalizeTranscript"; - -// What a video's relationship to a backfill is, right now, read from disk. -// -// present — has it, at the identity we would produce now. -// stale — has it, but from a different engine/model/threshold. -// partial — has SOME of it at the current identity and not the rest. -// Only meaningful for a kind whose output has parts; today -// that is digest alone, whose sections are generated and -// compared independently. See the digest entry for why this -// is not just a nicer word for `stale`. -// missing — does not have it, and the input to produce it is HERE. -// missing-input — does not have it, and the input is gone. Reachable only by -// re-acquiring the media, which is opt-in and bounded. -// deferred — does not have it, the input is here, and the kind refuses to -// attempt it under the current configuration. Today that is -// only the diarization duration cap. NEVER summed into -// reachable work, so a capped corpus cannot read as finished. -// blocked — does not have it, and what it is waiting for is the OUTPUT OF -// ANOTHER KIND IN THIS TABLE. Not the same thing as -// missing-input, and conflating them was a real defect: see -// below. -// -// `deferred` and `blocked` follow the house rule BackfillRunOutcome = "skipped" -// already sets on the run side: a deliberate non-action gets its own counter, -// and is never folded into the work total nor reported as a failure. -// -// WHY `blocked` IS NOT `missing-input`. `missing-input` means one specific -// thing to the rest of the system: THE MEDIA IS GONE, RE-ACQUIRE IT. It is the -// population `allowRedownload` exists for, and backfillReacquire answers it by -// fetching AUDIO. `attribution-diarized` waits on diarization.json — so -// reporting that as missing-input told the operator ~73,000 videos needed media -// re-fetched, and with re-download on, the lane would spend a download per video -// fetching audio that CANNOT satisfy the wait, then cleaning it up again. The -// prerequisite is not gone; it has not been produced yet, and this same table -// knows how to produce it. -export type BackfillState = - | "present" - | "stale" - | "partial" - | "missing" - | "missing-input" - | "deferred" - | "blocked"; - -// Videos this backfill has no opinion about (not transcribed, marked -// untranscribable). Kept out of BackfillState so it can never be counted. -export type BackfillClassification = BackfillState | "not-applicable"; - -// How expensive one video is, which decides where the work runs. -// -// inline — microseconds to cheap I/O; folds into a pass that already walks the -// corpus, and never gets a lane of its own. -// lane — expensive enough to need its own queue and a resource share. -// Diarization is ~500-680 s/audio-hour of CPU. -export type BackfillCostTier = "inline" | "lane"; - -// WHERE an operation's work runs, and what it fights with while it runs. -// -// This exists because "put every backfill on BACKFILL_QUEUE" is wrong, and -// measurably so. registry.ts submits every non-empty queueKey at concurrency 1, -// so a shared key SERIALIZES. Diarization is CPU and digest is GPU/ollama; -// today they run at the same time on separate queues, and collapsing them onto -// one key would idle the GPU while the CPU works and vice versa, across a -// multi-week sweep. So the queue is declared per operation rather than owned by -// the lane, and a scheduler dispatches one job per distinct queueKey. -export type BackfillLane = { - // The registry queue key. Distinct keys are the ONLY mechanism for - // concurrency between operations; the same key is the only mechanism for - // serializing an operation against itself. - queueKey: string; - // The scarce resource one unit of this work occupies. Not decoration: it is - // what decides whether the lane must stand aside for transcription. - // - // gpu — contends with the transcription engine for VRAM. Yields. - // cpu — contends for cores. Does not yield today; its share is governed - // by the backfill lane's own weight (idle-only at the default). - // network — a metered or remote API. Contends with nothing local, so it - // must NOT yield: doing so would park a lane that was costing - // nothing to keep running. - contendsFor: "gpu" | "cpu" | "network"; -}; - -// Whether this system dispatches the operation, or merely knows about it. -// -// "backfill" — the backfill machinery pulls candidates and runs it. -// "external" — something else owns dispatch (autoRunner, the worker pool). -// Registered anyway so the catalog is complete and a dependency -// on it resolves. See EXTERNAL_OPERATIONS. -export type BackfillDispatch = "backfill" | "external"; - -export type BackfillProbe = { - videoDir: string; - videoId: string; - // Already read by the caller. Taking it rather than re-reading is what makes - // the channel snapshot's per-video classification free — see VideoFiles.entries. - files: VideoFiles; - target: unknown; - // The live settings. Passed in rather than read here so a classification stays - // a pure function of what the caller already has, and so a kind can consult a - // knob that must NOT become part of its freshness identity — the diarization - // duration cap is exactly that: `target` is compared by isDiarizationFresh, so - // putting the cap there would mark every sidecar on disk stale the moment the - // cap moved. - settings: SiteSettings; -}; - -export type BackfillRunOptions = { - paths: Paths; - videoDir: string; - videoId: string; - channelSlug: string; - target: unknown; - // Redo a `present` video anyway (an operator forcing a regeneration). - force?: boolean; - // Engine-config override for the LLM-backed kinds (attribution today). The - // fan-out passes the primary's resolved config with only `baseUrl` swapped - // to a leased endpoint — baseUrl is not part of the freshness identity, so - // which endpoint served a call can vary freely while everything that IS - // identity (model, numCtx, timeouts) travels verbatim from the primary. - appConfig?: DigestAppConfig; - // Per-kind settings injection for a UNIT EXECUTOR (see workerServer's - // startWorkerUnit). The primary resolves these and ships them in the unit - // envelope; each kind's run() threads its own field into the controller's - // existing `settings` override — so a bare executor's default settings - // (attribution disabled, empty app config = a DIFFERENT identity) can never - // leak into provenance. On the primary these stay unset and the controllers - // read live settings exactly as before. - attributionSettings?: AttributionSettings; - diarizationSettings?: DiarizationSettings; - // Channel context, injected rather than shipped as a file: the primary - // already computed the note + hash, and digest pins contextHash in its - // identity — injecting the object makes hash equality true by construction. - digestContext?: DigestContext; - onLog?: (msg: string) => void; - signal?: AbortSignal; -}; - -// Deliberately mirrors DiarizeOneOutcome's discipline: an expected condition is -// an outcome, never a throw. The batch counts these and keeps going. -export type BackfillRunOutcome = - | "done" - | "already-present" - | "missing-input" - // Nothing to do for a reason state() could not see WITHOUT a per-video file - // read — the case that matters is a transcript whose cues.json is stale, which - // costs a read of the source file and cannot be paid 77,000 times per pass. - // Counted separately from `failed` so a transient condition that resolves - // itself does not report as a broken engine. - | "skipped" - | "not-configured" - | "disabled" - | "failed"; - -// WHICH PIPELINE THIS OPERATION BELONGS TO, as a declared field rather than a -// list somebody maintains in a component. -// -// Three surfaces need this same grouping and each used to hardcode its own copy: -// the channel transit line (which station does this operation live under), the -// /channels table (which columns sit together), and the lane card (which -// operations does this queue actually hold). Three hardcoded lists is three -// places to forget when a kind is added — and the last time one was added, the -// transit line kept summing three unrelated operations into one station because -// nobody updated its list. -// -// Deriving the label from the group also means a lane holding a MIX cannot go -// stale: it falls back to "Derived data" rather than naming two of its three -// members. -export type OperationGroup = "media" | "transcript" | "digest" | "speakers"; - -// Group order: upstream first. The /channels columns and the transit line both -// lay their pipelines out in this order, so a reader moving between the two -// pages sees the same left-to-right sequence. -export const OPERATION_GROUP_ORDER: readonly OperationGroup[] = [ - "media", - "transcript", - "digest", - "speakers", -]; - -export function groupLabel(group: OperationGroup): string { - switch (group) { - case "media": - return "Media"; - case "transcript": - return "Transcript"; - case "digest": - return "Digest"; - case "speakers": - return "Speakers"; - } -} - -export type BackfillKind = { - id: string; - label: string; - // One line of UI copy: what this backfill is, in the operator's terms. - hint: string; - // The pipeline this operation belongs to. See OperationGroup. - group: OperationGroup; - // The label at COLUMN width — one or two words, for a header that has to sit - // above a 48px band on a table 68 rows deep. - // - // Declared rather than abbreviated in the component, for the same reason - // `group` is: "Speaker names (from the transcript)" cannot be shortened - // mechanically, and a map of abbreviations maintained next to a table is a - // second place to forget when a kind is added. - shortLabel: string; - // WHAT ONE UNIT OF THIS OPERATION COSTS, as a phrase — "one audio pass per - // video", "~1 model call per transcript chunk". - // - // The fact this exists to surface: an operation can be armed at enormous cost - // and read as a quiet row. attribution-text is reachable on 11,337 videos of - // one channel and roughly 194,000 model calls corpus-wide, it has completed - // ONE video, and every screen that mentioned it said only "Backfill". Printing - // the unit beside the backlog is what makes 11,337 legible. - // - // Deliberately NO threshold and no editorialising. A "this is a lot" cutoff - // would be a magic number the next operation gets wrong, and the operator is - // the one who decides what is too expensive. - costBasis: string; - // What a `deferred` video of THIS kind is waiting for, and what an operator - // can do about it. Belongs to the kind, not to the card: BackfillStage used to - // hardcode "too long to diarize under the current limit", which was correct - // only while diarization was the sole kind that could defer. Digest defers for - // an unrelated reason (no current normalized transcript), so a card summing - // several kinds' `deferred` into one hardcoded sentence now states a cause - // that is false for most of what it counts. - // - // Written as a sentence FRAGMENT completing "N videos are …", so the card - // keeps ownership of the count and its pluralization. - deferredHint?: string; - tier: BackfillCostTier; - // Where this operation's work runs. See BackfillLane — the queue key is what - // keeps CPU and GPU operations overlapping instead of taking turns. - // - // This is the DECLARED lane, which for a kind with a choice means its default. - // Prefer laneFor() when a live answer is needed. - lane: BackfillLane; - // The lane this kind would actually use under the given settings, for the - // kinds whose scarce resource is a configuration choice rather than a fact. - // Diarization is one: sherpa-onnx is CPU-only, while sortformer on the Vulkan - // backend holds ~4.4 GB of the same 8 GB card the transcription engine wants. - // - // Optional because most kinds have no choice, and `lane` is the answer for - // them. Mirrors digestLaneFor, which solved the same problem for the digest - // operation's two lanes — and which digest itself now declares. - // - // OPTIONAL IS LOAD-BEARING, not tidiness. controller/backfillBatch.ts reads - // this field's PRESENCE as the marker for "this kind's resource depends on - // settings, so nothing else is deciding it for us" and makes the whole run - // idle-only when such a kind could take the GPU. Giving every kind a laneFor - // that defaults to `lane` would therefore not be a no-op: it would enrol - // every kind in that rule and make the lane idle-only whenever a statically - // GPU-bound kind was in the run — the exact regression the guard's comment - // records. Add one only where the lane genuinely varies. - laneFor?(settings: SiteSettings): BackfillLane; - // Ids of other kinds in this table whose output this one consumes. - // - // PURELY DECLARATIVE. It does not gate anything by itself — a kind still - // decides for itself, from disk, whether its input is there, and says so by - // returning `blocked`. What declaring it buys is two things nothing else - // could: resolveBackfillKinds can order a prerequisite before its dependant - // within a single pass (so a video diarized this pass can be attributed in - // the same one, rather than waiting for whatever LATER pass happens to find - // the sidecar on disk), and a surface can say what a blocked video is waiting - // FOR rather than just that it is stuck. - // - // An id naming a kind that is absent or disabled is not an error: the - // dependency simply imposes no ordering, and the dependant keeps reporting - // `blocked` until something produces its input. - dependsOn?: readonly string[]; - // The feature's OWN gate. A disabled feature reports no backfill at all — - // otherwise every surface would advertise catch-up work for something the - // operator has switched off. - enabled(settings: SiteSettings): boolean; - // The identity we would produce now, resolved ONCE per run rather than per - // video. `unknown` here is the one erasure point in the table: each entry - // narrows it back to its own type on the line below. The alternative — making - // the whole registry generic — infects every consumer with a type parameter - // for no gain, since none of them look inside a target. - // - // ASYNC AND CHANNEL-SCOPED, since the digest entry. A digest's identity - // includes the hash of the channel's context note, which is a file read — so - // this can no longer be a pure function of settings. It is still resolved - // ONCE PER RUN, which is the property that matters (deriving it per video is - // how a counter and a runner end up disagreeing about what is stale); the - // callers simply await it now. - resolveTarget(ctx: BackfillTargetContext): unknown | Promise<unknown>; - state(probe: BackfillProbe): Promise<BackfillClassification>; - run(opts: BackfillRunOptions): Promise<BackfillRunOutcome>; - - // --- The unit-executor contract: what one unit of this kind needs, what it - // produces, and how its output lands back on the primary. --- - - // Filenames (within the video dir) a unit executor must be shipped, derived - // from the listing the caller already has. ORDERED: the executor - // materializes them in this order, and every kind lists - // transcript.cues.json LAST — isCuesJsonFresh compares mtimes, and a cues - // file written before its metadata/raw transcript reads as stale, making - // the unit silently do nothing. - inputs(files: VideoFiles): string[]; - // Filenames the run writes — what the executor's result endpoint returns. - outputs: readonly string[]; - // Apply a unit's returned output files ON THE PRIMARY, through the guarded - // writers — never a raw file copy. Both sidecar writers are - // read-modify-write (writeDigestSection preserves the other section; - // attribution re-checks the downgrade rule against the primary's CURRENT - // disk, because the unit ran against a snapshot that is minutes old). - // "invalid" = the payload is not a usable record; "refused" = a guard said - // no (which is a success of the guard, not a failure of the unit). - applyResult( - videoDir: string, - payload: Record<string, string>, - ): Promise<ApplyResultOutcome>; -}; - -export type ApplyResultOutcome = "applied" | "refused" | "invalid"; - -export type BackfillTargetContext = { - settings: SiteSettings; - paths: Paths; - channelSlug: string; -}; - -// --------------------------------------------------------------------------- -// Unit-contract helpers, shared across the kinds. -// --------------------------------------------------------------------------- - -// The resolved primary raw transcript — needed by every transcript-derived -// kind because isCuesJsonFresh compares the cues sidecar's mtime against it. -function rawTranscriptOf(files: VideoFiles): string | null { - if (files.hasWhisper) return WHISPER_FILENAME; - return files.ytVttFile; -} - -// The common transcript-derived input set: metadata + raw transcript + the -// existing output sidecar (for the freshness/downgrade checks) + CUES LAST — -// see BackfillKind.inputs for why the order is load-bearing. -function transcriptUnitInputs( - files: VideoFiles, - extras: readonly string[], -): string[] { - const out: string[] = []; - if (files.entries.includes(META_FILENAME)) out.push(META_FILENAME); - const raw = rawTranscriptOf(files); - if (raw) out.push(raw); - for (const name of extras) { - if (files.entries.includes(name)) out.push(name); - } - if (files.entries.includes(CUES_JSON_FILENAME)) out.push(CUES_JSON_FILENAME); - return out; -} - -// Shared by both attribution kinds: parse, validate the same shape -// loadAttribution enforces, RE-CHECK the downgrade rule against the primary's -// current disk (the unit ran against a snapshot minutes old, and the diarized -// lane can have landed a better record meanwhile), then the guarded writer. -async function applyAttributionResult( - videoDir: string, - payload: Record<string, string>, -): Promise<ApplyResultOutcome> { - const raw = payload[ATTRIBUTION_FILENAME]; - if (!raw) return "invalid"; - let record: AttributionRecord; - try { - const parsed = JSON.parse(raw) as Partial<AttributionRecord>; - if ( - typeof parsed?.videoId !== "string" || - typeof parsed.generatedAt !== "string" || - !Array.isArray(parsed.speakers) || - !Array.isArray(parsed.segments) || - !parsed.provenance || - typeof parsed.provenance.method !== "string" - ) { - return "invalid"; - } - record = parsed as AttributionRecord; - } catch { - return "invalid"; - } - if ( - isAttributionDowngrade( - await loadAttribution(videoDir), - record.provenance.method as AttributionMethod, - ) - ) { - return "refused"; - } - await writeAttribution(videoDir, record); - return "applied"; -} - -// Digest results land SECTION-WISE through writeDigestSection, which preserves -// the other section, its warnings and the history from the record already on -// the primary's disk — a raw copy of the unit's file would clobber a section -// the primary wrote while the unit was in flight. -async function applyDigestResult( - videoDir: string, - payload: Record<string, string>, -): Promise<ApplyResultOutcome> { - const raw = payload[DIGEST_FILENAME]; - if (!raw) return "invalid"; - let parsed: Partial<DigestRecord>; - try { - parsed = JSON.parse(raw) as Partial<DigestRecord>; - } catch { - return "invalid"; - } - const sections = parsed?.sections; - if (!sections || typeof sections !== "object") return "invalid"; - let applied = false; - for (const [section, value] of Object.entries(sections)) { - if (!isDigestSectionKind(section)) continue; - const v = value as { - provenance?: DigestProvenance; - items?: DigestItem[]; - }; - if (!v?.provenance || !Array.isArray(v.items)) continue; - const warnings: DigestWarning[] = (parsed.warnings ?? []).filter( - (w) => w.section === section, - ); - await writeDigestSection(videoDir, { - section, - items: v.items, - provenance: v.provenance, - warnings, - }); - applied = true; - } - return applied ? "applied" : "invalid"; -} - -// Diarization is the ONE whole-file verbatim apply: the sidecar has a single -// writer and no sections to merge, so the shape check plus the atomic writer -// is the whole guard. -async function applyDiarizationResult( - videoDir: string, - payload: Record<string, string>, -): Promise<ApplyResultOutcome> { - const raw = payload[DIARIZATION_FILENAME]; - if (!raw) return "invalid"; - try { - const parsed = JSON.parse(raw) as Partial<DiarizationRecord>; - if ( - typeof parsed?.generatedAt !== "string" || - !Array.isArray(parsed.turns) - ) { - return "invalid"; - } - await writeDiarization(videoDir, parsed as DiarizationRecord); - return "applied"; - } catch { - return "invalid"; - } -} - -// Diarization: the first entry, and the reason the table exists. -const diarization: BackfillKind = { - id: "diarization", - label: "Speaker diarization", - hint: "Speaker turns captured from the audio, written to diarization.json beside the transcript.", - group: "speakers", - shortLabel: "Diarize", - costBasis: "one pass over the audio per video", - // The wording BackfillStage used to hardcode for every kind at once. - deferredHint: - "too long to diarize under the current limit — raise or clear Max audio hours in Settings to include them", - tier: "lane", - // CPU, on the shared backfill queue. Serialized against the other backfill - // kinds on purpose — two channels' worth of diarization at once just thrashes - // cores. How much of the machine it may take is backfillLimit()'s question, - // not the queue's. - // The DEFAULT engine's answer, written out rather than derived: settings.ts is - // a TYPE-only import here, and pulling defaultDiarization() in as a value would - // make this module's initialization depend on it at runtime. laneFor is the - // live answer, and diarizationLaneFor is where the rule actually lives. - lane: { queueKey: BACKFILL_QUEUE, contendsFor: "cpu" }, - laneFor: (settings) => diarizationLaneFor(settings.diarization), - enabled: (settings) => - settings.diarization.enabled && - !!settings.diarization.segModel && - !!settings.diarization.embModel, - // The SAME target diarizeOne's own short-circuit uses. Two derivations would - // let the counter and the runner disagree about what is stale. - resolveTarget: ({ settings }): DiarizationFreshnessTarget => - diarizationTarget(settings.diarization), - async state({ videoDir, files, target, settings }) { - // Same eligibility as diarizeAll's transcribedOnly default: the capture lane - // exists to pair speaker turns with a transcript, and an untranscribed - // video's audio is not at risk from the cleanup sweep yet. - if (!isVideoTranscribed(files) || files.isUntranscribable) { - return "not-applicable"; - } - // Cheap negative first: no sidecar in the listing means no read at all. - if (files.hasDiarization) { - const record = await loadDiarization(videoDir); - // A malformed file reads as ABSENT here, exactly as hasDiarization() in - // diarization-server.ts treats it: a half-written sidecar must never be - // what convinces anything the work is done. - if (record) { - return isDiarizationFresh(record, target as DiarizationFreshnessTarget) - ? "present" - : "stale"; - } - } - if (!(await hasDiarizableInput(videoDir, files))) return "missing-input"; - // The duration cap, read ONLY here. This is the would-be-`missing` branch, - // which is ~835 videos corpus-wide rather than 77,000, and that gating is - // not optional: countBackfillWork calls state() for every video on every job - // start, and readVideoDurationSec reads and parses a file. - // - // Unknown duration is NOT deferred — an absent or unparseable - // metadata.info.json must not silently remove a video from the work list. - return (await isOverDiarizationCap(videoDir, settings.diarization)) - ? "deferred" - : "missing"; - }, - async run(opts) { - // diarizeOneVideo re-reads settings when none is passed, which is what we - // want: the batch may run for hours and a model change mid-run should be - // picked up. `force` is how a `stale` video gets redone at all — the - // existence short-circuit inside is now a freshness check, but an operator - // forcing a regeneration still needs to win. - const outcome = await diarizeOneVideo({ - paths: opts.paths, - videoDir: opts.videoDir, - videoId: opts.videoId, - settings: opts.diarizationSettings, - force: opts.force, - onLog: opts.onLog, - signal: opts.signal, - }); - if (outcome === "diarized") return "done"; - if (outcome === "already-exists") return "already-present"; - if (outcome === "no-audio") return "missing-input"; - if (outcome === "not-configured") return "not-configured"; - if (outcome === "disabled") return "disabled"; - return "failed"; - }, - // The one kind whose input is MEDIA: the preferred extracted audio, else the - // persisted source container. Same envelope as the transcript kinds, bigger - // files — and a reachable population of only ~836 videos, so modest use. - inputs(files) { - const out: string[] = []; - if (files.entries.includes(META_FILENAME)) out.push(META_FILENAME); - const audio = - pickPreferredAudio(files.audioFiles) ?? findSourceMedia(files.entries); - if (audio) out.push(audio); - return out; - }, - outputs: [DIARIZATION_FILENAME], - applyResult: applyDiarizationResult, -}; - -// Is there anything on disk ffmpeg could read for this video? Mirrors -// controller/diarizeOne.ts's resolveDiarizableMedia, but answered from the -// listing the caller already has so the common cases cost no I/O: -// extracted audio, then a persisted source container, then — only when the -// pointer file is actually present — the saved-video store. -async function hasDiarizableInput( - videoDir: string, - files: VideoFiles, -): Promise<boolean> { - if (files.audioFiles.length > 0) return true; - if (findSourceMedia(files.entries)) return true; - if (!files.entries.includes(SAVED_VIDEO_POINTER_FILENAME)) return false; - // The pointer exists but the stored file may not (an unmounted backup disk), - // so this last step really does have to touch the filesystem. - return (await resolveSavedVideo(videoDir)) !== null; -} - -// Is this video longer than the diarization duration cap? -// -// THE CAP IS OFF BY DEFAULT NOW — windowed diarization removed the OOM it -// existed for. It remains because a smaller machine, or a recording longer than -// anything measured here, may still want it. This function is where its two -// honest limitations live. First, duration is a PROXY: the memory blowup is O(n^2) in -// speech-SEGMENT count, and turn density varies 40x across this corpus, so a -// sparse 7h42m video is cheaper than a dense 6h12m one. Duration is used anyway -// because it is the only predictor available from metadata already on disk, for -// free, before committing 45 minutes of CPU to find out the hard way. Second, -// duration is the CONTAINER's, so a video whose metadata is missing or lies gets -// the benefit of the doubt. -// -// Unknown duration therefore returns false — not deferred. Deferring on an -// unreadable metadata.info.json would quietly delete work from the list on the -// strength of a file that could not be parsed, which is the opposite of what a -// third counter is for. -async function isOverDiarizationCap( - videoDir: string, - diarization: SiteSettings["diarization"], -): Promise<boolean> { - const capHours = diarization.maxAudioHours; - if (!capHours || capHours <= 0) return false; // cap off - const seconds = await readVideoDurationSec(videoDir); - if (seconds === null) return false; - return seconds > capHours * 3600; -} - -// --------------------------------------------------------------------------- -// Attribution — the second and third entries, and the ones that make this a -// registry rather than a wrapper around diarization. -// -// TWO KINDS, ONE FILE. Both write attribution.json, and the ordering rule in -// lib/attribution.ts is the whole safety of that: the diarized lane may -// overwrite a text-only record (an UPGRADE — that is what the second kind is -// for), and the text lane must never overwrite a diarized one (a DOWNGRADE). -// state() encodes it here and attributeOne re-checks it against disk immediately -// before writing, because the pool can hold a candidate for minutes after -// state() ran. It has a test; a comment would not have been enough. -// -// THE SPLIT IS ALSO WHAT MAKES THE UPGRADE QUEUE FREE. PLAN.md describes a -// bespoke "upgrade job" for turning text-only records into diarized ones. It is -// not needed: `attribution-diarized` reports a text-only record as MISSING work, -// so the existing lane, sweep and indicators queue the upgrade with no new -// machinery. What it needs re-acquiring media for is already -// controller/backfillReacquire.ts. -// -// AND IT IS WHY missing-input MATTERS HERE MOST. `attribution-diarized`'s input -// is diarization.json, of which this corpus has ONE. So its missing-input -// population is ~73,000 videos on day one — the exact case the reachable / -// needs-input split exists to stop from poisoning every surface. A single -// "remaining" number would put every channel at the top of every list forever. - -// Shared by both attribution kinds: a video only has speakers worth naming if it -// has a transcript. Not-applicable rather than missing-input, since re-acquiring -// media would not help — the video needs transcribing, which is another lane's -// job entirely. -function attributionApplies(files: VideoFiles): boolean { - return isVideoTranscribed(files) && !files.isUntranscribable; -} - -// The identity, minus the per-video half. The diarized lane's identity also -// includes the generatedAt of the diarization.json it names clusters from, and -// that is a disk read — so it is added inside the one state() branch that has -// already paid for the read. See AttributionProvenance.diarizationGeneratedAt. -function attributionTargetFor( - settings: SiteSettings, - method: AttributionMethod, -): AttributionFreshnessTarget { - return resolveAttributionTarget(method, settings.attribution).target; -} - -const attributionText: BackfillKind = { - id: "attribution-text", - label: "Speaker names (from the transcript)", - hint: "Speakers reconstructed from the transcript alone, for videos with no diarization. Cheaper to reach, worse than the diarized lane, and it never overwrites one.", - group: "speakers", - shortLabel: "Names·T", - // The expensive one, and the reason costBasis is a field. A transcript is - // many chunks; this is the only lane in the table whose unit is not the video. - costBasis: "~1 model call per transcript chunk", - tier: "lane", - lane: { queueKey: BACKFILL_QUEUE, contendsFor: "network" }, - enabled: (settings) => - settings.attribution.enabled && settings.attribution.textOnlyEnabled, - resolveTarget: ({ settings }) => attributionTargetFor(settings, "text-only"), - async state({ videoDir, files, target }) { - if (!attributionApplies(files)) return "not-applicable"; - // NEVER missing-input. The input is the cue stream, and a transcribed video - // has one by definition — which is exactly why this lane can reach the whole - // corpus and why running it over the whole corpus costs ~194,000 model calls. - if (!files.entries.includes(ATTRIBUTION_FILENAME)) return "missing"; - const record = await loadAttribution(videoDir); - // A malformed file reads as ABSENT, the same rule the diarization entry - // uses. Here it also protects the write path: a half-written sidecar must - // not be able to masquerade as a diarized record and block this lane - // forever. - if (!record) return "missing"; - // THE DOWNGRADE RULE, in the counter as well as the runner. A diarized - // record is not stale for this lane and is not work — there is simply - // something better here. Reporting it as work would put this lane in a loop - // of "attempt, refuse, still outstanding" across every pass of a sweep. - if (isAttributionDowngrade(record, "text-only")) return "present"; - return isAttributionFresh(record, target as AttributionFreshnessTarget) - ? "present" - : "stale"; - }, - async run(opts) { - // Lazy, once, at the point of actually running something. See the import - // note at the top of this file. - const { attributeOneVideo } = await import("../controller/attributeOne"); - return toBackfillOutcome( - await attributeOneVideo({ - paths: opts.paths, - videoDir: opts.videoDir, - videoId: opts.videoId, - channelSlug: opts.channelSlug, - method: "text-only", - force: opts.force, - settings: opts.attributionSettings, - appConfig: opts.appConfig, - context: opts.digestContext, - onLog: opts.onLog, - signal: opts.signal, - }), - ); - }, - inputs: (files) => transcriptUnitInputs(files, [ATTRIBUTION_FILENAME]), - outputs: [ATTRIBUTION_FILENAME], - applyResult: applyAttributionResult, -}; - -const attributionDiarized: BackfillKind = { - id: "attribution-diarized", - label: "Speaker names (from the audio)", - hint: "Names put to the speaker clusters in diarization.json — about one model call per video, and better than the text-only lane. Needs diarization to have run first.", - group: "speakers", - shortLabel: "Names·A", - costBasis: "~1 model call per video", - tier: "lane", - // The dependency the hint has always stated in prose. Declaring it is what - // turns "needs diarization to have run first" from a sentence an operator - // reads into something the scheduler can order by and a counter can name. - dependsOn: ["diarization"], - lane: { queueKey: BACKFILL_QUEUE, contendsFor: "network" }, - enabled: (settings) => - settings.attribution.enabled && settings.attribution.diarizedEnabled, - resolveTarget: ({ settings }) => attributionTargetFor(settings, "diarized"), - async state({ videoDir, files, target }) { - if (!attributionApplies(files)) return "not-applicable"; - // The input is diarization.json, NOT audio. That distinction is the whole - // reason this kind is cheap: the perishable input was already captured, and - // what is left is a naming pass that can be redone at any time. - // - // Deliberately NOT gated on settings.diarization.enabled — a sidecar - // captured during a past run is a perfectly good input after capture is - // switched off again, and refusing to name it would strand exactly the work - // the capture lane exists to protect. - // - // BLOCKED, NOT MISSING-INPUT. This used to say missing-input, and that was - // wrong in a way that cost real work: it put ~73,000 videos into the - // "re-acquire the media" population, where allowRedownload would fetch - // AUDIO — which can never satisfy a wait for diarization.json — and then - // delete it again. What this video is waiting for is the `diarization` kind - // declared in dependsOn above, and that is a thing this table produces. - if (!files.hasDiarization) return "blocked"; - if (!files.entries.includes(ATTRIBUTION_FILENAME)) return "missing"; - const record = await loadAttribution(videoDir); - if (!record) return "missing"; - // A text-only record here is THE UPGRADE QUEUE: the diarized record this - // kind is responsible for genuinely does not exist yet, so it is `missing` - // rather than `stale`. Both are reachable work, but the two words mean - // different things to an operator reading a stage card — "stale" says - // something changed under a record, "missing" says a better one was never - // made. - if (record.provenance.method !== "diarized") return "missing"; - // Only now is the diarization read worth paying for: it is needed solely to - // ask whether the clusters these names point at are still the same clusters. - const diarization = await loadDiarization(videoDir); - // Present in the listing but unreadable — a half-written or corrupt - // sidecar. Blocked for the same reason as the branch above, and note that - // the `diarization` kind reads a malformed record as ABSENT too, so it will - // regenerate this file and unblock the video without anyone intervening. - if (!diarization) return "blocked"; - return isAttributionFresh(record, { - ...(target as AttributionFreshnessTarget), - diarizationGeneratedAt: diarization.generatedAt, - }) - ? "present" - : "stale"; - }, - async run(opts) { - // Lazy, once, at the point of actually running something. See the import - // note at the top of this file. - const { attributeOneVideo } = await import("../controller/attributeOne"); - return toBackfillOutcome( - await attributeOneVideo({ - paths: opts.paths, - videoDir: opts.videoDir, - videoId: opts.videoId, - channelSlug: opts.channelSlug, - method: "diarized", - force: opts.force, - settings: opts.attributionSettings, - appConfig: opts.appConfig, - context: opts.digestContext, - onLog: opts.onLog, - signal: opts.signal, - }), - ); - }, - inputs: (files) => - transcriptUnitInputs(files, [DIARIZATION_FILENAME, ATTRIBUTION_FILENAME]), - outputs: [ATTRIBUTION_FILENAME], - applyResult: applyAttributionResult, -}; - -// One mapping, shared by both kinds, so the two lanes cannot report the same -// condition differently. -function toBackfillOutcome(outcome: AttributeOneOutcome): BackfillRunOutcome { - switch (outcome) { - case "attributed": - return "done"; - case "already-exists": - // A better record already exists. Nothing to do here is the SAME answer as - // "already current" for the lane's purposes, and reporting it as a failure - // would make an untouched corpus look broken. - case "outranked": - return "already-present"; - case "no-diarization": - return "missing-input"; - case "disabled": - return "disabled"; - // A transcript that is absent or about to be rewritten. Not a failure of - // this lane and not something re-acquiring media fixes — it resolves itself - // when the normalize pass catches up, and the next sweep pass will see it. - case "no-transcript": - return "skipped"; - default: - return "failed"; - } -} - -// --------------------------------------------------------------------------- -// Digest — the fourth entry, and the one this registry was supposed to have had -// from the start. -// -// Digests were built FIRST and never merged in, so they grew a parallel -// implementation of this same idea: their own sweep, their own per-channel -// batch, their own yield probe, their own cost planner, their own queue keys, -// their own settings block with its own pause, and their own snapshot counter. -// backfillSweep.ts is, in its own words, a clone of digestSweep.ts. Registering -// the operation is how that convergence starts. -// -// NOTHING ABOUT DIGEST GENERATION IS REWRITTEN HERE. Every function this entry -// calls is the one the digest controller already calls — resolveDigestTarget for -// the identity, isCuesJsonFresh for the transcript gate, isSectionFresh for -// freshness, digestVideo to do the work. If this entry and digestBatch ever -// disagree about whether a video is digested, that is a bug in this file, not a -// second opinion. -// -// WHAT REGISTERING ACTUALLY BUYS, today: -// -// 1. THE TRANSCRIPT DEPENDENCY BECOMES DECLARED. PLAN.md states in prose that -// "a video cannot be digested until it has a transcript" and nothing -// enforced it — an undigestable video was simply absent from every bucket. -// It now reports `blocked` on `transcription`, so it is counted and named. -// 2. A DECLARED LANE. The digest queue keys stop being one subsystem's private -// constants and become this operation's declared lane, which is what lets a -// scheduler dispatch one job per lane and keep GPU and CPU work overlapping. -// 3. ONE DEFINITION OF "digested". The snapshot, the planner and the batch all -// derive it; this is where they can converge. -// -// IT IS DELIBERATELY NOT RUN BY backfillBatch — see backfillQueueKinds below. -const digest: BackfillKind = { - id: "digest", - label: "Digest", - hint: "Chapters and tags generated from the transcript by a local or metered model. Needs a transcript first.", - group: "digest", - shortLabel: "Digest", - costBasis: "~1 model call per transcript chunk", - // See the `deferred` branch in state() below for the measurement behind this - // wording. It says "run the normalize pass" and NOT "it clears itself", - // because nothing automatic ever will. - deferredHint: - "waiting on a normalized transcript (transcript.cues.json) that nothing produces automatically — run Normalize transcripts on the channel to make them digestable", - tier: "lane", - // The transcript, declared. Nothing in the repo enforced this before. - dependsOn: ["transcription"], - // The LOCAL lane's key is the declared one because it is the default and the - // only one enabled unless remoteEnabled is set. The metered lane runs on - // DIGEST_REMOTE_QUEUE, and the two must never share a key: one shared key - // would idle the network lane while the GPU works, across a multi-week sweep. - // `contendsFor: "gpu"` is what makes this lane — and only this lane — stand - // aside for transcription. - lane: digestLaneFor("local-gpu"), - // The live answer, for the same reason diarization has one: which lane this - // runs on is a CONFIGURATION CHOICE, not a fact about the operation, and - // `lane` above can only carry the default. laneForOperation asks this, so the - // arbiter dispatches a remote digest onto DIGEST_REMOTE_QUEUE instead of the - // local key it declares. - // - // Declaring it does NOT put digest into backfillBatch's idle-only rule, and - // the reason is worth stating because that rule keys off laneFor's PRESENCE: - // resolveBackfillKinds is filtered through laneBackfillKinds (BACKFILL_QUEUE - // only), so digest can never be among the `kinds` that guard inspects. See - // controller/backfillBatch.ts, where the same fact is written from the other - // side. - laneFor: (settings) => - digestLaneFor(settings.digest.remoteEnabled ? "remote-api" : "local-gpu"), - // Digests are gated by their own sweep/pause switches rather than a master - // "enabled" flag, so the feature is on whenever an app is configured. The - // pause is honoured at DISPATCH (digestBatch's limit()), not here: a paused - // lane must still report how much work is outstanding. - enabled: () => true, - async resolveTarget({ paths, channelSlug }) { - // The existing resolver, verbatim. This is the async, channel-scoped case - // BackfillTargetContext exists for: a digest's identity includes the hash of - // the channel's context note, which is a file read. - const { resolveDigestTarget } = await import("../controller/digestTarget"); - const resolved = await resolveDigestTarget({ paths, channelSlug }); - return { target: resolved.target, sections: resolved.sections }; - }, - async state({ videoDir, files, target }) { - const { target: freshness, sections } = target as DigestTarget; - // Untranscribable is not-applicable, exactly as the attribution kinds treat - // it: nothing will ever produce a transcript for it, so it is not blocked, - // it is out of scope. - if (files.isUntranscribable) return "not-applicable"; - // BLOCKED, not not-applicable and not missing-input. A video with no - // transcript is waiting on the transcription operation declared in - // dependsOn above — a thing this system produces. Before this it was simply - // invisible: absent from every digest bucket, so a channel of untranscribed - // videos read as fully digested. - if (!isVideoTranscribed(files)) return "blocked"; - const { fresh: cuesFresh } = await isCuesJsonFresh(videoDir); - if (!cuesFresh) { - // NO CURRENT NORMALIZED TRANSCRIPT. Digesting now would either fail for - // want of one or describe superseded text and then look fresh forever, so - // the video is held back — `deferred` is the classification with the - // matching meaning: not attempted, not broken, not counted as reachable - // work. digestVideo reports the same condition as `skipped` rather than a - // failure, for the same reason. - // - // THIS DOES NOT RESOLVE ITSELF, and an earlier version of this comment - // said it did. Measured over the whole corpus (79,219 video dirs): - // - // 1,942 have NO cues.json at all ← reason "missing" - // 47 have one that is superseded ← reason "stale" - // - // so the superseded case this branch was written for is 2.4% of what it - // actually catches. The missing case is permanent: transcribeOne is the - // ONLY automatic caller of normalizeTranscript, and a channel with - // `handling: "youtube"` fetches subtitles with --skip-download and so - // never runs it — 1,683 of the 1,942 are piratesoftware alone. It went - // unnoticed because buildIndex treats cues.json as a CACHE and silently - // re-parses the raw VTT when it is absent, so the published site is - // correct and only this lane, which has no such fallback, can see it. - // - // The fix is the normalize pass, run deliberately: normalizeChannel- - // Transcripts (controller/normalizeAll.ts), wired to a button on the - // digest stage card next to this count. Both reasons are fixed by it, - // which is why they share one classification — see CuesFreshReason. - return "deferred"; - } - const record = await loadDigest(videoDir); - // A digest SHARED from a duplicate cluster's canonical member counts as - // done. The canonical member's own freshness drives regeneration and the - // share is re-applied from it (isSharedFrom's contract) — so re-deriving it - // here would undo ~11% of the sweep's saving. Same rule as the snapshot's - // noDigest bucket, deliberately, because two definitions of "digested" is - // the exact failure this entry exists to stop. - if (record?.derivedFrom != null) return "present"; - // EVERY configured section must be fresh to count as done, matching - // countMissingDigests and the batch. What is new is that "not all of them" - // is no longer one answer. - // - // WHY `partial` IS NOT JUST A NICER WORD FOR `stale`. digestVideo does not - // regenerate a video, it regenerates SECTIONS: its `stale` list at - // digestVideo.ts:158 is `sections.filter(not fresh)`, and only those are - // generated. So a video with fresh chapters and no tags is a fraction of the - // cost of one with neither, and folding them together prices the work wrong - // in the direction that matters — the corpus is ~77,000 videos and the - // recorded surcharge for adding tags to an existing chapters pass is ~44% of - // a full pass, against 100% for a genuine re-generation. - // - // The case is not hypothetical: `sections` is a setting, and the pre-sweep - // decision is to turn tags on. The moment that happens every already-digested - // video in the corpus becomes part-done at once, and without this split it - // would read as `stale` — indistinguishable, on a stage card, from a - // PROMPT_VERSION bump that really did invalidate everything. - // - // The empty-sections case still reads `present` (0 of 0 fresh), exactly as - // the `every()` this replaces did, so a caller passing no sections is - // unchanged. - const fresh = sections.filter((section) => - isSectionFresh(record, section, freshness), - ).length; - if (fresh === sections.length) return "present"; - // No record at all cannot be part-done, and it is the one branch that must - // stay `missing`: it is what separates "never digested" from "digested and - // superseded" everywhere downstream. - if (!record) return "missing"; - return fresh > 0 ? "partial" : "stale"; - }, - async run(opts) { - // Lazy, at the point of running something: digestVideo drags in the prompt - // module, the markdown renderer and the transcript normalizer, and this - // module is imported by channelSnapshot on the editor's hot path. - const { digestVideo } = await import("../controller/digestVideo"); - const { sections } = opts.target as DigestTarget; - // BackfillRunOptions already carries channelSlug and videoId, which is - // exactly what digestVideo takes — no reshaping of the controller. - const outcome = await digestVideo({ - paths: opts.paths, - channelSlug: opts.channelSlug, - videoId: opts.videoId, - sections, - // Unit-executor injection. On the primary both are unset and digestVideo - // resolves exactly as before; on an executor the injected config (baseUrl - // stripped — the executor localises its own endpoint) and the injected - // context are what keep the identity the primary's. - ...(opts.appConfig ? { config: opts.appConfig } : {}), - ...(opts.digestContext ? { context: opts.digestContext } : {}), - force: opts.force, - onLog: opts.onLog, - signal: opts.signal, - }); - // The outcome union already lines up almost exactly. - if (outcome.status === "wrote") return "done"; - if (outcome.status === "fresh") return "already-present"; - return "skipped"; - }, - inputs: (files) => transcriptUnitInputs(files, [DIGEST_FILENAME]), - outputs: [DIGEST_FILENAME], - applyResult: applyDigestResult, -}; - -type DigestTarget = { - target: DigestFreshnessTarget; - sections: DigestSectionKind[]; -}; - -// The digest operation has TWO lanes, and which one a run uses follows from the -// engine it is configured with rather than from a separate setting. This is the -// declaration; digestBatch consults it instead of re-testing the app id, so -// "which lane must stand aside for transcription" is stated once. -// -// The queue keys must never be shared: registry.ts runs each key at concurrency -// 1, so one key would idle the network lane while the GPU works — across a -// sweep measured in weeks. -export function digestLaneFor(appLane: DigestLane): BackfillLane { - return appLane === "local-gpu" - ? // Competes with the transcription engine for the same VRAM. Yields. - { queueKey: DIGEST_LOCAL_QUEUE, contendsFor: "gpu" } - : // Metered and network-bound: it competes for nothing local, so yielding - // would park a lane that costs nothing to keep running. - { queueKey: DIGEST_REMOTE_QUEUE, contendsFor: "network" }; -} - -// Which resource a diarization run competes for, which follows from the engine -// it is configured with rather than from a separate setting — the same shape as -// digestLaneFor, and stated here so nothing has to re-test the engine id. -// -// The queue key does NOT change with the engine. Diarization serializes against -// itself either way, and giving the GPU variant its own key would only let two -// diarizations run at once — which is precisely what must not happen when each -// holds ~4.4 GB of an 8 GB card. -export function diarizationLaneFor(diarization: { - engine: DiarizationEngineId; - backend: DiarizationBackend; -}): BackfillLane { - return diarization.engine === SORTFORMER_DIARIZATION_ENGINE && - diarization.backend === "vulkan" - ? // Competes with the transcription engine for the same VRAM. Yields. - { queueKey: BACKFILL_QUEUE, contendsFor: "gpu" } - : // sherpa-onnx is ONNX/CPU, and sortformer on the CPU backend is likewise - // only after cores. Contends for CPU, whose share the backfill lane's own - // weight already governs. - { queueKey: BACKFILL_QUEUE, contendsFor: "cpu" }; -} - -// Whether a lane must stand aside while transcription is working. One rule, so -// the digest lane and any future GPU operation cannot answer it differently. -export function laneYieldsToTranscription(lane: BackfillLane): boolean { - return lane.contendsFor === "gpu"; -} - -// --------------------------------------------------------------------------- -// The operations this system knows about but does NOT dispatch. -// -// Registering them costs nothing and gets the dependency graph right: without a -// `transcription` entry, digest.dependsOn = ["transcription"] names nothing and -// no surface can say what a blocked video is waiting for. -// -// They stay externally dispatched DELIBERATELY, and the reasons are concrete -// rather than a lack of time: -// -// - A SELF-REFERENTIAL YIELD. transcriptionActivity() reports busy exactly -// when transcription is running, so a transcription lane that yielded to it -// would yield to itself. -// - NO WORKER-LEASE SURFACE. BackfillRunOptions has no notion of leasing a -// worker, of a tier, of draining, or of a partial stop — all of which the -// worker pool provides and transcription requires. -// - A SCALAR AGAINST A POOL. backfillLimit() returns one number; the worker -// pool is per-worker with individual enable flags and priorities. -// - PER-CHANNEL VS CROSS-CHANNEL SCOPE. backfillBatch is one job per channel; -// autoRunner arbitrates across every channel at once, which is the whole -// point of its policy tree. -export type ExternalOperation = { - id: string; - label: string; - hint: string; - group: OperationGroup; - shortLabel: string; - costBasis: string; - lane: BackfillLane; - dependsOn?: readonly string[]; - dispatch: "external"; - // The auto-queue runner that dispatches this, when one does. `dispatch` alone - // cannot answer it: download and transcription are `external` WITH a runner, - // and a future `transcode` would be `external` with none. A console that - // guessed from the id would hand transcode the transcription runner's - // controls — a live Start button over the wrong lane. - runner?: AutoQueueKind; -}; - -export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [ - { - id: "download", - label: "Download", - hint: "Fetching the media. Dispatched by the auto-download runner and the per-channel pipeline actions.", - group: "media", - shortLabel: "Download", - costBasis: "one fetch per video, over the network", - // Really one queue per platform (downloadQueueKey), not a single key. Named - // here as the shape rather than the exact key, because the catalog's job is - // the dependency graph, not dispatch. - lane: { queueKey: "download:<platform>", contendsFor: "network" }, - dispatch: "external", - runner: "download", - }, - { - id: "transcription", - label: "Transcription", - hint: "Turning audio into a transcript. Dispatched by the auto-transcribe runner across the worker pool.", - group: "transcript", - shortLabel: "Transcribe", - costBasis: "one pass over the audio per video, on a worker", - lane: { queueKey: TRANSCRIPTION_QUEUE, contendsFor: "gpu" }, - dependsOn: ["download"], - dispatch: "external", - runner: "transcription", - }, -]; - -// Every media-derived operation, dispatched here or not. The catalog — what a -// dependency id resolves against, and what a future scheduler enumerates. -export type OperationDescriptor = { - id: string; - label: string; - hint: string; - group: OperationGroup; - shortLabel: string; - costBasis: string; - lane: BackfillLane; - dependsOn?: readonly string[]; - dispatch: BackfillDispatch; - // See ExternalOperation.runner. Absent for every backfill kind: those are - // dispatched by the sweep and the arbiter, not by a runner. - runner?: AutoQueueKind; -}; - -export function operationCatalog(): OperationDescriptor[] { - return [ - ...EXTERNAL_OPERATIONS, - ...BACKFILL_KINDS.map((k) => ({ - id: k.id, - label: k.label, - hint: k.hint, - group: k.group, - shortLabel: k.shortLabel, - costBasis: k.costBasis, - lane: k.lane, - dependsOn: k.dependsOn, - dispatch: "backfill" as const, - })), - ]; -} - -// The label for a dependency id, from anywhere in the catalog. Returns the id -// itself for something unknown rather than throwing — a dangling dependency is -// already tolerated everywhere else here. -export function operationLabel(id: string): string { - return operationCatalog().find((o) => o.id === id)?.label ?? id; -} - -// The group an operation belongs to, or null for an id the catalog does not -// know. Null rather than a fallback group: a caller grouping by this must be -// able to tell "unknown" from "media", and silently filing a dangling id under -// the first group would put it on the wrong station. -export function operationGroup(id: string): OperationGroup | null { - return operationCatalog().find((o) => o.id === id)?.group ?? null; -} - -// What one unit of an operation costs, in words. Empty string for an unknown -// id, so a surface can print it unconditionally without a placeholder. -export function operationShortLabel(id: string): string { - return operationCatalog().find((o) => o.id === id)?.shortLabel ?? id; -} - -export function operationCostBasis(id: string): string { - return operationCatalog().find((o) => o.id === id)?.costBasis ?? ""; -} - -// The label for a SET of operations — a lane card, a transit-line station, a -// column group. One group means that group's name; a mix means "Derived data". -// -// DERIVED, never hardcoded, which is the point: a lane that gains a kind from a -// different group degrades to the honest generic name instead of continuing to -// advertise a label that now describes two thirds of what it holds. -// The same set, named as a THING AN OPERATOR RUNS rather than as a stage on a -// line — "Run speaker work", "N videos are waiting on digests". -// -// Two labels for one group is not duplication: "Speakers" is a station on a -// transit line and has to be a noun at eyebrow width; "speaker work" is the -// object of a verb and has to survive being lower-cased into a sentence. The -// alternative — deriving one from the other — produces "Run speakers work", -// which is why this is declared. -export function groupActionLabel(group: OperationGroup): string { - switch (group) { - case "media": - return "downloads"; - case "transcript": - return "transcripts"; - case "digest": - return "digests"; - case "speakers": - return "speaker work"; - } -} - -export function operationsActionLabel(ids: ReadonlyArray<string>): string { - const groups = new Set<OperationGroup>(); - for (const id of ids) { - const group = operationGroup(id); - if (group) groups.add(group); - } - if (groups.size !== 1) return "derived data"; - return groupActionLabel([...groups][0]); -} - -export function operationsGroupLabel(ids: ReadonlyArray<string>): string { - const groups = new Set<OperationGroup>(); - for (const id of ids) { - const group = operationGroup(id); - if (group) groups.add(group); - } - if (groups.size !== 1) return "Derived data"; - return groupLabel([...groups][0]); -} - -// The digest operation's id, named once. Surfaces that read one specific -// operation off a snapshot (the digest stage card, the dashboard's coverage -// instrument) need this string, and a typo in it fails the way a missing -// snapshot entry does — silently, as "nothing to do". -export const DIGEST_KIND_ID = "digest"; - -// The diarization operation's id, named once for the same reason. The cleanup -// accounting needs it to ask a question no other surface asks: whether the lane -// that would release a held video is even running (allBackfillKinds drops a kind -// whose enabled() is false, so an ABSENT entry is the answer, not a zero). -export const DIARIZATION_KIND_ID = "diarization"; - -// One entry per backfill known to the system. -export const BACKFILL_KINDS: readonly BackfillKind[] = [ - diarization, - attributionDiarized, - // Text-only LAST, deliberately. resolveBackfillKinds preserves this order and - // backfillBatch walks the kinds in it, so on a video that has diarization the - // cheap, better lane gets there first and the text lane then finds a record it - // must not overwrite — one wasted classification instead of ~30 model calls. - attributionText, - // Digest is in the table, but on its own lane — so it is counted and - // classified by everything that reads this registry, and dispatched by none - // of it. See backfillQueueKinds. - digest, -]; - -export const BACKFILL_KIND_BY_ID: Record<string, BackfillKind> = - Object.fromEntries(BACKFILL_KINDS.map((k) => [k.id, k])); - -export function getBackfillKind(id: string): BackfillKind | undefined { - return BACKFILL_KIND_BY_ID[id]; -} - -// Every registered kind whose feature is switched on, whatever lane it runs on. -// The catalog view: what EXISTS and is live, for a scheduler or a dependency -// lookup. Callers that mean "the backfill lane" want laneBackfillKinds below. -export function allBackfillKinds(settings: SiteSettings): BackfillKind[] { - return BACKFILL_KINDS.filter((k) => k.tier === "lane" && k.enabled(settings)); -} - -// THE BACKFILL LANE's kinds: enabled, `lane` tier, and running on the shared -// BACKFILL_QUEUE. Three filters, and the third is new with the digest entry. -// -// The queue filter is a SAFETY RAIL, not a tidy-up. Everything downstream of -// this function — backfillBatch's dispatch, the channel Backfill card, the -// dashboard's backfill instrument, /actionable's backfill rows — treats these -// as "one lane, one job, one set of counters". Digest satisfies none of that: -// -// - DISPATCH. backfillBatch runs its kinds in one job on one queue under -// backfillLimit(). Handing it digest would SERIALIZE the GPU digest lane -// behind CPU diarization, when the entire reason they hold separate queue -// keys is that they currently overlap. -// - GUARDS. digest carries digestsPaused, the yield-to-transcription -// carve-out (with its CPU-worker exemption), spendCapUsd on the metered -// lane, the remoteEnabled fail-fast, shortest-first ordering, -// duplicate-cluster sharing and the engine probe() fail-fast. Those are -// measured decisions in digestBatch's limit(), and not one of them is -// expressible as backfillLimit()'s single scalar. -// - COUNTERS. Digest already has its own instrument, its own stage card and -// its own snapshot bucket. Folding it in here would double-count it against -// surfaces that are live on a 78,000-video corpus mid-sweep. -// -// So digest is CLASSIFIED and CATALOGUED through this registry and DISPATCHED -// through its own controller. Collapsing the two sets of counters and the two -// schedulers is the unified-rule-model work, not this function's job. A kind -// joins this list when its lane rule can be expressed here without losing a -// guard. -export function laneBackfillKinds(settings: SiteSettings): BackfillKind[] { - return allBackfillKinds(settings).filter( - (k) => k.lane.queueKey === BACKFILL_QUEUE, - ); -} - -// The READ-SIDE twin of laneBackfillKinds: given a snapshot's per-kind map, -// return only the entries belonging to the shared backfill lane — KEYED, so a -// surface can say WHICH kind a number came from. laneEntriesOf below is this -// with the ids dropped, for the callers that only sum. -// -// The keyed form is what the corpus-wide backfill card needs. Summed, this lane -// reads "77,952 reachable · 77,134 need media" — both figures correct, and -// together meaningless: 99.5% of the first is attribution-text (one model call -// per transcript CHUNK) and all of the second is diarization (329 runs). Adding -// kinds gives a number in no unit at all, which is the mistake the header -// forbids one level up for `missing` vs `missing-input`. -// -// THIS EXISTS BECAUSE THE SNAPSHOT MAP STOPPED BEING THE LANE. It used to be -// written from laneBackfillKinds, so `Object.values(snapshot.backfill)` and "the -// backfill lane" were the same set by construction, and four surfaces summed it -// generically on that basis — the channel dashboard's backfill instrument, -// /actionable's two backfill functions and the widget's sync payload. The moment -// channelSnapshot writes an entry per CATALOG operation, that identity breaks: -// those four would silently absorb ~75,000 digest videos into a number that has -// only ever meant diarization plus attribution. -// -// FILTERED BY THE DECLARATION, NOT BY ID. `key !== "digest"` would fix today and -// leave the identical trap armed for the next operation registered on a lane of -// its own — which is the whole direction of the unified-operations work. The -// rule is the same one laneBackfillKinds applies on the write side, asked of the -// registry: does this kind run on BACKFILL_QUEUE? -// -// An id the catalog does not know is EXCLUDED. A snapshot is a file on disk that -// may have been written by an older build and may name a kind that has since -// been renamed or removed; there is no lane declaration to check it against, so -// it cannot be asserted to belong to this one. Deliberately not filtered on -// `enabled(settings)` — these are counts already written to disk, and a feature -// switched off after a snapshot was taken does not retroactively unmake the work -// it recorded. -export function laneKindEntriesOf<T>( - backfill: Record<string, T> | undefined | null, -): [string, T][] { - if (!backfill) return []; - return Object.entries(backfill).filter( - ([id]) => getBackfillKind(id)?.lane.queueKey === BACKFILL_QUEUE, - ); -} - -// The same set with the ids dropped, for the callers that only ever sum. Defined -// in terms of the above rather than beside it: the filter and every word of the -// rule above it must stay in ONE place, or the next surface that wants per-kind -// detail copies a `key !== "digest"` in and re-arms the trap. -export function laneEntriesOf<T>( - backfill: Record<string, T> | undefined | null, -): T[] { - return laneKindEntriesOf(backfill).map(([, entry]) => entry); -} - -// Resolve a caller-supplied list of kind ids against the registry. An empty or -// absent list means "every enabled lane kind" — the sweep's scope default. -// Unknown ids are dropped rather than throwing: a settings file may name a kind -// from a newer build, and a stale scope must not wedge the lane. -export function resolveBackfillKinds( - settings: SiteSettings, - ids: readonly string[] | undefined, -): BackfillKind[] { - const lane = laneBackfillKinds(settings); - const selected = - !ids || ids.length === 0 - ? lane - : lane.filter((k) => new Set(ids).has(k.id)); - return orderByDependencies(selected); -} - -// Order kinds so a prerequisite is attempted before anything that declares it. -// -// WHY THIS IS WORTH DOING AT ALL. backfillBatch walks the kinds in the order it -// is given, all the way through the video list, before starting the next kind. -// So with `attribution-diarized` ahead of `diarization`, a video diarized -// during a pass becomes eligible for attribution only on whatever LATER pass -// happens to find the sidecar on disk. Ordering by the declaration collapses -// that into one pass, and costs a topological sort over three entries. -// -// STABLE, and that is load-bearing rather than tidiness. BACKFILL_KINDS puts -// attribution-text LAST on purpose (see the comment there): on a video that has -// diarization, the better lane must get there first so the text lane finds a -// record it must not overwrite — one wasted classification instead of ~30 model -// calls. Kahn's algorithm with a queue seeded and drained in declaration order -// preserves every ordering the declarations do not contradict, so that decision -// survives. -export function orderByDependencies(kinds: BackfillKind[]): BackfillKind[] { - const byId = new Map(kinds.map((k) => [k.id, k])); - // Only dependencies that are actually IN this selection constrain anything. A - // kind that names a disabled or unselected prerequisite is not held back — - // it will report `blocked` per video, which is the honest answer, rather than - // being silently dropped from the run. - const remaining = new Map( - kinds.map((k) => [ - k.id, - (k.dependsOn ?? []).filter((d) => byId.has(d) && d !== k.id).length, - ]), - ); - const dependants = new Map<string, string[]>(); - for (const k of kinds) { - for (const d of k.dependsOn ?? []) { - if (!byId.has(d) || d === k.id) continue; - const list = dependants.get(d); - if (list) list.push(k.id); - else dependants.set(d, [k.id]); - } - } - const out: BackfillKind[] = []; - const emitted = new Set<string>(); - // Repeatedly take the FIRST still-unemitted kind in declaration order whose - // prerequisites are all out. Quadratic in the number of kinds, which is three. - for (;;) { - const next = kinds.find( - (k) => !emitted.has(k.id) && (remaining.get(k.id) ?? 0) === 0, - ); - if (!next) break; - emitted.add(next.id); - out.push(next); - for (const id of dependants.get(next.id) ?? []) { - remaining.set(id, (remaining.get(id) ?? 1) - 1); - } - } - // A CYCLE leaves entries unemitted. Append them in declaration order rather - // than throwing or dropping them: a mis-declared dependency should degrade to - // the old behaviour (run in table order), never wedge the lane or silently - // stop a backfill from running at all. - for (const k of kinds) if (!emitted.has(k.id)) out.push(k); - return out; -} - -// Per-kind counts, the shape every indicator reads. `missing` and `missingInput` -// are never summed — see the header. -export type BackfillCounts = { - missing: number; - stale: number; - missingInput: number; - // Work the kind is refusing to attempt under the current configuration (the - // diarization duration cap). A THIRD number, alongside the other two that are - // never summed. Snapshots written before this field existed do not carry it, - // so every read site needs `?? 0` — `.toLocaleString()` on undefined throws. - deferred: number; - // Waiting on a prerequisite kind's output. A FOURTH number, and the same rule - // applies: never summed with the others, and `?? 0` at every read site, - // because every snapshot currently on disk predates it. - // - // This number should FALL on its own as the prerequisite lane works, which is - // the whole difference from missingInput — that one only falls if an operator - // turns re-download on. - blocked: number; - // Reachable work that is PART DONE. Unlike deferred and blocked, this one IS - // summed into reachableBackfillWork — it is work the lane can do today. It is - // split out of `stale` because the two cost different amounts and want - // different decisions: see the digest entry's state(). - // - // `?? 0` at every read site, like deferred and blocked before it. Every - // snapshot currently on disk predates this field. - partial: number; -}; - -export function emptyBackfillCounts(): BackfillCounts { - return { - missing: 0, - stale: 0, - partial: 0, - missingInput: 0, - deferred: 0, - blocked: 0, - }; -} - -// Fold one classification into a counts record. Central so no surface invents -// its own accounting: `present` and `not-applicable` add to nothing, which is -// what makes these counts a WORK LIST rather than a coverage measure. -export function addBackfillState( - counts: BackfillCounts, - state: BackfillClassification, -): void { - if (state === "missing") counts.missing++; - else if (state === "stale") counts.stale++; - else if (state === "partial") counts.partial++; - else if (state === "missing-input") counts.missingInput++; - else if (state === "deferred") counts.deferred++; - else if (state === "blocked") counts.blocked++; -} - -// What the lane can act on WITHOUT re-acquiring media. The number every "how -// much is left?" surface should lead with. -// -// DELIBERATELY UNCHANGED by the addition of `deferred`, and unchanged again by -// `blocked`. This function is the guard: adding a state to the union raises no -// TypeScript error here (the exhaustiveness check lives on the DISPATCH -// decision, in backfillBatch's candidateAction, which is the branch that can do -// harm), so the only thing keeping capped and blocked videos out of the work -// total is that they are not added here. If a future state belongs in the -// total, it goes in on purpose. -// -// A blocked video is emphatically not reachable work: there is nothing this -// lane can do about it this pass. Counting it would make a corpus with one -// diarization and 73,000 waiting attributions report 73,000 jobs ready to run. -// -// `partial` IS in the total, and that is the on-purpose case the paragraph -// above reserves. A part-done video is work the lane can pick up right now and -// the run will write to it; leaving it out would make a corpus mid-tags-backfill -// report less work than it has. Splitting it from `stale` is about what the -// work COSTS, not about whether it is reachable — so the sum is unchanged from -// what it would have been before the split, which is the property that keeps -// this a refinement rather than a behaviour change. -// -// `?? 0` because every snapshot on disk predates the field, and undefined would -// poison the sum to NaN rather than merely under-report. -export function reachableBackfillWork(counts: BackfillCounts): number { - return counts.missing + counts.stale + (counts.partial ?? 0); -} - -// What a channel snapshot stores per kind: the three counts, plus the ids of the -// REACHABLE work only. -// -// The asymmetry is deliberate. A stage card has to list what it would act on, so -// those ids have to be somewhere the render path can read without walking the -// corpus (there is a guard test forbidding exactly that). But `missingInput` is -// ~76,000 videos corpus-wide, and writing that list into all 66 snapshots would -// put tens of megabytes of ids on disk to say a number we already have. So: ids -// for the actionable half, a count for the other. -export type BackfillSnapshotEntry = BackfillCounts & { - // Exactly the reachable set — missing + stale + partial — sorted. Never - // includes missing-input, deferred or blocked. Kept equal to - // reachableBackfillWork(entry) by a test, because a policy leaf hands this - // list out as work while the cards render the count. - ids: string[]; - // How many videos this operation has an OPINION about: everything it did not - // classify not-applicable. The denominator, and deliberately not part of - // BackfillCounts — those are a work list, and mixing a coverage measure into - // them is what would let a surface add "done" to "to do". - // - // It is stored rather than derived because `present` is the one classification - // addBackfillState throws away, so nothing downstream can reconstruct the - // total from the counts alone. With it, present = eligible - (every work - // count), which is what presentBackfillWork below computes. - // - // Optional: every snapshot written before this field lacks it, and a reader - // that cannot tell how many videos were considered must say so rather than - // divide by a zero it invented. - eligible?: number; -}; - -// How many videos this operation is DONE with, derived from the stored -// denominator minus every work state. Returns null when the snapshot predates -// `eligible`, because the honest answer there is "unknown" — a 0 would render as -// "nothing digested" on a fully digested channel. -export function presentBackfillWork( - entry: BackfillSnapshotEntry, -): number | null { - if (entry.eligible == null) return null; - return Math.max( - 0, - entry.eligible - - (entry.missing + - entry.stale + - (entry.partial ?? 0) + - entry.missingInput + - (entry.deferred ?? 0) + - (entry.blocked ?? 0)), - ); -} diff --git a/common/lib/backfillUnit.test.ts b/common/lib/backfillUnit.test.ts @@ -1,228 +0,0 @@ -import { test } from "node:test"; -import assert from "node:assert/strict"; -import { mkdtemp, readFile } from "node:fs/promises"; -import { tmpdir } from "node:os"; -import path from "node:path"; -import { - BACKFILL_KINDS, - getBackfillKind, -} from "./backfillKinds"; -import { - ATTRIBUTION_FILENAME, - type AttributionRecord, -} from "./attribution"; -import { loadAttribution, writeAttribution } from "./attribution-server"; -import { DIGEST_FILENAME, type DigestProvenance } from "./digest"; -import { loadDigest, writeDigestSection } from "./digest-server"; -import { loadDiarization } from "./diarization-server"; -import { - CUES_JSON_FILENAME, - DIARIZATION_FILENAME, - META_FILENAME, - WHISPER_FILENAME, - type VideoFiles, -} from "./videoStatus"; - -// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/backfillUnit.test.ts -// -// The unit-executor contract: every lane kind declares what a unit needs -// (inputs), what it produces (outputs), and how its output lands back on the -// primary (applyResult — through the GUARDED writers, never a raw copy). The -// two guards that must hold whatever a unit returns: the text-only lane can -// never overwrite a diarized attribution record, and applying one digest -// section never clobbers the other. - -function files(entries: string[]): VideoFiles { - return { - hasMeta: entries.includes(META_FILENAME), - hasYtVtt: entries.includes("transcript.en.vtt"), - ytVttFile: entries.includes("transcript.en.vtt") - ? "transcript.en.vtt" - : null, - hasNonCanonicalVtt: false, - hasWhisper: entries.includes(WHISPER_FILENAME), - hasCuesJson: entries.includes(CUES_JSON_FILENAME), - hasDiarization: entries.includes(DIARIZATION_FILENAME), - isUntranscribable: false, - audioFiles: entries.filter((e) => e.startsWith("audio.")), - partAudioFiles: [], - entries, - }; -} - -const FULL_DIR = files([ - META_FILENAME, - WHISPER_FILENAME, - CUES_JSON_FILENAME, - DIARIZATION_FILENAME, - ATTRIBUTION_FILENAME, - DIGEST_FILENAME, - "audio.mp3", -]); - -test("every lane kind declares the full unit contract", () => { - for (const kind of BACKFILL_KINDS) { - assert.equal(typeof kind.inputs, "function", `${kind.id} inputs`); - assert.ok(kind.outputs.length > 0, `${kind.id} outputs`); - assert.equal(typeof kind.applyResult, "function", `${kind.id} applyResult`); - // Outputs are the SIDECAR the kind writes — never the shipped transcript - // or metadata, which applying would silently rewrite on the primary. - for (const out of kind.outputs) { - assert.ok( - out !== CUES_JSON_FILENAME && - out !== META_FILENAME && - out !== WHISPER_FILENAME, - `${kind.id} output ${out} collides with a shipped input`, - ); - } - } -}); - -test("every transcript-derived kind ships transcript.cues.json LAST", () => { - // The executor materializes inputs in order and isCuesJsonFresh compares - // mtimes: a cues file written before its metadata/raw transcript reads as - // stale and the unit silently does nothing. - for (const id of ["attribution-text", "attribution-diarized", "digest"]) { - const inputs = getBackfillKind(id)!.inputs(FULL_DIR); - assert.equal( - inputs[inputs.length - 1], - CUES_JSON_FILENAME, - `${id} must list cues last, got: ${inputs.join(", ")}`, - ); - assert.ok(inputs.includes(META_FILENAME), `${id} ships metadata`); - assert.ok(inputs.includes(WHISPER_FILENAME), `${id} ships the raw transcript`); - } -}); - -test("the diarized attribution unit ships diarization.json; diarization ships audio", () => { - assert.ok( - getBackfillKind("attribution-diarized")! - .inputs(FULL_DIR) - .includes(DIARIZATION_FILENAME), - ); - assert.ok(getBackfillKind("diarization")!.inputs(FULL_DIR).includes("audio.mp3")); -}); - -function attributionPayload(method: "text-only" | "diarized"): string { - const record: AttributionRecord = { - videoId: "vid", - generatedAt: "2026-08-24T00:00:00.000Z", - speakers: [{ index: 0, label: "Host", seconds: 10 }], - segments: [{ start: 0, end: 10, speaker: 0 }], - provenance: { - method, - appId: "ollama-direct", - model: "m", - modelRequested: "m", - promptVersion: 2, - generatedAt: "2026-08-24T00:00:00.000Z", - durationMs: 1, - }, - }; - return JSON.stringify(record); -} - -test("applyResult refuses a text-only record over a diarized one (the downgrade rule)", async () => { - const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); - await writeAttribution( - dir, - JSON.parse(attributionPayload("diarized")) as AttributionRecord, - ); - const outcome = await getBackfillKind("attribution-text")!.applyResult(dir, { - [ATTRIBUTION_FILENAME]: attributionPayload("text-only"), - }); - assert.equal(outcome, "refused"); - assert.equal( - (await loadAttribution(dir))?.provenance.method, - "diarized", - "the diarized record must survive", - ); -}); - -test("applyResult upgrades a text-only record to a diarized one", async () => { - const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); - await writeAttribution( - dir, - JSON.parse(attributionPayload("text-only")) as AttributionRecord, - ); - const outcome = await getBackfillKind("attribution-diarized")!.applyResult( - dir, - { [ATTRIBUTION_FILENAME]: attributionPayload("diarized") }, - ); - assert.equal(outcome, "applied"); - assert.equal((await loadAttribution(dir))?.provenance.method, "diarized"); -}); - -test("applyResult rejects a malformed attribution payload", async () => { - const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); - const kind = getBackfillKind("attribution-text")!; - assert.equal(await kind.applyResult(dir, {}), "invalid"); - assert.equal( - await kind.applyResult(dir, { [ATTRIBUTION_FILENAME]: "not json" }), - "invalid", - ); - assert.equal( - await kind.applyResult(dir, { [ATTRIBUTION_FILENAME]: "{}" }), - "invalid", - ); -}); - -function digestProvenance(model: string): DigestProvenance { - return { - appId: "ollama-direct", - model, - promptVersion: 2, - contextHash: "none", - generatedAt: "2026-08-24T00:00:00.000Z", - } as DigestProvenance; -} - -test("applying one digest section preserves the other (guarded read-modify-write)", async () => { - const dir = await mkdtemp(path.join(tmpdir(), "unit-digest-")); - // The primary already holds a tags section… - await writeDigestSection(dir, { - section: "tags", - items: [], - provenance: digestProvenance("local-model"), - warnings: [], - }); - // …and a unit returns a record carrying only chapters. - const payload = JSON.stringify({ - sections: { - chapters: { provenance: digestProvenance("m"), items: [] }, - }, - }); - const outcome = await getBackfillKind("digest")!.applyResult(dir, { - [DIGEST_FILENAME]: payload, - }); - assert.equal(outcome, "applied"); - const record = await loadDigest(dir); - assert.ok(record?.sections?.chapters, "the unit's section landed"); - assert.ok( - record?.sections?.tags, - "the section the primary wrote must survive the apply", - ); - assert.equal(record?.sections?.tags?.provenance.model, "local-model"); -}); - -test("diarization results apply verbatim (single-writer sidecar) after a shape check", async () => { - const dir = await mkdtemp(path.join(tmpdir(), "unit-diar-")); - const kind = getBackfillKind("diarization")!; - assert.equal( - await kind.applyResult(dir, { [DIARIZATION_FILENAME]: "{}" }), - "invalid", - ); - const outcome = await kind.applyResult(dir, { - [DIARIZATION_FILENAME]: JSON.stringify({ - videoId: "vid", - generatedAt: "2026-08-24T00:00:00.000Z", - speakers: 1, - turns: [{ start: 0, end: 5, speaker: 0 }], - }), - }); - assert.equal(outcome, "applied"); - const raw = await readFile(path.join(dir, DIARIZATION_FILENAME), "utf8"); - assert.equal((JSON.parse(raw) as { speakers: number }).speakers, 1); - // And the server-side loader accepts what applyResult wrote. - assert.notEqual(await loadDiarization(dir), null); -}); diff --git a/common/lib/diarization.ts b/common/lib/diarization.ts @@ -164,7 +164,7 @@ function baseName(p: string): string { // settings import (settings.ts imports the threshold default FROM here). // // One definition, two callers — controller/diarizeOne.ts's short-circuit and -// lib/backfillKinds.ts's state() — because a comparator and the writer it +// lib/operations.ts's state() — because a comparator and the writer it // guards disagreeing about the identity is how a corpus ends up either // regenerating forever or never. export function diarizationTarget(cfg: { diff --git a/common/lib/operationUnit.test.ts b/common/lib/operationUnit.test.ts @@ -0,0 +1,228 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdtemp, readFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { + OPERATIONS, + getOperation, +} from "./operations"; +import { + ATTRIBUTION_FILENAME, + type AttributionRecord, +} from "./attribution"; +import { loadAttribution, writeAttribution } from "./attribution-server"; +import { DIGEST_FILENAME, type DigestProvenance } from "./digest"; +import { loadDigest, writeDigestSection } from "./digest-server"; +import { loadDiarization } from "./diarization-server"; +import { + CUES_JSON_FILENAME, + DIARIZATION_FILENAME, + META_FILENAME, + WHISPER_FILENAME, + type VideoFiles, +} from "./videoStatus"; + +// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test lib/backfillUnit.test.ts +// +// The unit-executor contract: every lane kind declares what a unit needs +// (inputs), what it produces (outputs), and how its output lands back on the +// primary (applyResult — through the GUARDED writers, never a raw copy). The +// two guards that must hold whatever a unit returns: the text-only lane can +// never overwrite a diarized attribution record, and applying one digest +// section never clobbers the other. + +function files(entries: string[]): VideoFiles { + return { + hasMeta: entries.includes(META_FILENAME), + hasYtVtt: entries.includes("transcript.en.vtt"), + ytVttFile: entries.includes("transcript.en.vtt") + ? "transcript.en.vtt" + : null, + hasNonCanonicalVtt: false, + hasWhisper: entries.includes(WHISPER_FILENAME), + hasCuesJson: entries.includes(CUES_JSON_FILENAME), + hasDiarization: entries.includes(DIARIZATION_FILENAME), + isUntranscribable: false, + audioFiles: entries.filter((e) => e.startsWith("audio.")), + partAudioFiles: [], + entries, + }; +} + +const FULL_DIR = files([ + META_FILENAME, + WHISPER_FILENAME, + CUES_JSON_FILENAME, + DIARIZATION_FILENAME, + ATTRIBUTION_FILENAME, + DIGEST_FILENAME, + "audio.mp3", +]); + +test("every lane kind declares the full unit contract", () => { + for (const kind of OPERATIONS) { + assert.equal(typeof kind.inputs, "function", `${kind.id} inputs`); + assert.ok(kind.outputs.length > 0, `${kind.id} outputs`); + assert.equal(typeof kind.applyResult, "function", `${kind.id} applyResult`); + // Outputs are the SIDECAR the kind writes — never the shipped transcript + // or metadata, which applying would silently rewrite on the primary. + for (const out of kind.outputs) { + assert.ok( + out !== CUES_JSON_FILENAME && + out !== META_FILENAME && + out !== WHISPER_FILENAME, + `${kind.id} output ${out} collides with a shipped input`, + ); + } + } +}); + +test("every transcript-derived kind ships transcript.cues.json LAST", () => { + // The executor materializes inputs in order and isCuesJsonFresh compares + // mtimes: a cues file written before its metadata/raw transcript reads as + // stale and the unit silently does nothing. + for (const id of ["attribution-text", "attribution-diarized", "digest"]) { + const inputs = getOperation(id)!.inputs(FULL_DIR); + assert.equal( + inputs[inputs.length - 1], + CUES_JSON_FILENAME, + `${id} must list cues last, got: ${inputs.join(", ")}`, + ); + assert.ok(inputs.includes(META_FILENAME), `${id} ships metadata`); + assert.ok(inputs.includes(WHISPER_FILENAME), `${id} ships the raw transcript`); + } +}); + +test("the diarized attribution unit ships diarization.json; diarization ships audio", () => { + assert.ok( + getOperation("attribution-diarized")! + .inputs(FULL_DIR) + .includes(DIARIZATION_FILENAME), + ); + assert.ok(getOperation("diarization")!.inputs(FULL_DIR).includes("audio.mp3")); +}); + +function attributionPayload(method: "text-only" | "diarized"): string { + const record: AttributionRecord = { + videoId: "vid", + generatedAt: "2026-08-24T00:00:00.000Z", + speakers: [{ index: 0, label: "Host", seconds: 10 }], + segments: [{ start: 0, end: 10, speaker: 0 }], + provenance: { + method, + appId: "ollama-direct", + model: "m", + modelRequested: "m", + promptVersion: 2, + generatedAt: "2026-08-24T00:00:00.000Z", + durationMs: 1, + }, + }; + return JSON.stringify(record); +} + +test("applyResult refuses a text-only record over a diarized one (the downgrade rule)", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); + await writeAttribution( + dir, + JSON.parse(attributionPayload("diarized")) as AttributionRecord, + ); + const outcome = await getOperation("attribution-text")!.applyResult(dir, { + [ATTRIBUTION_FILENAME]: attributionPayload("text-only"), + }); + assert.equal(outcome, "refused"); + assert.equal( + (await loadAttribution(dir))?.provenance.method, + "diarized", + "the diarized record must survive", + ); +}); + +test("applyResult upgrades a text-only record to a diarized one", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); + await writeAttribution( + dir, + JSON.parse(attributionPayload("text-only")) as AttributionRecord, + ); + const outcome = await getOperation("attribution-diarized")!.applyResult( + dir, + { [ATTRIBUTION_FILENAME]: attributionPayload("diarized") }, + ); + assert.equal(outcome, "applied"); + assert.equal((await loadAttribution(dir))?.provenance.method, "diarized"); +}); + +test("applyResult rejects a malformed attribution payload", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "unit-attr-")); + const kind = getOperation("attribution-text")!; + assert.equal(await kind.applyResult(dir, {}), "invalid"); + assert.equal( + await kind.applyResult(dir, { [ATTRIBUTION_FILENAME]: "not json" }), + "invalid", + ); + assert.equal( + await kind.applyResult(dir, { [ATTRIBUTION_FILENAME]: "{}" }), + "invalid", + ); +}); + +function digestProvenance(model: string): DigestProvenance { + return { + appId: "ollama-direct", + model, + promptVersion: 2, + contextHash: "none", + generatedAt: "2026-08-24T00:00:00.000Z", + } as DigestProvenance; +} + +test("applying one digest section preserves the other (guarded read-modify-write)", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "unit-digest-")); + // The primary already holds a tags section… + await writeDigestSection(dir, { + section: "tags", + items: [], + provenance: digestProvenance("local-model"), + warnings: [], + }); + // …and a unit returns a record carrying only chapters. + const payload = JSON.stringify({ + sections: { + chapters: { provenance: digestProvenance("m"), items: [] }, + }, + }); + const outcome = await getOperation("digest")!.applyResult(dir, { + [DIGEST_FILENAME]: payload, + }); + assert.equal(outcome, "applied"); + const record = await loadDigest(dir); + assert.ok(record?.sections?.chapters, "the unit's section landed"); + assert.ok( + record?.sections?.tags, + "the section the primary wrote must survive the apply", + ); + assert.equal(record?.sections?.tags?.provenance.model, "local-model"); +}); + +test("diarization results apply verbatim (single-writer sidecar) after a shape check", async () => { + const dir = await mkdtemp(path.join(tmpdir(), "unit-diar-")); + const kind = getOperation("diarization")!; + assert.equal( + await kind.applyResult(dir, { [DIARIZATION_FILENAME]: "{}" }), + "invalid", + ); + const outcome = await kind.applyResult(dir, { + [DIARIZATION_FILENAME]: JSON.stringify({ + videoId: "vid", + generatedAt: "2026-08-24T00:00:00.000Z", + speakers: 1, + turns: [{ start: 0, end: 5, speaker: 0 }], + }), + }); + assert.equal(outcome, "applied"); + const raw = await readFile(path.join(dir, DIARIZATION_FILENAME), "utf8"); + assert.equal((JSON.parse(raw) as { speakers: number }).speakers, 1); + // And the server-side loader accepts what applyResult wrote. + assert.notEqual(await loadDiarization(dir), null); +}); diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts @@ -0,0 +1,1494 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import path from "node:path"; +import os from "node:os"; +import { mkdtemp, mkdir, writeFile, rm, utimes } from "node:fs/promises"; +import { + getOperation, + allOperations, + backfillLaneOperations, + resolveBackfillLaneOperations, + orderByDependencies, + operationCatalog, + operationCostBasis, + operationGroup, + operationLabel, + operationsActionLabel, + operationsGroupLabel, + digestLaneFor, + diarizationLaneFor, + laneYieldsToTranscription, + addOperationState, + emptyOperationCounts, + reachableOperationWork, + type OperationClassification, + type Operation, + backfillLaneEntriesOf, + backfillLaneOperationEntriesOf, + presentOperationWork, +} from "./operations"; +import { candidateAction } from "../controller/backfillBatch"; +import { + readVideoFiles, + CUES_JSON_FILENAME, + META_FILENAME, + SOURCE_MEDIA_BASENAME, +} from "./videoStatus"; +import { CUES_FILE_VERSION } from "../controller/normalizeTranscript"; +import { attributeOneVideo } from "../controller/attributeOne"; +import { DIGEST_FILENAME, OLLAMA_DIGEST_APP_ID } from "./digest"; +import { BACKFILL_QUEUE, DIGEST_LOCAL_QUEUE } from "./queueKeys"; +import { + ATTRIBUTION_FILENAME, + ATTRIBUTION_PROMPT_VERSION, + type AttributionRecord, +} from "./attribution"; +import { + DEFAULT_DIARIZATION_THRESHOLD, + DIARIZATION_FILENAME, + SORTFORMER_DIARIZATION_ENGINE, + diarizationTarget, + isDiarizationFresh, + type DiarizationRecord, +} from "./diarization"; +import { defaultSiteSettings, type SiteSettings } from "./settings"; +import { SAVED_VIDEO_POINTER_FILENAME } from "./savedVideo"; + +// Run with: +// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/operations.test.ts +// +// Every case here is a FIXTURE DIRECTORY, on purpose: `state()` is defined as +// "read from disk, never stored", and a test that hands it a hand-built record +// would not be testing the thing the indicators actually call. + +const diarization = getOperation("diarization")!; + +// resolveTarget is channel-scoped since the digest entry (a digest's identity +// includes the hash of the channel's context note). None of the kinds tested +// here reads paths/channelSlug, so a stub is honest as well as convenient. +function targetCtx(settings: SiteSettings) { + return { + settings, + paths: { channelsDir: "/nonexistent" } as unknown as Parameters< + typeof diarization.resolveTarget + >[0]["paths"], + channelSlug: "test-channel", + }; +} + +function settingsWithDiarization( + over: Partial<SiteSettings["diarization"]> = {}, +): SiteSettings { + const s = defaultSiteSettings(); + return { + ...s, + diarization: { + ...s.diarization, + enabled: true, + segModel: "/opt/models/seg-1.onnx", + embModel: "/opt/models/emb-1.onnx", + threshold: DEFAULT_DIARIZATION_THRESHOLD, + ...over, + }, + }; +} + +// The duration cap is OFF by default now that windowing exists, so a test of the +// cap has to ask for one. Stated here once rather than inline, so it is obvious +// that every OTHER test in this file runs with the cap disabled — which is the +// shipped configuration. +function settingsWithCap(hours: number): SiteSettings { + return settingsWithDiarization({ maxAudioHours: hours }); +} + +// A video dir built from a description of what is on disk. `sidecar` is written +// verbatim so a MALFORMED file can be tested — that case is not hypothetical, +// it is what a crash mid-write leaves behind. +async function fixture(opts: { + transcript?: boolean; + audio?: boolean; + container?: boolean; + savedPointer?: boolean; + sidecar?: string; + // Written into metadata.info.json. Absent means NO metadata file at all, + // which is the "duration unknown" case the cap has to get right. + durationSec?: number; +}): Promise<{ dir: string; cleanup: () => Promise<void> }> { + const root = await mkdtemp(path.join(os.tmpdir(), "backfill-kinds-")); + const dir = path.join(root, "vid1"); + await mkdir(dir, { recursive: true }); + if (opts.transcript !== false) { + await writeFile( + path.join(dir, "transcript.json"), + JSON.stringify({ transcription: [{ text: "hi" }] }), + ); + } + if (opts.audio) await writeFile(path.join(dir, "audio.mp3"), "x"); + // `source-media.<ext>` specifically — isSourceMediaFile keys off the basename, + // and an arbitrarily-named mp4 in a video dir is not a persisted container. + if (opts.container) { + await writeFile(path.join(dir, `${SOURCE_MEDIA_BASENAME}.mp4`), "x"); + } + if (opts.savedPointer) { + await writeFile( + path.join(dir, SAVED_VIDEO_POINTER_FILENAME), + JSON.stringify({ storedAt: "now", dir: "/nowhere", file: "video.mp4" }), + ); + } + if (opts.sidecar !== undefined) { + await writeFile(path.join(dir, DIARIZATION_FILENAME), opts.sidecar); + } + if (opts.durationSec !== undefined) { + await writeFile( + path.join(dir, META_FILENAME), + JSON.stringify({ id: "vid1", duration: opts.durationSec }), + ); + } + return { dir, cleanup: () => rm(root, { recursive: true, force: true }) }; +} + +function sidecar(engine: DiarizationRecord["engine"]): string { + return JSON.stringify({ + videoId: "vid1", + generatedAt: "2026-08-07T00:00:00.000Z", + speakers: 2, + turns: [{ start: 0, end: 4, speaker: 0 }], + engine, + } satisfies DiarizationRecord); +} + +// The current identity, as scripts/diarize.mjs would record it: BASENAMES, not +// the configured full paths. +const CURRENT = { + engine: "sherpa-onnx", + segmentationModel: "seg-1.onnx", + embeddingModel: "emb-1.onnx", + threshold: DEFAULT_DIARIZATION_THRESHOLD, +}; + +async function classify( + dirOpts: Parameters<typeof fixture>[0], + settings: SiteSettings = settingsWithDiarization(), +): Promise<OperationClassification> { + const { dir, cleanup } = await fixture(dirOpts); + try { + const files = await readVideoFiles(dir, { checkUntranscribable: true }); + return await diarization.state({ + videoDir: dir, + videoId: "vid1", + files, + target: await diarization.resolveTarget(targetCtx(settings)), + settings, + }); + } finally { + await cleanup(); + } +} + +test("present: the sidecar matches the identity we would produce now", async () => { + assert.equal( + await classify({ audio: true, sidecar: sidecar(CURRENT) }), + "present", + ); + // Still present with the audio already cleaned away — the whole point of + // capturing while the audio exists is that the result outlives it. + assert.equal(await classify({ sidecar: sidecar(CURRENT) }), "present"); +}); + +test("stale: a different threshold or model is work, not coverage", async () => { + // The threshold is the single most consequential knob (it decides how many + // speakers come out), so a change to it MUST show as work. Before the + // comparator, this read as done. + assert.equal( + await classify({ + audio: true, + sidecar: sidecar({ ...CURRENT, threshold: 0.5 }), + }), + "stale", + ); + assert.equal( + await classify({ + audio: true, + sidecar: sidecar({ ...CURRENT, segmentationModel: "seg-OLD.onnx" }), + }), + "stale", + ); + assert.equal( + await classify({ + audio: true, + sidecar: sidecar({ ...CURRENT, embeddingModel: "emb-OLD.onnx" }), + }), + "stale", + ); +}); + +// THE ENGINE IS NOT COMPARED, and that is a fix rather than an omission. +// scripts/diarize.mjs records whichever binary actually ran, and nothing in +// DiarizationSettings can predict that — so a hardcoded engine in the target +// would mark every sidecar from any other wrapper permanently stale, which at +// ~500-680 s/audio-hour is an infinite regeneration loop. (The e2e fake engine +// records "fake-diarize" and would have tripped it on the first run.) +test("a different engine binary is not, by itself, stale", async () => { + assert.equal( + await classify({ + audio: true, + sidecar: sidecar({ ...CURRENT, engine: "some-other-engine" }), + }), + "present", + ); + // The comparison is still WRITTEN, so adding an engine setting later needs no + // new logic: a target that does declare one still rejects a mismatch. + assert.equal( + isDiarizationFresh( + { + videoId: "v", + generatedAt: "now", + speakers: 1, + turns: [], + engine: { ...CURRENT, engine: "some-other-engine" }, + }, + { ...CURRENT, engine: "sherpa-onnx" }, + ), + false, + ); +}); + +// The engine-as-a-setting case the test above says needs "no new logic". It is +// asserted for sortformer and NOT for the default, and that asymmetry is the +// whole design: the default engine still cannot predict which binary a +// `--engine` override runs, so it must keep asserting nothing. +test("selecting sortformer asserts the engine; selecting the default still does not", () => { + const rec = (engine: DiarizationRecord["engine"]): DiarizationRecord => ({ + videoId: "v", + generatedAt: "now", + speakers: 1, + turns: [], + engine, + }); + const sherpaSidecar = rec({ + engine: "sherpa-onnx", + segmentationModel: "seg-1.onnx", + embeddingModel: "emb-1.onnx", + threshold: DEFAULT_DIARIZATION_THRESHOLD, + }); + const sortformerSidecar = rec({ + engine: SORTFORMER_DIARIZATION_ENGINE, + model: "sortformer-4spk.gguf", + }); + + const sortformerTarget = diarizationTarget({ + engine: SORTFORMER_DIARIZATION_ENGINE, + sortformerModel: "/abs/path/sortformer-4spk.gguf", + // Still configured, because settings carry one set of fields for both + // engines. None of it may leak into the sortformer identity. + segModel: "/abs/seg-1.onnx", + embModel: "/abs/emb-1.onnx", + threshold: DEFAULT_DIARIZATION_THRESHOLD, + }); + + // Basename only, as everywhere else: the corpus is rsynced between shards. + assert.equal(sortformerTarget.model, "sortformer-4spk.gguf"); + assert.equal(sortformerTarget.segmentationModel, undefined); + assert.equal(sortformerTarget.embeddingModel, undefined); + + assert.equal(isDiarizationFresh(sortformerSidecar, sortformerTarget), true); + // The point of switching: every sherpa sidecar becomes work the backfill lane + // will offer to redo, rather than silently staying half a corpus. + assert.equal(isDiarizationFresh(sherpaSidecar, sortformerTarget), false); + + // And back the other way, with no special case needed — a sortformer record + // carries neither segmentation nor embedding model, so it fails the sherpa + // comparison on the models alone. + const sherpaTarget = diarizationTarget({ + engine: "sherpa-onnx", + segModel: "/abs/seg-1.onnx", + embModel: "/abs/emb-1.onnx", + threshold: DEFAULT_DIARIZATION_THRESHOLD, + }); + assert.equal(sherpaTarget.engine, undefined); + assert.equal(isDiarizationFresh(sherpaSidecar, sherpaTarget), true); + assert.equal(isDiarizationFresh(sortformerSidecar, sherpaTarget), false); +}); + +// Sortformer has no clustering step, so the threshold cannot have changed any of +// its turns. Letting it into the identity would regenerate the whole corpus for +// an edit that provably could not affect it. +test("the clustering threshold does not stale a sortformer sidecar", () => { + const sidecar: DiarizationRecord = { + videoId: "v", + generatedAt: "now", + speakers: 1, + turns: [], + engine: { engine: SORTFORMER_DIARIZATION_ENGINE, model: "m.gguf" }, + }; + const at = (threshold: number) => + diarizationTarget({ + engine: SORTFORMER_DIARIZATION_ENGINE, + sortformerModel: "m.gguf", + threshold, + }); + assert.equal(isDiarizationFresh(sidecar, at(DEFAULT_DIARIZATION_THRESHOLD)), true); + assert.equal(isDiarizationFresh(sidecar, at(0.4)), true); + // The model itself IS the identity, though — a different one is a redo. + assert.equal( + isDiarizationFresh( + sidecar, + diarizationTarget({ + engine: SORTFORMER_DIARIZATION_ENGINE, + sortformerModel: "other.gguf", + }), + ), + false, + ); +}); + +// Which resource diarization competes for is a CONFIGURATION outcome, and the +// backfill lane reads it to decide whether a guaranteed share is even available. +// Getting this wrong in the permissive direction puts ~4.4 GB of sortformer next +// to parakeet on an 8 GB card. +test("diarization contends for the GPU only as sortformer on vulkan", () => { + const lane = (engine: "sherpa-onnx" | "sortformer", backend: "vulkan" | "cpu") => + diarizationLaneFor({ engine, backend }); + + assert.equal(lane("sortformer", "vulkan").contendsFor, "gpu"); + assert.equal(laneYieldsToTranscription(lane("sortformer", "vulkan")), true); + + // The same engine on the CPU backend is only after cores. + assert.equal(lane("sortformer", "cpu").contendsFor, "cpu"); + assert.equal(laneYieldsToTranscription(lane("sortformer", "cpu")), false); + + // sherpa-onnx is ONNX/CPU, so a stale `backend: vulkan` left in settings must + // NOT make it claim the card — that would park the lane behind transcription + // for work using no shaders at all, which is the bug digestYield.ts already + // records having hit once. + assert.equal(lane("sherpa-onnx", "vulkan").contendsFor, "cpu"); + assert.equal(laneYieldsToTranscription(lane("sherpa-onnx", "vulkan")), false); + + // The queue key never changes: two diarizations must not run at once whichever + // engine is selected. + assert.equal(lane("sortformer", "vulkan").queueKey, BACKFILL_QUEUE); + assert.equal(lane("sherpa-onnx", "cpu").queueKey, BACKFILL_QUEUE); +}); + +// THE COMPATIBILITY RULE, and the reason it is written down: without it, adding +// a field to the provenance would mark all 77,000 videos stale at once. +test("an absent recorded field compares equal to today's default", async () => { + assert.equal( + await classify({ + audio: true, + // A sidecar from before the threshold was recorded at all. + sidecar: sidecar({ + engine: "sherpa-onnx", + segmentationModel: "seg-1.onnx", + embeddingModel: "emb-1.onnx", + }), + }), + "present", + ); + // ...and it is equal to the DEFAULT specifically, not to anything: with a + // non-default threshold configured, the same record is stale. + assert.equal( + await classify( + { + audio: true, + sidecar: sidecar({ + engine: "sherpa-onnx", + segmentationModel: "seg-1.onnx", + embeddingModel: "emb-1.onnx", + }), + }, + settingsWithDiarization({ threshold: 0.5 }), + ), + "stale", + ); +}); + +test("a version bump alone does not invalidate captured work", async () => { + // ~500-680 s/audio-hour of CPU says an engine point-release is not a reason to + // redo everything when the models and the threshold are unchanged. + assert.equal( + await classify({ + audio: true, + sidecar: sidecar({ ...CURRENT, version: "9.9.9" }), + }), + "present", + ); +}); + +test("missing: no sidecar, and the input is still here", async () => { + assert.equal(await classify({ audio: true }), "missing"); + // A persisted source container counts — ffmpeg reads it directly, which is + // what resolveDiarizableMedia does. + assert.equal(await classify({ container: true }), "missing"); +}); + +test("missing-input: no sidecar and nothing to diarize from", async () => { + assert.equal(await classify({}), "missing-input"); + // A saved-video POINTER whose stored file has gone (an unmounted backup disk) + // is not an input either — the pointer is not the media. + assert.equal(await classify({ savedPointer: true }), "missing-input"); +}); + +// --------------------------------------------------------------------------- +// The duration cap. A stopgap for an OOM that kills 6 of 10 videos over 6 hours +// on this box, burning ~40 minutes each and producing nothing. + +test("deferred: over the cap, with the input right there", async () => { + // 5 hours against the 4-hour default. The input EXISTS — that is the whole + // point of a third state: this is not missing-input (nothing to work from) and + // not missing (work to do); it is work deliberately not attempted. + assert.equal( + await classify({ audio: true, durationSec: 5 * 3600 }, settingsWithCap(4)), + "deferred", + ); +}); + +test("under the cap is ordinary missing work", async () => { + assert.equal( + await classify({ audio: true, durationSec: 3 * 3600 }, settingsWithCap(4)), + "missing", + ); + // Exactly at the cap is not over it. + assert.equal( + await classify({ audio: true, durationSec: 4 * 3600 }, settingsWithCap(4)), + "missing", + ); +}); + +test("unknown duration is NOT deferred", async () => { + // No metadata.info.json at all. Deferring here would quietly remove a video + // from the work list on the strength of a file that could not be read, and it + // would break every fixture in this file that predates the cap. + assert.equal(await classify({ audio: true }, settingsWithCap(4)), "missing"); + // Present but useless — the same answer, for the same reason. + assert.equal( + await classify({ audio: true, durationSec: 0 }, settingsWithCap(4)), + "missing", + ); +}); + +test("maxAudioHours 0 turns the cap off — and 0 is the shipped default", async () => { + // Where this setting went once windowed diarization landed. + assert.equal( + await classify( + { audio: true, durationSec: 12 * 3600 }, + settingsWithDiarization({ maxAudioHours: 0 }), + ), + "missing", + ); +}); + +test("the cap never overrules a sidecar that is already there", async () => { + // A long video ALREADY diarized stays `present`: the cap decides what to + // attempt, not what counts as done. Otherwise raising the cap would look like + // work appearing and lowering it would look like work being undone. + assert.equal( + await classify( + { audio: true, durationSec: 12 * 3600, sidecar: sidecar(CURRENT) }, + settingsWithCap(4), + ), + "present", + ); +}); + +test("the cap does not resurrect a video whose input is gone", async () => { + // missing-input is checked FIRST. A 12-hour video with nothing to diarize from + // is unreachable, not deferred — deferred promises "we could do this if you + // raised the cap", and that would be a lie here. + assert.equal( + await classify({ durationSec: 12 * 3600 }, settingsWithCap(4)), + "missing-input", + ); +}); + +test("a malformed sidecar reads as absent, never as done", async () => { + // The same rule diarization-server.ts's hasDiarization() encodes: a + // half-written file must not be what convinces anything the work is captured + // — that is what would let the cleanup sweep delete the only copy of the audio. + assert.equal( + await classify({ audio: true, sidecar: "{ not json" }), + "missing", + ); + assert.equal( + await classify({ audio: true, sidecar: JSON.stringify({ videoId: "x" }) }), + "missing", + ); +}); + +test("not-applicable: an untranscribed video is not this backfill's business", async () => { + assert.equal( + await classify({ transcript: false, audio: true }), + "not-applicable", + ); +}); + +test("a disabled or unconfigured feature reports no backfill at all", () => { + const s = defaultSiteSettings(); + // Off by default, so nothing advertises catch-up work for it. + assert.equal(diarization.enabled(s), false); + assert.equal(backfillLaneOperations(s).length, 0); + // Enabled but with no models is "not set up", which must also report nothing + // rather than a corpus-sized work list nobody can act on. + assert.equal( + diarization.enabled(settingsWithDiarization({ segModel: "" })), + false, + ); + assert.equal(diarization.enabled(settingsWithDiarization()), true); + assert.equal(backfillLaneOperations(settingsWithDiarization()).length, 1); +}); + +test("an unknown or stale kind id in the sweep scope is dropped, not fatal", () => { + const s = settingsWithDiarization(); + // Empty scope = every enabled lane kind. + assert.deepEqual( + resolveBackfillLaneOperations(s, []).map((k) => k.id), + ["diarization"], + ); + assert.deepEqual( + resolveBackfillLaneOperations(s, undefined).map((k) => k.id), + ["diarization"], + ); + // A settings file naming a kind from another build must not wedge the lane. + assert.deepEqual(resolveBackfillLaneOperations(s, ["from-the-future"]), []); + assert.deepEqual( + resolveBackfillLaneOperations(s, ["diarization", "from-the-future"]).map((k) => k.id), + ["diarization"], + ); +}); + +// --------------------------------------------------------------------------- +// Attribution — the second and third kinds, and the pair that shares one file +// --------------------------------------------------------------------------- + +const attrText = getOperation("attribution-text")!; +const attrDiarized = getOperation("attribution-diarized")!; + +function settingsWithAttribution( + over: Partial<SiteSettings["attribution"]> = {}, +): SiteSettings { + const s = defaultSiteSettings(); + return { + ...s, + attribution: { + ...s.attribution, + enabled: true, + diarizedEnabled: true, + textOnlyEnabled: true, + appId: OLLAMA_DIGEST_APP_ID, + model: "qwen2.5:7b", + ...over, + }, + }; +} + +function attrSidecar(over: Partial<AttributionRecord["provenance"]> = {}): string { + return JSON.stringify({ + videoId: "vid1", + generatedAt: "2026-08-07T00:00:00.000Z", + speakers: [{ index: 0, label: "Host" }], + segments: [{ start: 0, end: 10, speaker: 0 }], + provenance: { + method: "diarized", + appId: OLLAMA_DIGEST_APP_ID, + model: "qwen2.5:7b", + modelRequested: "qwen2.5:7b", + promptVersion: ATTRIBUTION_PROMPT_VERSION, + generatedAt: "2026-08-07T00:00:00.000Z", + ...over, + }, + } satisfies AttributionRecord); +} + +// Same fixture discipline as above — a real directory, because state() is +// defined as "read from disk, never stored". +async function attrFixture(opts: { + transcript?: boolean; + diarization?: string; + attribution?: string; +}): Promise<{ dir: string; cleanup: () => Promise<void> }> { + const root = await mkdtemp(path.join(os.tmpdir(), "backfill-attr-")); + const dir = path.join(root, "vid1"); + await mkdir(dir, { recursive: true }); + if (opts.transcript !== false) { + await writeFile( + path.join(dir, "transcript.json"), + JSON.stringify({ transcription: [{ text: "hi" }] }), + ); + } + if (opts.diarization !== undefined) { + await writeFile(path.join(dir, DIARIZATION_FILENAME), opts.diarization); + } + if (opts.attribution !== undefined) { + await writeFile(path.join(dir, ATTRIBUTION_FILENAME), opts.attribution); + } + return { dir, cleanup: () => rm(root, { recursive: true, force: true }) }; +} + +async function classifyAttr( + kind: typeof attrText, + dirOpts: Parameters<typeof attrFixture>[0], + settings: SiteSettings = settingsWithAttribution(), +): Promise<OperationClassification> { + const { dir, cleanup } = await attrFixture(dirOpts); + try { + const files = await readVideoFiles(dir, { checkUntranscribable: true }); + return await kind.state({ + videoDir: dir, + videoId: "vid1", + files, + target: await kind.resolveTarget(targetCtx(settings)), + settings, + }); + } finally { + await cleanup(); + } +} + +// A diarization sidecar at a known generatedAt, so the "re-diarizing invalidates +// the naming" case has something to compare against. +const DIARIZED_AT = "2026-08-06T00:00:00.000Z"; +function diarizationSidecar(generatedAt = DIARIZED_AT): string { + return JSON.stringify({ + videoId: "vid1", + generatedAt, + speakers: 2, + turns: [{ start: 0, end: 10, speaker: 0 }], + engine: { engine: "sherpa-onnx" }, + } satisfies DiarizationRecord); +} + +test("attribution-text reaches every transcribed video and never reports missing-input", async () => { + // The claim that prices this lane: its input is the cue stream, which every + // transcribed video has. That is why it can reach the whole corpus, and why + // running it over the whole corpus costs ~194,000 model calls. + assert.equal(await classifyAttr(attrText, {}), "missing"); + assert.equal( + await classifyAttr(attrText, { transcript: false }), + "not-applicable", + ); +}); + +test("attribution-text: fresh, stale, and a malformed file that reads as absent", async () => { + assert.equal( + await classifyAttr(attrText, { + attribution: attrSidecar({ method: "text-only" }), + }), + "present", + ); + assert.equal( + await classifyAttr(attrText, { + attribution: attrSidecar({ method: "text-only", model: "llama3:8b", modelRequested: "llama3:8b" }), + }), + "stale", + ); + // A half-written sidecar must read as absent — and here that matters twice + // over, because a malformed file that read as "present" could masquerade as a + // diarized record and block this lane forever. + assert.equal( + await classifyAttr(attrText, { attribution: "{ not json" }), + "missing", + ); +}); + +// THE DOWNGRADE RULE, in the counter. Reporting a diarized record as work would +// put this lane in a loop of "attempt, refuse, still outstanding" on every pass +// of a multi-day sweep. +test("attribution-text has nothing to do where a diarized record exists", async () => { + assert.equal( + await classifyAttr(attrText, { attribution: attrSidecar() }), + "present", + ); + // Not even when that diarized record is itself stale — it is not this lane's + // record to redo. + assert.equal( + await classifyAttr(attrText, { + attribution: attrSidecar({ model: "llama3:8b", modelRequested: "llama3:8b" }), + }), + "present", + ); +}); + +test("attribution-diarized: no diarization.json is BLOCKED, not missing-input", async () => { + // ~73,000 videos on this corpus, against a handful reachable — so this state + // has to be right or every surface is wrong. + // + // THIS ASSERTION USED TO SAY missing-input, AND THAT WAS THE DEFECT. + // missing-input means one thing to the rest of the system: the media is gone, + // re-acquire it. So these videos were counted as needing media re-fetched, + // and with allowRedownload on the lane would have spent a download per video + // fetching AUDIO — which cannot satisfy a wait for diarization.json — and + // then deleted it again. What they are waiting for is the `diarization` kind, + // which this table produces. + assert.equal(await classifyAttr(attrDiarized, {}), "blocked"); + assert.equal( + await classifyAttr(attrDiarized, { attribution: attrSidecar() }), + "blocked", + ); +}); + +test("blocked is never dispatched, and re-download cannot change that", () => { + // The dispatch decision is the consequential one: `blocked` must not reach a + // runner whatever the flags say. Note allowRedownload — the flag that DOES + // turn missing-input into a dispatch — is deliberately inert here. + for (const force of [false, true]) { + for (const allowRedownload of [false, true]) { + assert.equal( + candidateAction("blocked", { force, allowRedownload }), + "blocked", + `force=${force} allowRedownload=${allowRedownload}`, + ); + } + } + // The contrast, so this test fails if the two ever get conflated again. + assert.equal( + candidateAction("missing-input", { force: false, allowRedownload: true }), + "dispatch", + ); +}); + +test("blocked is counted, and is NOT reachable work", () => { + const counts = emptyOperationCounts(); + addOperationState(counts, "blocked"); + addOperationState(counts, "blocked"); + addOperationState(counts, "missing"); + assert.equal(counts.blocked, 2); + assert.equal(counts.missing, 1); + // The load-bearing line. A corpus with one diarization and 73,000 waiting + // attributions must not report 73,000 jobs ready to run. + assert.equal(reachableOperationWork(counts), 1); + // And it is its own number, not folded into the re-acquire population. + assert.equal(counts.missingInput, 0); + assert.equal(counts.deferred, 0); +}); + +test("a prerequisite is ordered before the kind that declares it", () => { + const base = settingsWithDiarization(); + const kinds = resolveBackfillLaneOperations( + { + ...base, + attribution: { + ...base.attribution, + enabled: true, + diarizedEnabled: true, + textOnlyEnabled: true, + }, + }, + undefined, + ); + const ids = kinds.map((k) => k.id); + const diarizationAt = ids.indexOf("diarization"); + const diarizedAttrAt = ids.indexOf("attribution-diarized"); + assert.ok(diarizationAt >= 0 && diarizedAttrAt >= 0, ids.join(",")); + // Without this, a video diarized during a pass only becomes attributable on + // whatever LATER pass happens to find the sidecar on disk. + assert.ok( + diarizationAt < diarizedAttrAt, + `diarization must precede attribution-diarized, got ${ids.join(", ")}`, + ); + // AND the sort is STABLE: attribution-text stays last, which is a separate + // deliberate decision (the better lane must reach a diarized video first, or + // the text lane spends ~30 model calls to produce a record it must not write). + assert.equal(ids[ids.length - 1], "attribution-text", ids.join(",")); +}); + +test("ordering degrades safely when a prerequisite is absent or cyclic", () => { + const a = { id: "a", dependsOn: ["b"] } as unknown as Operation; + const b = { id: "b", dependsOn: ["a"] } as unknown as Operation; + // A cycle must not wedge the lane or silently drop a kind: every entry comes + // back, in declaration order. + assert.deepEqual( + orderByDependencies([a, b]).map((k) => k.id), + ["a", "b"], + ); + // A dependency on something not in the selection imposes no ordering, and + // does not remove the dependant from the run. + const lonely = { id: "lonely", dependsOn: ["not-here"] } as unknown as + Operation; + assert.deepEqual( + orderByDependencies([lonely]).map((k) => k.id), + ["lonely"], + ); +}); + +test("attribution-diarized treats a text-only record as the upgrade queue", async () => { + // `missing`, not `stale`: the diarized record this kind is responsible for + // genuinely was never made. This is the whole of PLAN.md's bespoke "upgrade + // job", and it falls out of the registry rather than needing new machinery. + assert.equal( + await classifyAttr(attrDiarized, { + diarization: diarizationSidecar(), + attribution: attrSidecar({ method: "text-only" }), + }), + "missing", + ); + assert.equal( + await classifyAttr(attrDiarized, { diarization: diarizationSidecar() }), + "missing", + ); +}); + +test("attribution-diarized: present, and stale when the clusters underneath change", async () => { + assert.equal( + await classifyAttr(attrDiarized, { + diarization: diarizationSidecar(), + attribution: attrSidecar({ diarizationGeneratedAt: DIARIZED_AT }), + }), + "present", + ); + // Re-diarized since. Cluster 3 is now a different person, or nobody, so the + // names that pointed at it are work again. + assert.equal( + await classifyAttr(attrDiarized, { + diarization: diarizationSidecar("2026-08-09T00:00:00.000Z"), + attribution: attrSidecar({ diarizationGeneratedAt: DIARIZED_AT }), + }), + "stale", + ); + // A model change is stale the ordinary way too. + assert.equal( + await classifyAttr(attrDiarized, { + diarization: diarizationSidecar(), + attribution: attrSidecar({ + diarizationGeneratedAt: DIARIZED_AT, + model: "llama3:8b", + modelRequested: "llama3:8b", + }), + }), + "stale", + ); +}); + +// Capture may legitimately be switched off after a run: a sidecar on disk is a +// perfectly good input, and refusing to name it would strand exactly the work +// the capture lane exists to protect. +test("attribution-diarized does not require diarization CAPTURE to still be on", async () => { + const s = settingsWithAttribution(); + assert.equal(s.diarization.enabled, false); + assert.equal(attrDiarized.enabled(s), true); + assert.equal( + await classifyAttr( + attrDiarized, + { diarization: diarizationSidecar() }, + s, + ), + "missing", + ); +}); + +test("each attribution lane is gated separately, under one master switch", () => { + const off = defaultSiteSettings(); + assert.equal(attrText.enabled(off), false); + assert.equal(attrDiarized.enabled(off), false); + // Turning the feature on must not by itself arm a ~194,000-call sweep. + const onlyMaster = settingsWithAttribution({ + diarizedEnabled: false, + textOnlyEnabled: false, + }); + assert.equal(attrText.enabled(onlyMaster), false); + assert.equal(attrDiarized.enabled(onlyMaster), false); + assert.equal( + attrText.enabled(settingsWithAttribution({ textOnlyEnabled: false })), + false, + ); + assert.equal( + attrDiarized.enabled(settingsWithAttribution({ diarizedEnabled: false })), + false, + ); + // Both lanes on. Diarization CAPTURE is still off in this fixture, so two — + // which is also the point: an attribution backfill does not need the capture + // lane armed to have work. + assert.equal(backfillLaneOperations(settingsWithAttribution()).length, 2); +}); + +// The cheap, better lane must get to a video first: on a video that has +// diarization, running the text lane first would spend ~30 model calls producing +// a record the diarized lane then replaces. +test("the diarized lane is ordered ahead of the text-only lane", () => { + assert.deepEqual( + resolveBackfillLaneOperations(settingsWithAttribution(), []).map((k) => k.id), + ["attribution-diarized", "attribution-text"], + ); +}); + +// THE RUNNER'S OWN GUARD, and it is not the same test as the counter's. state() +// runs at pull time; the pool can hold a candidate for minutes afterwards, and +// the diarized lane can land a better record in that window. This asserts the +// refusal happens with NO engine call at all — no settings resolved, no model +// chosen, nothing that could fail for an unrelated reason. +test("the text lane refuses to downgrade a diarized record, without calling an engine", async () => { + const { dir, cleanup } = await attrFixture({ + attribution: attrSidecar(), + }); + try { + // A full transcript fixture, so the run gets PAST the cues guard and the + // refusal is genuinely the downgrade rule rather than a missing transcript. + await writeTranscriptFixture(dir); + const outcome = await attributeOneVideo({ + // A bogus engine: reaching it at all is the failure this test is looking + // for. The guard fires before anything resolves an app. + paths: { channelsDir: dir } as never, + videoDir: dir, + videoId: "vid1", + channelSlug: "chan", + method: "text-only", + settings: { + ...settingsWithAttribution().attribution, + appId: "no-such-engine", + }, + // Even FORCED. Forcing a regeneration is not the same as asking for a + // worse record, and nothing in the UI should be able to request the second + // by accident. + force: true, + onLog: () => {}, + }); + assert.equal(outcome, "outranked"); + } finally { + await cleanup(); + } +}); + +// mtimes have to ascend: isCuesJsonFresh compares cues.json against the metadata +// and the raw transcript, and a cues.json older than either means the transcript +// changed underneath and must not be attributed. +async function writeTranscriptFixture(dir: string): Promise<void> { + await writeFile( + path.join(dir, META_FILENAME), + JSON.stringify({ id: "vid1", title: "A video" }), + ); + await writeFile( + path.join(dir, CUES_JSON_FILENAME), + JSON.stringify({ + version: CUES_FILE_VERSION, + id: "vid1", + title: "A video", + cues: [{ start: 0, end: 5, text: "hello" }], + }), + ); + const base = Date.now() / 1000; + await utimes(path.join(dir, META_FILENAME), base, base); + await utimes(path.join(dir, "transcript.json"), base, base); + await utimes(path.join(dir, CUES_JSON_FILENAME), base + 10, base + 10); +} + +// --------------------------------------------------------------------------- +// The digest entry. Registered but NOT dispatched by the backfill lane — see +// backfillLaneOperations — so these tests pin the classification, which is the half +// that is now shared, and the lane declaration, which is what keeps the GPU and +// CPU lanes from serializing. +// --------------------------------------------------------------------------- + +const digestKind = getOperation("digest")!; + +// The digest freshness target, hand-built rather than resolved: resolveTarget +// reads the channel's context note through paths, and these cases are about +// what state() does with a record, not about how the identity is derived. +const DIGEST_TARGET = { + target: { + appId: OLLAMA_DIGEST_APP_ID, + model: "llama3:8b", + promptVersion: 2, + contextHash: "ctx-1", + }, + sections: ["chapters"] as const, +}; + +function digestSidecar(over: Record<string, unknown> = {}): string { + return JSON.stringify({ + videoId: "vid1", + digestSchemaVersion: 1, + sections: { + chapters: { + items: [{ start: 0, title: "Intro" }], + provenance: { + appId: OLLAMA_DIGEST_APP_ID, + model: "llama3:8b", + modelRequested: "llama3:8b", + promptVersion: 2, + contextHash: "ctx-1", + }, + }, + }, + ...over, + }); +} + +async function classifyDigest(opts: { + transcript?: boolean; + cues?: boolean; + staleCues?: boolean; + sidecar?: string; + // Which sections must ALL be fresh. Defaults to the single-section shape the + // live corpus is configured with; the two-section form is what makes a + // part-done digest possible at all. + sections?: readonly string[]; +}): Promise<OperationClassification> { + const { dir, cleanup } = await fixture({ transcript: opts.transcript }); + try { + if (opts.cues !== false && opts.transcript !== false) { + await writeTranscriptFixture(dir); + if (opts.staleCues) { + // cues.json OLDER than the raw transcript: the transcript changed + // underneath and the normalize pass owes this video a rewrite. + const base = Date.now() / 1000; + await utimes(path.join(dir, CUES_JSON_FILENAME), base - 100, base - 100); + } + } + if (opts.sidecar) { + await writeFile(path.join(dir, DIGEST_FILENAME), opts.sidecar); + } + const files = await readVideoFiles(dir, { checkUntranscribable: true }); + return await digestKind.state({ + videoDir: dir, + videoId: "vid1", + files, + target: opts.sections + ? { ...DIGEST_TARGET, sections: opts.sections } + : DIGEST_TARGET, + settings: defaultSiteSettings(), + }); + } finally { + await cleanup(); + } +} + +test("digest: a video with no transcript is BLOCKED on transcription", async () => { + // PLAN.md states "a video cannot be digested until it has a transcript" in + // prose and nothing enforced it. Now it is declared (dependsOn) and reported. + assert.equal(await classifyDigest({ transcript: false }), "blocked"); + assert.deepEqual(digestKind.dependsOn, ["transcription"]); + // And the dependency resolves to something nameable, which is the whole + // reason the externally-dispatched operations are in the catalog at all. + assert.equal(operationLabel("transcription"), "Transcription"); +}); + +test("digest: a fresh sidecar is present, a mismatched one is stale", async () => { + assert.equal( + await classifyDigest({ sidecar: digestSidecar() }), + "present", + ); + // A model change is work. This is the case the snapshot's old + // "does an ai-digest.json exist?" test got wrong, reading "All digested" + // while the batch reported the whole channel as stale. + assert.equal( + await classifyDigest({ + sidecar: digestSidecar({ + sections: { + chapters: { + items: [{ start: 0, title: "Intro" }], + provenance: { + appId: OLLAMA_DIGEST_APP_ID, + model: "llama3:70b", + modelRequested: "llama3:70b", + promptVersion: 2, + contextHash: "ctx-1", + }, + }, + }, + }), + }), + "stale", + ); +}); + +test("digest: no sidecar at all is missing, not stale", async () => { + assert.equal(await classifyDigest({}), "missing"); +}); + +test("digest: a stale cues.json defers rather than digesting superseded text", async () => { + // Digesting now would describe text that is about to be rewritten, and would + // then look fresh forever. Not a failure — the normalize pass fixes it — so + // it must not be counted as reachable work or as a broken engine. + assert.equal(await classifyDigest({ staleCues: true }), "deferred"); +}); + +test("digest: a transcript with NO cues.json defers too — the case that is 97.6% of them", async () => { + // The population this branch actually catches. Measured over 79,219 video + // dirs: 1,942 have a raw transcript and no cues.json at all, against 47 with + // a superseded one — and 1,683 of the 1,942 are a single `handling: "youtube"` + // channel, which downloads subtitles with --skip-download and so never runs + // transcribeOne, the only automatic caller of normalizeTranscript. + // + // Same classification as the stale case on purpose: one normalize pass fixes + // both, so the distinction is about COPY (see CuesFreshReason), not dispatch. + assert.equal( + await classifyDigest({ transcript: true, cues: false }), + "deferred", + ); +}); + +test("a kind that can defer says WHY, and digest's reason is not 'it clears itself'", () => { + // BackfillStage used to hardcode one sentence about the diarization duration + // cap for every kind's deferred videos at once. Correct only while diarization + // was the sole kind that could defer. + assert.match( + getOperation("diarization")!.deferredHint!, + /Max audio hours/, + ); + // The claim this whole change exists to retract: the old copy said the + // normalize pass clears these on its own. Nothing runs it on its own, so the + // hint has to name the action. + const digestHint = digestKind.deferredHint!; + assert.match(digestHint, /Normalize/); + assert.doesNotMatch(digestHint, /clears? (itself|them|it)/i); +}); + +test("digest: a digest shared from a duplicate cluster counts as done", async () => { + // Worth ~11% of the sweep. If this entry disagreed with the snapshot's + // noDigest bucket here, every mirror would be regenerated. + assert.equal( + await classifyDigest({ + sidecar: digestSidecar({ derivedFrom: { videoId: "canonical" } }), + }), + "present", + ); +}); + +test("digest declares its own lane, and it is NOT the backfill queue", () => { + // THE PROPERTY THAT KEEPS THEM CONCURRENT. registry.ts runs every non-empty + // queueKey at concurrency 1, so sharing a key with diarization would make the + // GPU lane wait on the CPU lane and vice versa. + assert.notEqual(digestKind.lane.queueKey, BACKFILL_QUEUE); + assert.equal(digestKind.lane.queueKey, DIGEST_LOCAL_QUEUE); + assert.equal(getOperation("diarization")!.lane.queueKey, BACKFILL_QUEUE); + // The two digest lanes are distinct from each other for the same reason. + assert.notEqual( + digestLaneFor("local-gpu").queueKey, + digestLaneFor("remote-api").queueKey, + ); + // Only the GPU lane stands aside for transcription. Yielding the metered lane + // would park something that costs nothing to keep running. + assert.equal(laneYieldsToTranscription(digestLaneFor("local-gpu")), true); + assert.equal(laneYieldsToTranscription(digestLaneFor("remote-api")), false); +}); + +test("the backfill lane never dispatches digest", () => { + // backfillLaneOperations feeds backfillBatch, the channel Backfill card and the + // dashboard instrument. Digest must be in the CATALOG and out of THAT list, + // or it both serializes behind diarization and gets double-counted. + const settings = settingsWithDiarization(); + const laneIds = backfillLaneOperations(settings).map((k) => k.id); + const allIds = allOperations(settings).map((k) => k.id); + assert.ok(allIds.includes("digest"), allIds.join(",")); + assert.ok(!laneIds.includes("digest"), laneIds.join(",")); + // Which means backfillBatch's kind resolution cannot reach it either, even + // when it is asked for by name. + assert.deepEqual(resolveBackfillLaneOperations(settings, ["digest"]), []); +}); + +test("the catalog covers every operation, dispatched here or not", () => { + const ids = operationCatalog().map((o) => o.id); + for (const id of [ + "download", + "transcription", + "diarization", + "attribution-diarized", + "attribution-text", + "digest", + ]) { + assert.ok(ids.includes(id), `${id} missing from ${ids.join(",")}`); + } + // Every declared dependency resolves to a catalogued operation. This is the + // check that would have caught digest.dependsOn naming something that did not + // exist. + for (const op of operationCatalog()) { + for (const dep of op.dependsOn ?? []) { + assert.ok(ids.includes(dep), `${op.id} depends on unknown ${dep}`); + } + } +}); + +test("`runner` names the auto-queue runner, and only for the two that have one", () => { + // The console reads this instead of asking whether the id happens to be + // "download" or "transcription". `dispatch` cannot answer it: both of those + // are `external` WITH a runner, and a future `transcode` would be `external` + // with none — so an id-shaped guess would hand transcode the transcription + // runner's controls. + const runners = new Map(operationCatalog().map((o) => [o.id, o.runner])); + assert.equal(runners.get("download"), "download"); + assert.equal(runners.get("transcription"), "transcription"); + // Everything the sweep dispatches must leave it unset — a backfill kind with + // a runner would render a runner console over a lane no runner feeds. + for (const op of operationCatalog()) { + if (op.id === "download" || op.id === "transcription") continue; + assert.equal(op.runner, undefined, `${op.id} declares a runner`); + } +}); + +test("every catalogued operation declares a group and a cost basis", () => { + // Both are read unconditionally by the UI — the transit line groups stations + // by `group`, and every armed operation prints `costBasis` beside its + // backlog. An entry missing either renders a blank where a fact should be, + // which is the failure mode that let attribution-text sit armed at ~194,000 + // model calls while every screen called it "Backfill". + for (const op of operationCatalog()) { + assert.ok(op.group, `${op.id} has no group`); + assert.ok(op.shortLabel.length > 0, `${op.id} has no shortLabel`); + assert.ok( + op.shortLabel.length <= 12, + `${op.id} shortLabel "${op.shortLabel}" is too long for a column header`, + ); + assert.ok(op.costBasis.length > 0, `${op.id} has no costBasis`); + } +}); + +test("the three speaker operations share one group; digest does not", () => { + assert.equal(operationGroup("diarization"), "speakers"); + assert.equal(operationGroup("attribution-diarized"), "speakers"); + assert.equal(operationGroup("attribution-text"), "speakers"); + assert.equal(operationGroup("digest"), "digest"); + assert.equal(operationGroup("download"), "media"); + assert.equal(operationGroup("transcription"), "transcript"); + // An id the catalog does not know is null, NOT filed under the first group. + assert.equal(operationGroup("no-such-operation"), null); +}); + +test("a set's label is derived, so a mixed lane cannot claim one member's name", () => { + // The whole reason this is derived: the backfill lane holds three speaker + // operations today and its station can honestly say "Speakers". Add a kind + // from another group and it degrades to the generic name rather than + // continuing to advertise a label that now describes two thirds of it. + assert.equal( + operationsGroupLabel([ + "diarization", + "attribution-diarized", + "attribution-text", + ]), + "Speakers", + ); + assert.equal( + operationsGroupLabel(["diarization", "digest"]), + "Derived data", + ); + assert.equal(operationsGroupLabel([]), "Derived data"); + // Unknown ids contribute nothing rather than poisoning a single-group set. + assert.equal(operationsGroupLabel(["digest", "no-such-op"]), "Digest"); +}); + +test("a group has a station name AND a name you can put a verb in front of", () => { + // "Run speakers work" is why these are two declarations rather than one + // lower-cased derivation. The station eyebrow needs a noun; the button needs + // an object. + assert.equal( + operationsActionLabel([ + "diarization", + "attribution-diarized", + "attribution-text", + ]), + "speaker work", + ); + assert.equal(operationsActionLabel(["digest"]), "digests"); + assert.equal(operationsActionLabel([]), "derived data"); + assert.equal(operationsActionLabel(["diarization", "digest"]), "derived data"); +}); + +test("attribution-text's cost basis states the CHUNK unit, not the video", () => { + // The measured fact that hid behind the word "backfill": this lane's unit is + // the transcript chunk, so its 11,337 reachable videos are on the order of + // 194,000 model calls. Every other lane in the table is per-video, and a + // reader who assumes that of this one is wrong by ~17x. + assert.match(operationCostBasis("attribution-text"), /chunk/); + assert.match(operationCostBasis("attribution-diarized"), /per video/); + assert.match(operationCostBasis("diarization"), /per video/); + assert.equal(operationCostBasis("no-such-operation"), ""); +}); + +// The accounting rule the whole feature turns on: reachable work and +// needs-re-acquiring are never added together. +test("counts keep reachable work and needs-re-acquiring apart", () => { + const counts = emptyOperationCounts(); + for (const state of [ + "missing", + "missing", + "stale", + "missing-input", + "missing-input", + "missing-input", + "present", + "not-applicable", + ] as OperationClassification[]) { + addOperationState(counts, state); + } + assert.deepEqual(counts, { + missing: 2, + stale: 1, + partial: 0, + missingInput: 3, + deferred: 0, + blocked: 0, + }); + // 3, not 6. Measured on the real corpus the difference is 835 vs 77,105, and + // reporting the larger number is what would make every surface useless. + assert.equal(reachableOperationWork(counts), 3); +}); + +test("deferred is counted, and is NOT reachable work", async () => { + const counts = emptyOperationCounts(); + for (const state of [ + "missing", + "deferred", + "deferred", + "stale", + ] as OperationClassification[]) { + addOperationState(counts, state); + } + assert.deepEqual(counts, { + missing: 1, + stale: 1, + partial: 0, + missingInput: 0, + deferred: 2, + blocked: 0, + }); + // 2, not 4. This is the assertion that keeps a capped corpus from ever reading + // as finished, and the one that fails if someone "tidies up" by folding + // deferred into the total. + assert.equal(reachableOperationWork(counts), 2); +}); + +// --------------------------------------------------------------------------- +// PARTIAL: some sections at the current identity, some not. + +test("digest: some sections fresh and some not is PARTIAL, not stale", async () => { + // THE CASE THIS EXISTS FOR. `sections` is a setting and the pre-sweep decision + // is to turn tags on; the moment that happens every already-digested video in + // the corpus is part-done at once. Without the split all ~77,000 would read as + // `stale` — indistinguishable on a stage card from a PROMPT_VERSION bump that + // really did invalidate everything, when in fact digestVideo would regenerate + // only the tags. + assert.equal( + await classifyDigest({ + sidecar: digestSidecar(), + sections: ["chapters", "tags"], + }), + "partial", + ); + // The same sidecar against the section list it was made for is DONE, which is + // what makes the line above about the sections and not about the sidecar. + assert.equal( + await classifyDigest({ sidecar: digestSidecar(), sections: ["chapters"] }), + "present", + ); +}); + +test("digest: partial is REACHABLE work, unlike deferred and blocked", () => { + // The distinction is about what the work COSTS, never about whether the lane + // can do it. A part-done video is dispatched exactly like a stale one, so the + // sum must be unchanged by splitting them — otherwise a corpus mid-tags- + // backfill reports less work than it has. + const split = emptyOperationCounts(); + for (const st of ["missing", "partial", "stale"] as OperationClassification[]) { + addOperationState(split, st); + } + assert.equal(reachableOperationWork(split), 3); + const blockedAndDeferred = emptyOperationCounts(); + for (const st of ["blocked", "deferred"] as OperationClassification[]) { + addOperationState(blockedAndDeferred, st); + } + assert.equal(reachableOperationWork(blockedAndDeferred), 0); +}); + +test("reachableOperationWork survives a snapshot written before `partial`", () => { + // Every snapshot on disk predates the field. Reading it as undefined and + // adding it would produce NaN, which renders as "NaN" and sorts unpredictably + // — strictly worse than under-reporting. + const old = { + missing: 2, + stale: 1, + missingInput: 5, + deferred: 0, + blocked: 0, + } as unknown as ReturnType<typeof emptyOperationCounts>; + assert.equal(reachableOperationWork(old), 3); +}); + +// --------------------------------------------------------------------------- +// backfillLaneEntriesOf: the read-side twin of backfillLaneOperations. + +test("backfillLaneEntriesOf keeps a digest entry OUT of the lane's sums", () => { + // THE REGRESSION THIS WHOLE DESIGN EXISTS TO PREVENT. Four surfaces summed + // Object.values(snapshot.backfill) on the assumption that the map WAS the + // backfill lane. Now that the snapshot carries an entry per catalog operation, + // that assumption would fold ~75,000 digest videos into the dashboard's + // backfill instrument, /actionable's backfill rows and the widget. + const backfill = { + diarization: { ...emptyOperationCounts(), missing: 3, ids: [], eligible: 3 }, + digest: { + ...emptyOperationCounts(), + missing: 75_000, + ids: [], + eligible: 75_000, + }, + }; + const lane = backfillLaneEntriesOf(backfill); + assert.equal(lane.length, 1); + assert.equal( + lane.reduce((n, e) => n + reachableOperationWork(e), 0), + 3, + ); +}); + +test("backfillLaneEntriesOf filters by the DECLARATION, not by a hardcoded id", () => { + // `key !== "digest"` would pass the test above and leave the identical trap + // armed for the next operation registered on a lane of its own. The rule is + // the queue key, asked of the registry — so every lane kind is in, and an id + // the catalog does not know is out. + const laneIds = backfillLaneOperations(settingsWithDiarization()).map((k) => k.id); + const backfill: Record<string, ReturnType<typeof emptyOperationCounts> & { ids: string[] }> = {}; + for (const id of [...laneIds, "digest", "some-kind-from-a-newer-build"]) { + backfill[id] = { ...emptyOperationCounts(), missing: 1, ids: [] }; + } + assert.equal(backfillLaneEntriesOf(backfill).length, laneIds.length); + assert.ok(laneIds.length > 0, "expected at least one lane kind enabled"); + // Empty and absent are both simply nothing, never a throw: a snapshot may + // predate the field entirely. + assert.deepEqual(backfillLaneEntriesOf(undefined), []); + assert.deepEqual(backfillLaneEntriesOf({}), []); +}); + +test("backfillLaneOperationEntriesOf applies the SAME filter, keyed by kind", () => { + // The keyed form exists so the corpus-wide backfill card can say WHICH kind a + // number came from: summed, this lane reads "77,952 reachable · 77,134 need + // media", where 99.5% of the first is attribution-text at ~1 model call per + // transcript CHUNK and all of the second is diarization at a few hundred audio + // passes. A breakdown that re-derived its own filter is how a "Digest" row + // ends up on the backfill card contradicting the figure above it — so the two + // are one function, and this asserts they cannot drift. + const laneIds = backfillLaneOperations(settingsWithDiarization()).map((k) => k.id); + const backfill: Record< + string, + ReturnType<typeof emptyOperationCounts> & { ids: string[] } + > = {}; + for (const id of [...laneIds, "digest", "some-kind-from-a-newer-build"]) { + backfill[id] = { ...emptyOperationCounts(), missing: 1, ids: [] }; + } + assert.deepEqual( + backfillLaneOperationEntriesOf(backfill).map(([id]) => id).sort(), + [...laneIds].sort(), + ); + // Exactly the entries backfillLaneEntriesOf returns, in the same order. + assert.deepEqual( + backfillLaneOperationEntriesOf(backfill).map(([, e]) => e), + backfillLaneEntriesOf(backfill), + ); + assert.deepEqual(backfillLaneOperationEntriesOf(undefined), []); + assert.deepEqual(backfillLaneOperationEntriesOf({}), []); +}); + +test("presentOperationWork says UNKNOWN rather than zero on an old snapshot", () => { + // A 0 here would render as "nothing digested" on a fully digested channel, + // which is the most dangerous direction for a coverage number to be wrong. + assert.equal( + presentOperationWork({ ...emptyOperationCounts(), ids: [] }), + null, + ); + assert.equal( + presentOperationWork({ + ...emptyOperationCounts(), + missing: 2, + blocked: 1, + ids: [], + eligible: 10, + }), + 7, + ); +}); diff --git a/common/lib/operations.ts b/common/lib/operations.ts @@ -0,0 +1,1728 @@ +// The single source of truth for what an OPERATION is. +// +// An operation is a derived-data feature that declares what it needs — its +// inputs, its per-video state probe, what one video of it costs — and the +// system supplies the lane it runs on, the resource share and the indicator. +// +// The problem this exists for repeats: a derived-data feature lands, and the +// corpus that already exists does not have what it needs. For diarization that +// input is AUDIO, which cleanAudioFromTranscribed deletes once a video is +// transcribed. Before this table the only catch-up was a per-channel button on +// a controller written for that one feature (controller/diarizeAll.ts) — there +// was no way to ask "how much of the corpus is missing this?", and no way to run +// catch-up alongside new-video work without one starving the other. +// +// So, in the shape jobKinds.ts already uses: adding an operation should mean +// adding ONE entry here, and the system supplies the lane, the resource share +// and the indicator. +// +// WHAT "BACKFILL" STILL MEANS IN THIS FILE. The backfill LANE — BACKFILL_QUEUE, +// backfillLaneOperations, backfillLaneEntriesOf, controller/backfillBatch.ts and +// controller/backfillSweep.ts — is ONE QUEUE that several operations share: one +// pause, one sweep, one share. It is the only thing this file still calls +// "backfill", and its persisted contracts (the `backfill` key on the snapshot, +// `settings.backfill`, the `backfill-channel` and `backfill-sweep` job kinds) +// keep the word because they are on disk. An operation is not a backfill; a +// backfill is what the lane does to the corpus that predates an operation. +// Anything named `*Kind*` outside this file's exports means "one entry of this +// registry", which stays true. +// +// FOUR STATES, NOT TWO, and the split is the load-bearing part. Measured on this +// corpus at the time of writing: 77,106 videos, 836 with media still on disk, 1 +// diarized. A single "remaining" number would therefore read 77,105 — and 91x of +// that is unreachable without re-downloading. The repo has already been burned by +// exactly this once: editor/app/api/widget/actionable/route.ts deliberately +// refuses to filter on `noDigest` because during the backfill that is 99.87% of +// the corpus and counting it would put every channel in the list forever. So +// `missing` (reachable now) and `missing-input` (needs re-acquiring) are +// SEPARATE numbers, everywhere, and no surface is allowed to add them together. +// +// STALENESS IS PROVENANCE, NOT AGE. `state` is derived from disk on every read +// and never stored, and an operation that records what produced its output +// compares that against what we would produce now — lib/digest.ts's +// isSectionFresh, whose absent-field-equals-today's-default trick is what stops +// adding a field from invalidating the whole corpus. Diarization writes that +// provenance and, before this, had no comparator at all: diarizeOne +// short-circuited on mere existence, so the two most likely reasons to re-run (a +// threshold or model change) left everything looking done. +// +// SERVER-ONLY, despite living in lib/. It reads the filesystem and calls a +// controller, so it is `-server.ts` in everything but name; the path is the one +// the plan named. No client component imports it — the UI is handed plain +// numbers off the channel snapshot, and labels as props. +// +// THREE ENTRIES, AND THE SECOND PAIR IS WHAT MAKES THIS AN ABSTRACTION. A +// registry with one entry is a wrapper: nothing proved that "a feature declares +// what it needs and the system supplies the lane, the share and the indicator" +// was true. Attribution is the test of it, and it passed — registering +// `attribution-diarized` and `attribution-text` lit the channel stage card, +// /actionable, the dashboard instrument and the widget strip with ZERO UI +// changes, because all four iterate snapshot.backfill[operationId]. +// +// It also exercised the parts of the shape that one entry could not: +// `missing-input` for something other than audio (diarization.json, of which +// this corpus has one), and two operations writing the SAME FILE at different +// quality tiers — see the ordering rule above the attribution entries. +// +// WHAT IS STILL NOT HERE. The other catch-up mechanisms in the repo do not fit +// this per-video probe, and forcing them in would make the table lie: +// controller/backfillAvailability.ts is CHANNEL-scoped (one JSON map, folded +// into sync at runYtdlp.ts), and controller/normalizeAll.ts has no recorded +// provenance to compare, so its "stale" is undefined. `tier` still exists +// because it is what keeps them apart if they are ever added: `inline` folds +// into an existing pass and `lane` gets the concurrent queue and the share. + +import type { Paths } from "./paths"; +import type { + AttributionSettings, + DiarizationSettings, + SiteSettings, +} from "./settings"; +import { + BACKFILL_QUEUE, + DIGEST_LOCAL_QUEUE, + DIGEST_REMOTE_QUEUE, + TRANSCRIPTION_QUEUE, +} from "./queueKeys"; +// TYPE-ONLY, and that is what keeps this safe: the import is erased at compile +// time, so naming the runner union here cannot create a cycle no matter what +// jobs/autoQueueState.ts imports. +import type { AutoQueueKind } from "../jobs/autoQueueState"; +import { + DIGEST_FILENAME, + isDigestSectionKind, + isSectionFresh, + type DigestAppConfig, + type DigestFreshnessTarget, + type DigestItem, + type DigestLane, + type DigestProvenance, + type DigestRecord, + type DigestSectionKind, + type DigestWarning, +} from "./digest"; +import { loadDigest, writeDigestSection } from "./digest-server"; +import type { DigestContext } from "./digestContext-server"; +import { + SORTFORMER_DIARIZATION_ENGINE, + diarizationTarget, + isDiarizationFresh, + type DiarizationBackend, + type DiarizationEngineId, + type DiarizationFreshnessTarget, + type DiarizationRecord, +} from "./diarization"; +import { loadDiarization, writeDiarization } from "./diarization-server"; +import { + ATTRIBUTION_FILENAME, + isAttributionDowngrade, + isAttributionFresh, + type AttributionFreshnessTarget, + type AttributionMethod, + type AttributionRecord, +} from "./attribution"; +import { loadAttribution, writeAttribution } from "./attribution-server"; +import { + CUES_JSON_FILENAME, + DIARIZATION_FILENAME, + META_FILENAME, + WHISPER_FILENAME, + findSourceMedia, + isVideoTranscribed, + readVideoDurationSec, + type VideoFiles, +} from "./videoStatus"; +import { pickPreferredAudio } from "./mediaFiles"; +import { SAVED_VIDEO_POINTER_FILENAME } from "./savedVideo"; +import { resolveSavedVideo } from "./savedVideo-server"; +import { diarizeOneVideo } from "../controller/diarizeOne"; +// The TARGET resolver only — a settings read plus the digest app registry. +// attributeOne is loaded LAZILY inside run() below: this module is imported by +// controller/channelSnapshot.ts, which classifies every video of every channel, +// so its eager import graph sits on the editor's hot path, and a classification +// needs none of the runner's (the transcript normalizer, the markdown renderer, +// the digest prompt module, the channel-context reader). See +// controller/attributionTarget.ts for what this is and is not worth — it is a +// structural argument, not a measured speedup. +import { resolveAttributionTarget } from "../controller/attributionTarget"; +import type { AttributeOneOutcome } from "../controller/attributeOne"; +import { isCuesJsonFresh } from "../controller/normalizeTranscript"; + +// What a video's relationship to a backfill is, right now, read from disk. +// +// present — has it, at the identity we would produce now. +// stale — has it, but from a different engine/model/threshold. +// partial — has SOME of it at the current identity and not the rest. +// Only meaningful for a kind whose output has parts; today +// that is digest alone, whose sections are generated and +// compared independently. See the digest entry for why this +// is not just a nicer word for `stale`. +// missing — does not have it, and the input to produce it is HERE. +// missing-input — does not have it, and the input is gone. Reachable only by +// re-acquiring the media, which is opt-in and bounded. +// deferred — does not have it, the input is here, and the kind refuses to +// attempt it under the current configuration. Today that is +// only the diarization duration cap. NEVER summed into +// reachable work, so a capped corpus cannot read as finished. +// blocked — does not have it, and what it is waiting for is the OUTPUT OF +// ANOTHER KIND IN THIS TABLE. Not the same thing as +// missing-input, and conflating them was a real defect: see +// below. +// +// `deferred` and `blocked` follow the house rule OperationRunOutcome = "skipped" +// already sets on the run side: a deliberate non-action gets its own counter, +// and is never folded into the work total nor reported as a failure. +// +// WHY `blocked` IS NOT `missing-input`. `missing-input` means one specific +// thing to the rest of the system: THE MEDIA IS GONE, RE-ACQUIRE IT. It is the +// population `allowRedownload` exists for, and backfillReacquire answers it by +// fetching AUDIO. `attribution-diarized` waits on diarization.json — so +// reporting that as missing-input told the operator ~73,000 videos needed media +// re-fetched, and with re-download on, the lane would spend a download per video +// fetching audio that CANNOT satisfy the wait, then cleaning it up again. The +// prerequisite is not gone; it has not been produced yet, and this same table +// knows how to produce it. +export type OperationState = + | "present" + | "stale" + | "partial" + | "missing" + | "missing-input" + | "deferred" + | "blocked"; + +// Videos this backfill has no opinion about (not transcribed, marked +// untranscribable). Kept out of OperationState so it can never be counted. +export type OperationClassification = OperationState | "not-applicable"; + +// How expensive one video is, which decides where the work runs. +// +// inline — microseconds to cheap I/O; folds into a pass that already walks the +// corpus, and never gets a lane of its own. +// lane — expensive enough to need its own queue and a resource share. +// Diarization is ~500-680 s/audio-hour of CPU. +export type OperationTier = "inline" | "lane"; + +// WHERE an operation's work runs, and what it fights with while it runs. +// +// This exists because "put every backfill on BACKFILL_QUEUE" is wrong, and +// measurably so. registry.ts submits every non-empty queueKey at concurrency 1, +// so a shared key SERIALIZES. Diarization is CPU and digest is GPU/ollama; +// today they run at the same time on separate queues, and collapsing them onto +// one key would idle the GPU while the CPU works and vice versa, across a +// multi-week sweep. So the queue is declared per operation rather than owned by +// the lane, and a scheduler dispatches one job per distinct queueKey. +export type Lane = { + // The registry queue key. Distinct keys are the ONLY mechanism for + // concurrency between operations; the same key is the only mechanism for + // serializing an operation against itself. + queueKey: string; + // The scarce resource one unit of this work occupies. Not decoration: it is + // what decides whether the lane must stand aside for transcription. + // + // gpu — contends with the transcription engine for VRAM. Yields. + // cpu — contends for cores. Does not yield today; its share is governed + // by the backfill lane's own weight (idle-only at the default). + // network — a metered or remote API. Contends with nothing local, so it + // must NOT yield: doing so would park a lane that was costing + // nothing to keep running. + contendsFor: "gpu" | "cpu" | "network"; +}; + +// Whether this system dispatches the operation, or merely knows about it. +// +// "backfill" — the backfill machinery pulls candidates and runs it. +// "external" — something else owns dispatch (autoRunner, the worker pool). +// Registered anyway so the catalog is complete and a dependency +// on it resolves. See EXTERNAL_OPERATIONS. +export type OperationDispatch = "backfill" | "external"; + +export type OperationProbe = { + videoDir: string; + videoId: string; + // Already read by the caller. Taking it rather than re-reading is what makes + // the channel snapshot's per-video classification free — see VideoFiles.entries. + files: VideoFiles; + target: unknown; + // The live settings. Passed in rather than read here so a classification stays + // a pure function of what the caller already has, and so a kind can consult a + // knob that must NOT become part of its freshness identity — the diarization + // duration cap is exactly that: `target` is compared by isDiarizationFresh, so + // putting the cap there would mark every sidecar on disk stale the moment the + // cap moved. + settings: SiteSettings; +}; + +export type OperationRunOptions = { + paths: Paths; + videoDir: string; + videoId: string; + channelSlug: string; + target: unknown; + // Redo a `present` video anyway (an operator forcing a regeneration). + force?: boolean; + // Engine-config override for the LLM-backed kinds (attribution today). The + // fan-out passes the primary's resolved config with only `baseUrl` swapped + // to a leased endpoint — baseUrl is not part of the freshness identity, so + // which endpoint served a call can vary freely while everything that IS + // identity (model, numCtx, timeouts) travels verbatim from the primary. + appConfig?: DigestAppConfig; + // Per-kind settings injection for a UNIT EXECUTOR (see workerServer's + // startWorkerUnit). The primary resolves these and ships them in the unit + // envelope; each kind's run() threads its own field into the controller's + // existing `settings` override — so a bare executor's default settings + // (attribution disabled, empty app config = a DIFFERENT identity) can never + // leak into provenance. On the primary these stay unset and the controllers + // read live settings exactly as before. + attributionSettings?: AttributionSettings; + diarizationSettings?: DiarizationSettings; + // Channel context, injected rather than shipped as a file: the primary + // already computed the note + hash, and digest pins contextHash in its + // identity — injecting the object makes hash equality true by construction. + digestContext?: DigestContext; + onLog?: (msg: string) => void; + signal?: AbortSignal; +}; + +// Deliberately mirrors DiarizeOneOutcome's discipline: an expected condition is +// an outcome, never a throw. The batch counts these and keeps going. +export type OperationRunOutcome = + | "done" + | "already-present" + | "missing-input" + // Nothing to do for a reason state() could not see WITHOUT a per-video file + // read — the case that matters is a transcript whose cues.json is stale, which + // costs a read of the source file and cannot be paid 77,000 times per pass. + // Counted separately from `failed` so a transient condition that resolves + // itself does not report as a broken engine. + | "skipped" + | "not-configured" + | "disabled" + | "failed"; + +// WHICH PIPELINE THIS OPERATION BELONGS TO, as a declared field rather than a +// list somebody maintains in a component. +// +// Three surfaces need this same grouping and each used to hardcode its own copy: +// the channel transit line (which station does this operation live under), the +// /channels table (which columns sit together), and the lane card (which +// operations does this queue actually hold). Three hardcoded lists is three +// places to forget when a kind is added — and the last time one was added, the +// transit line kept summing three unrelated operations into one station because +// nobody updated its list. +// +// Deriving the label from the group also means a lane holding a MIX cannot go +// stale: it falls back to "Derived data" rather than naming two of its three +// members. +export type OperationGroup = "media" | "transcript" | "digest" | "speakers"; + +// Group order: upstream first. The /channels columns and the transit line both +// lay their pipelines out in this order, so a reader moving between the two +// pages sees the same left-to-right sequence. +export const OPERATION_GROUP_ORDER: readonly OperationGroup[] = [ + "media", + "transcript", + "digest", + "speakers", +]; + +export function groupLabel(group: OperationGroup): string { + switch (group) { + case "media": + return "Media"; + case "transcript": + return "Transcript"; + case "digest": + return "Digest"; + case "speakers": + return "Speakers"; + } +} + +export type Operation = { + id: string; + label: string; + // One line of UI copy: what this backfill is, in the operator's terms. + hint: string; + // The pipeline this operation belongs to. See OperationGroup. + group: OperationGroup; + // The label at COLUMN width — one or two words, for a header that has to sit + // above a 48px band on a table 68 rows deep. + // + // Declared rather than abbreviated in the component, for the same reason + // `group` is: "Speaker names (from the transcript)" cannot be shortened + // mechanically, and a map of abbreviations maintained next to a table is a + // second place to forget when a kind is added. + shortLabel: string; + // WHAT ONE VIDEO OF THIS OPERATION COSTS, as a cost basis — "one audio pass + // per video", "~1 model call per transcript chunk". + // + // "UNIT" IS NOT THE WORD FOR THIS, and the rule is one-sided. A UNIT is one + // item of dispatchable work: what ArbiterUnit, startWorkerUnit, + // runUnitViaRemote and /api/worker/unit move around, and three of those are + // persisted contracts (the route, the `worker-unit` and `auto-download-unit` + // job kinds). What one video of an operation costs is its COST BASIS, and no + // surface calls that a unit — a prop that once said `unit` for a population + // label ("reachable", "blocked") is `population`, and prose that said "unit" + // for cost says "cost basis". The dispatch side keeps its name. + // + // The fact this exists to surface: an operation can be armed at enormous cost + // and read as a quiet row. attribution-text is reachable on 11,337 videos of + // one channel and roughly 194,000 model calls corpus-wide, it has completed + // ONE video, and every screen that mentioned it said only "Backfill". Printing + // the cost basis beside the backlog is what makes 11,337 legible. + // + // Deliberately NO threshold and no editorialising. A "this is a lot" cutoff + // would be a magic number the next operation gets wrong, and the operator is + // the one who decides what is too expensive. + costBasis: string; + // What a `deferred` video of THIS kind is waiting for, and what an operator + // can do about it. Belongs to the kind, not to the card: BackfillStage used to + // hardcode "too long to diarize under the current limit", which was correct + // only while diarization was the sole kind that could defer. Digest defers for + // an unrelated reason (no current normalized transcript), so a card summing + // several kinds' `deferred` into one hardcoded sentence now states a cause + // that is false for most of what it counts. + // + // Written as a sentence FRAGMENT completing "N videos are …", so the card + // keeps ownership of the count and its pluralization. + deferredHint?: string; + tier: OperationTier; + // Where this operation's work runs. See Lane — the queue key is what + // keeps CPU and GPU operations overlapping instead of taking turns. + // + // This is the DECLARED lane, which for a kind with a choice means its default. + // Prefer laneFor() when a live answer is needed. + lane: Lane; + // The lane this kind would actually use under the given settings, for the + // kinds whose scarce resource is a configuration choice rather than a fact. + // Diarization is one: sherpa-onnx is CPU-only, while sortformer on the Vulkan + // backend holds ~4.4 GB of the same 8 GB card the transcription engine wants. + // + // Optional because most kinds have no choice, and `lane` is the answer for + // them. Mirrors digestLaneFor, which solved the same problem for the digest + // operation's two lanes — and which digest itself now declares. + // + // OPTIONAL IS LOAD-BEARING, not tidiness. controller/backfillBatch.ts reads + // this field's PRESENCE as the marker for "this kind's resource depends on + // settings, so nothing else is deciding it for us" and makes the whole run + // idle-only when such a kind could take the GPU. Giving every kind a laneFor + // that defaults to `lane` would therefore not be a no-op: it would enrol + // every kind in that rule and make the lane idle-only whenever a statically + // GPU-bound kind was in the run — the exact regression the guard's comment + // records. Add one only where the lane genuinely varies. + laneFor?(settings: SiteSettings): Lane; + // Ids of other kinds in this table whose output this one consumes. + // + // PURELY DECLARATIVE. It does not gate anything by itself — a kind still + // decides for itself, from disk, whether its input is there, and says so by + // returning `blocked`. What declaring it buys is two things nothing else + // could: resolveBackfillLaneOperations can order a prerequisite before its dependant + // within a single pass (so a video diarized this pass can be attributed in + // the same one, rather than waiting for whatever LATER pass happens to find + // the sidecar on disk), and a surface can say what a blocked video is waiting + // FOR rather than just that it is stuck. + // + // An id naming a kind that is absent or disabled is not an error: the + // dependency simply imposes no ordering, and the dependant keeps reporting + // `blocked` until something produces its input. + dependsOn?: readonly string[]; + // The feature's OWN gate. A disabled feature reports no backfill at all — + // otherwise every surface would advertise catch-up work for something the + // operator has switched off. + enabled(settings: SiteSettings): boolean; + // The identity we would produce now, resolved ONCE per run rather than per + // video. `unknown` here is the one erasure point in the table: each entry + // narrows it back to its own type on the line below. The alternative — making + // the whole registry generic — infects every consumer with a type parameter + // for no gain, since none of them look inside a target. + // + // ASYNC AND CHANNEL-SCOPED, since the digest entry. A digest's identity + // includes the hash of the channel's context note, which is a file read — so + // this can no longer be a pure function of settings. It is still resolved + // ONCE PER RUN, which is the property that matters (deriving it per video is + // how a counter and a runner end up disagreeing about what is stale); the + // callers simply await it now. + resolveTarget(ctx: OperationTargetContext): unknown | Promise<unknown>; + state(probe: OperationProbe): Promise<OperationClassification>; + run(opts: OperationRunOptions): Promise<OperationRunOutcome>; + + // --- The unit-executor contract: what one unit of this kind needs, what it + // produces, and how its output lands back on the primary. --- + + // Filenames (within the video dir) a unit executor must be shipped, derived + // from the listing the caller already has. ORDERED: the executor + // materializes them in this order, and every kind lists + // transcript.cues.json LAST — isCuesJsonFresh compares mtimes, and a cues + // file written before its metadata/raw transcript reads as stale, making + // the unit silently do nothing. + inputs(files: VideoFiles): string[]; + // Filenames the run writes — what the executor's result endpoint returns. + outputs: readonly string[]; + // Apply a unit's returned output files ON THE PRIMARY, through the guarded + // writers — never a raw file copy. Both sidecar writers are + // read-modify-write (writeDigestSection preserves the other section; + // attribution re-checks the downgrade rule against the primary's CURRENT + // disk, because the unit ran against a snapshot that is minutes old). + // "invalid" = the payload is not a usable record; "refused" = a guard said + // no (which is a success of the guard, not a failure of the unit). + applyResult( + videoDir: string, + payload: Record<string, string>, + ): Promise<ApplyResultOutcome>; +}; + +export type ApplyResultOutcome = "applied" | "refused" | "invalid"; + +export type OperationTargetContext = { + settings: SiteSettings; + paths: Paths; + channelSlug: string; +}; + +// --------------------------------------------------------------------------- +// Unit-contract helpers, shared across the kinds. +// --------------------------------------------------------------------------- + +// The resolved primary raw transcript — needed by every transcript-derived +// kind because isCuesJsonFresh compares the cues sidecar's mtime against it. +function rawTranscriptOf(files: VideoFiles): string | null { + if (files.hasWhisper) return WHISPER_FILENAME; + return files.ytVttFile; +} + +// The common transcript-derived input set: metadata + raw transcript + the +// existing output sidecar (for the freshness/downgrade checks) + CUES LAST — +// see Operation.inputs for why the order is load-bearing. +function transcriptUnitInputs( + files: VideoFiles, + extras: readonly string[], +): string[] { + const out: string[] = []; + if (files.entries.includes(META_FILENAME)) out.push(META_FILENAME); + const raw = rawTranscriptOf(files); + if (raw) out.push(raw); + for (const name of extras) { + if (files.entries.includes(name)) out.push(name); + } + if (files.entries.includes(CUES_JSON_FILENAME)) out.push(CUES_JSON_FILENAME); + return out; +} + +// Shared by both attribution kinds: parse, validate the same shape +// loadAttribution enforces, RE-CHECK the downgrade rule against the primary's +// current disk (the unit ran against a snapshot minutes old, and the diarized +// lane can have landed a better record meanwhile), then the guarded writer. +async function applyAttributionResult( + videoDir: string, + payload: Record<string, string>, +): Promise<ApplyResultOutcome> { + const raw = payload[ATTRIBUTION_FILENAME]; + if (!raw) return "invalid"; + let record: AttributionRecord; + try { + const parsed = JSON.parse(raw) as Partial<AttributionRecord>; + if ( + typeof parsed?.videoId !== "string" || + typeof parsed.generatedAt !== "string" || + !Array.isArray(parsed.speakers) || + !Array.isArray(parsed.segments) || + !parsed.provenance || + typeof parsed.provenance.method !== "string" + ) { + return "invalid"; + } + record = parsed as AttributionRecord; + } catch { + return "invalid"; + } + if ( + isAttributionDowngrade( + await loadAttribution(videoDir), + record.provenance.method as AttributionMethod, + ) + ) { + return "refused"; + } + await writeAttribution(videoDir, record); + return "applied"; +} + +// Digest results land SECTION-WISE through writeDigestSection, which preserves +// the other section, its warnings and the history from the record already on +// the primary's disk — a raw copy of the unit's file would clobber a section +// the primary wrote while the unit was in flight. +async function applyDigestResult( + videoDir: string, + payload: Record<string, string>, +): Promise<ApplyResultOutcome> { + const raw = payload[DIGEST_FILENAME]; + if (!raw) return "invalid"; + let parsed: Partial<DigestRecord>; + try { + parsed = JSON.parse(raw) as Partial<DigestRecord>; + } catch { + return "invalid"; + } + const sections = parsed?.sections; + if (!sections || typeof sections !== "object") return "invalid"; + let applied = false; + for (const [section, value] of Object.entries(sections)) { + if (!isDigestSectionKind(section)) continue; + const v = value as { + provenance?: DigestProvenance; + items?: DigestItem[]; + }; + if (!v?.provenance || !Array.isArray(v.items)) continue; + const warnings: DigestWarning[] = (parsed.warnings ?? []).filter( + (w) => w.section === section, + ); + await writeDigestSection(videoDir, { + section, + items: v.items, + provenance: v.provenance, + warnings, + }); + applied = true; + } + return applied ? "applied" : "invalid"; +} + +// Diarization is the ONE whole-file verbatim apply: the sidecar has a single +// writer and no sections to merge, so the shape check plus the atomic writer +// is the whole guard. +async function applyDiarizationResult( + videoDir: string, + payload: Record<string, string>, +): Promise<ApplyResultOutcome> { + const raw = payload[DIARIZATION_FILENAME]; + if (!raw) return "invalid"; + try { + const parsed = JSON.parse(raw) as Partial<DiarizationRecord>; + if ( + typeof parsed?.generatedAt !== "string" || + !Array.isArray(parsed.turns) + ) { + return "invalid"; + } + await writeDiarization(videoDir, parsed as DiarizationRecord); + return "applied"; + } catch { + return "invalid"; + } +} + +// Diarization: the first entry, and the reason the table exists. +const diarization: Operation = { + id: "diarization", + label: "Speaker diarization", + hint: "Speaker turns captured from the audio, written to diarization.json beside the transcript.", + group: "speakers", + shortLabel: "Diarize", + costBasis: "one pass over the audio per video", + // The wording BackfillStage used to hardcode for every kind at once. + deferredHint: + "too long to diarize under the current limit — raise or clear Max audio hours in Settings to include them", + tier: "lane", + // CPU, on the shared backfill queue. Serialized against the other backfill + // kinds on purpose — two channels' worth of diarization at once just thrashes + // cores. How much of the machine it may take is backfillLimit()'s question, + // not the queue's. + // The DEFAULT engine's answer, written out rather than derived: settings.ts is + // a TYPE-only import here, and pulling defaultDiarization() in as a value would + // make this module's initialization depend on it at runtime. laneFor is the + // live answer, and diarizationLaneFor is where the rule actually lives. + lane: { queueKey: BACKFILL_QUEUE, contendsFor: "cpu" }, + laneFor: (settings) => diarizationLaneFor(settings.diarization), + enabled: (settings) => + settings.diarization.enabled && + !!settings.diarization.segModel && + !!settings.diarization.embModel, + // The SAME target diarizeOne's own short-circuit uses. Two derivations would + // let the counter and the runner disagree about what is stale. + resolveTarget: ({ settings }): DiarizationFreshnessTarget => + diarizationTarget(settings.diarization), + async state({ videoDir, files, target, settings }) { + // Same eligibility as diarizeAll's transcribedOnly default: the capture lane + // exists to pair speaker turns with a transcript, and an untranscribed + // video's audio is not at risk from the cleanup sweep yet. + if (!isVideoTranscribed(files) || files.isUntranscribable) { + return "not-applicable"; + } + // Cheap negative first: no sidecar in the listing means no read at all. + if (files.hasDiarization) { + const record = await loadDiarization(videoDir); + // A malformed file reads as ABSENT here, exactly as hasDiarization() in + // diarization-server.ts treats it: a half-written sidecar must never be + // what convinces anything the work is done. + if (record) { + return isDiarizationFresh(record, target as DiarizationFreshnessTarget) + ? "present" + : "stale"; + } + } + if (!(await hasDiarizableInput(videoDir, files))) return "missing-input"; + // The duration cap, read ONLY here. This is the would-be-`missing` branch, + // which is ~835 videos corpus-wide rather than 77,000, and that gating is + // not optional: countBackfillWork calls state() for every video on every job + // start, and readVideoDurationSec reads and parses a file. + // + // Unknown duration is NOT deferred — an absent or unparseable + // metadata.info.json must not silently remove a video from the work list. + return (await isOverDiarizationCap(videoDir, settings.diarization)) + ? "deferred" + : "missing"; + }, + async run(opts) { + // diarizeOneVideo re-reads settings when none is passed, which is what we + // want: the batch may run for hours and a model change mid-run should be + // picked up. `force` is how a `stale` video gets redone at all — the + // existence short-circuit inside is now a freshness check, but an operator + // forcing a regeneration still needs to win. + const outcome = await diarizeOneVideo({ + paths: opts.paths, + videoDir: opts.videoDir, + videoId: opts.videoId, + settings: opts.diarizationSettings, + force: opts.force, + onLog: opts.onLog, + signal: opts.signal, + }); + if (outcome === "diarized") return "done"; + if (outcome === "already-exists") return "already-present"; + if (outcome === "no-audio") return "missing-input"; + if (outcome === "not-configured") return "not-configured"; + if (outcome === "disabled") return "disabled"; + return "failed"; + }, + // The one kind whose input is MEDIA: the preferred extracted audio, else the + // persisted source container. Same envelope as the transcript kinds, bigger + // files — and a reachable population of only ~836 videos, so modest use. + inputs(files) { + const out: string[] = []; + if (files.entries.includes(META_FILENAME)) out.push(META_FILENAME); + const audio = + pickPreferredAudio(files.audioFiles) ?? findSourceMedia(files.entries); + if (audio) out.push(audio); + return out; + }, + outputs: [DIARIZATION_FILENAME], + applyResult: applyDiarizationResult, +}; + +// Is there anything on disk ffmpeg could read for this video? Mirrors +// controller/diarizeOne.ts's resolveDiarizableMedia, but answered from the +// listing the caller already has so the common cases cost no I/O: +// extracted audio, then a persisted source container, then — only when the +// pointer file is actually present — the saved-video store. +async function hasDiarizableInput( + videoDir: string, + files: VideoFiles, +): Promise<boolean> { + if (files.audioFiles.length > 0) return true; + if (findSourceMedia(files.entries)) return true; + if (!files.entries.includes(SAVED_VIDEO_POINTER_FILENAME)) return false; + // The pointer exists but the stored file may not (an unmounted backup disk), + // so this last step really does have to touch the filesystem. + return (await resolveSavedVideo(videoDir)) !== null; +} + +// Is this video longer than the diarization duration cap? +// +// THE CAP IS OFF BY DEFAULT NOW — windowed diarization removed the OOM it +// existed for. It remains because a smaller machine, or a recording longer than +// anything measured here, may still want it. This function is where its two +// honest limitations live. First, duration is a PROXY: the memory blowup is O(n^2) in +// speech-SEGMENT count, and turn density varies 40x across this corpus, so a +// sparse 7h42m video is cheaper than a dense 6h12m one. Duration is used anyway +// because it is the only predictor available from metadata already on disk, for +// free, before committing 45 minutes of CPU to find out the hard way. Second, +// duration is the CONTAINER's, so a video whose metadata is missing or lies gets +// the benefit of the doubt. +// +// Unknown duration therefore returns false — not deferred. Deferring on an +// unreadable metadata.info.json would quietly delete work from the list on the +// strength of a file that could not be parsed, which is the opposite of what a +// third counter is for. +async function isOverDiarizationCap( + videoDir: string, + diarization: SiteSettings["diarization"], +): Promise<boolean> { + const capHours = diarization.maxAudioHours; + if (!capHours || capHours <= 0) return false; // cap off + const seconds = await readVideoDurationSec(videoDir); + if (seconds === null) return false; + return seconds > capHours * 3600; +} + +// --------------------------------------------------------------------------- +// Attribution — the second and third entries, and the ones that make this a +// registry rather than a wrapper around diarization. +// +// TWO KINDS, ONE FILE. Both write attribution.json, and the ordering rule in +// lib/attribution.ts is the whole safety of that: the diarized lane may +// overwrite a text-only record (an UPGRADE — that is what the second kind is +// for), and the text lane must never overwrite a diarized one (a DOWNGRADE). +// state() encodes it here and attributeOne re-checks it against disk immediately +// before writing, because the pool can hold a candidate for minutes after +// state() ran. It has a test; a comment would not have been enough. +// +// THE SPLIT IS ALSO WHAT MAKES THE UPGRADE QUEUE FREE. PLAN.md describes a +// bespoke "upgrade job" for turning text-only records into diarized ones. It is +// not needed: `attribution-diarized` reports a text-only record as MISSING work, +// so the existing lane, sweep and indicators queue the upgrade with no new +// machinery. What it needs re-acquiring media for is already +// controller/backfillReacquire.ts. +// +// AND IT IS WHY missing-input MATTERS HERE MOST. `attribution-diarized`'s input +// is diarization.json, of which this corpus has ONE. So its missing-input +// population is ~73,000 videos on day one — the exact case the reachable / +// needs-input split exists to stop from poisoning every surface. A single +// "remaining" number would put every channel at the top of every list forever. + +// Shared by both attribution kinds: a video only has speakers worth naming if it +// has a transcript. Not-applicable rather than missing-input, since re-acquiring +// media would not help — the video needs transcribing, which is another lane's +// job entirely. +function attributionApplies(files: VideoFiles): boolean { + return isVideoTranscribed(files) && !files.isUntranscribable; +} + +// The identity, minus the per-video half. The diarized lane's identity also +// includes the generatedAt of the diarization.json it names clusters from, and +// that is a disk read — so it is added inside the one state() branch that has +// already paid for the read. See AttributionProvenance.diarizationGeneratedAt. +function attributionTargetFor( + settings: SiteSettings, + method: AttributionMethod, +): AttributionFreshnessTarget { + return resolveAttributionTarget(method, settings.attribution).target; +} + +const attributionText: Operation = { + id: "attribution-text", + label: "Speaker names (from the transcript)", + hint: "Speakers reconstructed from the transcript alone, for videos with no diarization. Cheaper to reach, worse than the diarized lane, and it never overwrites one.", + group: "speakers", + shortLabel: "Names·T", + // The expensive one, and the reason costBasis is a field. A transcript is + // many chunks; this is the only lane in the table whose unit is not the video. + costBasis: "~1 model call per transcript chunk", + tier: "lane", + lane: { queueKey: BACKFILL_QUEUE, contendsFor: "network" }, + enabled: (settings) => + settings.attribution.enabled && settings.attribution.textOnlyEnabled, + resolveTarget: ({ settings }) => attributionTargetFor(settings, "text-only"), + async state({ videoDir, files, target }) { + if (!attributionApplies(files)) return "not-applicable"; + // NEVER missing-input. The input is the cue stream, and a transcribed video + // has one by definition — which is exactly why this lane can reach the whole + // corpus and why running it over the whole corpus costs ~194,000 model calls. + if (!files.entries.includes(ATTRIBUTION_FILENAME)) return "missing"; + const record = await loadAttribution(videoDir); + // A malformed file reads as ABSENT, the same rule the diarization entry + // uses. Here it also protects the write path: a half-written sidecar must + // not be able to masquerade as a diarized record and block this lane + // forever. + if (!record) return "missing"; + // THE DOWNGRADE RULE, in the counter as well as the runner. A diarized + // record is not stale for this lane and is not work — there is simply + // something better here. Reporting it as work would put this lane in a loop + // of "attempt, refuse, still outstanding" across every pass of a sweep. + if (isAttributionDowngrade(record, "text-only")) return "present"; + return isAttributionFresh(record, target as AttributionFreshnessTarget) + ? "present" + : "stale"; + }, + async run(opts) { + // Lazy, once, at the point of actually running something. See the import + // note at the top of this file. + const { attributeOneVideo } = await import("../controller/attributeOne"); + return toBackfillOutcome( + await attributeOneVideo({ + paths: opts.paths, + videoDir: opts.videoDir, + videoId: opts.videoId, + channelSlug: opts.channelSlug, + method: "text-only", + force: opts.force, + settings: opts.attributionSettings, + appConfig: opts.appConfig, + context: opts.digestContext, + onLog: opts.onLog, + signal: opts.signal, + }), + ); + }, + inputs: (files) => transcriptUnitInputs(files, [ATTRIBUTION_FILENAME]), + outputs: [ATTRIBUTION_FILENAME], + applyResult: applyAttributionResult, +}; + +const attributionDiarized: Operation = { + id: "attribution-diarized", + label: "Speaker names (from the audio)", + hint: "Names put to the speaker clusters in diarization.json — about one model call per video, and better than the text-only lane. Needs diarization to have run first.", + group: "speakers", + shortLabel: "Names·A", + costBasis: "~1 model call per video", + tier: "lane", + // The dependency the hint has always stated in prose. Declaring it is what + // turns "needs diarization to have run first" from a sentence an operator + // reads into something the scheduler can order by and a counter can name. + dependsOn: ["diarization"], + lane: { queueKey: BACKFILL_QUEUE, contendsFor: "network" }, + enabled: (settings) => + settings.attribution.enabled && settings.attribution.diarizedEnabled, + resolveTarget: ({ settings }) => attributionTargetFor(settings, "diarized"), + async state({ videoDir, files, target }) { + if (!attributionApplies(files)) return "not-applicable"; + // The input is diarization.json, NOT audio. That distinction is the whole + // reason this kind is cheap: the perishable input was already captured, and + // what is left is a naming pass that can be redone at any time. + // + // Deliberately NOT gated on settings.diarization.enabled — a sidecar + // captured during a past run is a perfectly good input after capture is + // switched off again, and refusing to name it would strand exactly the work + // the capture lane exists to protect. + // + // BLOCKED, NOT MISSING-INPUT. This used to say missing-input, and that was + // wrong in a way that cost real work: it put ~73,000 videos into the + // "re-acquire the media" population, where allowRedownload would fetch + // AUDIO — which can never satisfy a wait for diarization.json — and then + // delete it again. What this video is waiting for is the `diarization` kind + // declared in dependsOn above, and that is a thing this table produces. + if (!files.hasDiarization) return "blocked"; + if (!files.entries.includes(ATTRIBUTION_FILENAME)) return "missing"; + const record = await loadAttribution(videoDir); + if (!record) return "missing"; + // A text-only record here is THE UPGRADE QUEUE: the diarized record this + // kind is responsible for genuinely does not exist yet, so it is `missing` + // rather than `stale`. Both are reachable work, but the two words mean + // different things to an operator reading a stage card — "stale" says + // something changed under a record, "missing" says a better one was never + // made. + if (record.provenance.method !== "diarized") return "missing"; + // Only now is the diarization read worth paying for: it is needed solely to + // ask whether the clusters these names point at are still the same clusters. + const diarization = await loadDiarization(videoDir); + // Present in the listing but unreadable — a half-written or corrupt + // sidecar. Blocked for the same reason as the branch above, and note that + // the `diarization` kind reads a malformed record as ABSENT too, so it will + // regenerate this file and unblock the video without anyone intervening. + if (!diarization) return "blocked"; + return isAttributionFresh(record, { + ...(target as AttributionFreshnessTarget), + diarizationGeneratedAt: diarization.generatedAt, + }) + ? "present" + : "stale"; + }, + async run(opts) { + // Lazy, once, at the point of actually running something. See the import + // note at the top of this file. + const { attributeOneVideo } = await import("../controller/attributeOne"); + return toBackfillOutcome( + await attributeOneVideo({ + paths: opts.paths, + videoDir: opts.videoDir, + videoId: opts.videoId, + channelSlug: opts.channelSlug, + method: "diarized", + force: opts.force, + settings: opts.attributionSettings, + appConfig: opts.appConfig, + context: opts.digestContext, + onLog: opts.onLog, + signal: opts.signal, + }), + ); + }, + inputs: (files) => + transcriptUnitInputs(files, [DIARIZATION_FILENAME, ATTRIBUTION_FILENAME]), + outputs: [ATTRIBUTION_FILENAME], + applyResult: applyAttributionResult, +}; + +// One mapping, shared by both kinds, so the two lanes cannot report the same +// condition differently. +function toBackfillOutcome(outcome: AttributeOneOutcome): OperationRunOutcome { + switch (outcome) { + case "attributed": + return "done"; + case "already-exists": + // A better record already exists. Nothing to do here is the SAME answer as + // "already current" for the lane's purposes, and reporting it as a failure + // would make an untouched corpus look broken. + case "outranked": + return "already-present"; + case "no-diarization": + return "missing-input"; + case "disabled": + return "disabled"; + // A transcript that is absent or about to be rewritten. Not a failure of + // this lane and not something re-acquiring media fixes — it resolves itself + // when the normalize pass catches up, and the next sweep pass will see it. + case "no-transcript": + return "skipped"; + default: + return "failed"; + } +} + +// --------------------------------------------------------------------------- +// Digest — the fourth entry, and the one this registry was supposed to have had +// from the start. +// +// Digests were built FIRST and never merged in, so they grew a parallel +// implementation of this same idea: their own sweep, their own per-channel +// batch, their own yield probe, their own cost planner, their own queue keys, +// their own settings block with its own pause, and their own snapshot counter. +// backfillSweep.ts is, in its own words, a clone of digestSweep.ts. Registering +// the operation is how that convergence starts. +// +// NOTHING ABOUT DIGEST GENERATION IS REWRITTEN HERE. Every function this entry +// calls is the one the digest controller already calls — resolveDigestTarget for +// the identity, isCuesJsonFresh for the transcript gate, isSectionFresh for +// freshness, digestVideo to do the work. If this entry and digestBatch ever +// disagree about whether a video is digested, that is a bug in this file, not a +// second opinion. +// +// WHAT REGISTERING ACTUALLY BUYS, today: +// +// 1. THE TRANSCRIPT DEPENDENCY BECOMES DECLARED. PLAN.md states in prose that +// "a video cannot be digested until it has a transcript" and nothing +// enforced it — an undigestable video was simply absent from every bucket. +// It now reports `blocked` on `transcription`, so it is counted and named. +// 2. A DECLARED LANE. The digest queue keys stop being one subsystem's private +// constants and become this operation's declared lane, which is what lets a +// scheduler dispatch one job per lane and keep GPU and CPU work overlapping. +// 3. ONE DEFINITION OF "digested". The snapshot, the planner and the batch all +// derive it; this is where they can converge. +// +// IT IS DELIBERATELY NOT RUN BY backfillBatch — see backfillQueueKinds below. +const digest: Operation = { + id: "digest", + label: "Digest", + hint: "Chapters and tags generated from the transcript by a local or metered model. Needs a transcript first.", + group: "digest", + shortLabel: "Digest", + costBasis: "~1 model call per transcript chunk", + // See the `deferred` branch in state() below for the measurement behind this + // wording. It says "run the normalize pass" and NOT "it clears itself", + // because nothing automatic ever will. + deferredHint: + "waiting on a normalized transcript (transcript.cues.json) that nothing produces automatically — run Normalize transcripts on the channel to make them digestable", + tier: "lane", + // The transcript, declared. Nothing in the repo enforced this before. + dependsOn: ["transcription"], + // The LOCAL lane's key is the declared one because it is the default and the + // only one enabled unless remoteEnabled is set. The metered lane runs on + // DIGEST_REMOTE_QUEUE, and the two must never share a key: one shared key + // would idle the network lane while the GPU works, across a multi-week sweep. + // `contendsFor: "gpu"` is what makes this lane — and only this lane — stand + // aside for transcription. + lane: digestLaneFor("local-gpu"), + // The live answer, for the same reason diarization has one: which lane this + // runs on is a CONFIGURATION CHOICE, not a fact about the operation, and + // `lane` above can only carry the default. laneForOperation asks this, so the + // arbiter dispatches a remote digest onto DIGEST_REMOTE_QUEUE instead of the + // local key it declares. + // + // Declaring it does NOT put digest into backfillBatch's idle-only rule, and + // the reason is worth stating because that rule keys off laneFor's PRESENCE: + // resolveBackfillLaneOperations is filtered through backfillLaneOperations (BACKFILL_QUEUE + // only), so digest can never be among the `kinds` that guard inspects. See + // controller/backfillBatch.ts, where the same fact is written from the other + // side. + laneFor: (settings) => + digestLaneFor(settings.digest.remoteEnabled ? "remote-api" : "local-gpu"), + // Digests are gated by their own sweep/pause switches rather than a master + // "enabled" flag, so the feature is on whenever an app is configured. The + // pause is honoured at DISPATCH (digestBatch's limit()), not here: a paused + // lane must still report how much work is outstanding. + enabled: () => true, + async resolveTarget({ paths, channelSlug }) { + // The existing resolver, verbatim. This is the async, channel-scoped case + // OperationTargetContext exists for: a digest's identity includes the hash of + // the channel's context note, which is a file read. + const { resolveDigestTarget } = await import("../controller/digestTarget"); + const resolved = await resolveDigestTarget({ paths, channelSlug }); + return { target: resolved.target, sections: resolved.sections }; + }, + async state({ videoDir, files, target }) { + const { target: freshness, sections } = target as DigestTarget; + // Untranscribable is not-applicable, exactly as the attribution kinds treat + // it: nothing will ever produce a transcript for it, so it is not blocked, + // it is out of scope. + if (files.isUntranscribable) return "not-applicable"; + // BLOCKED, not not-applicable and not missing-input. A video with no + // transcript is waiting on the transcription operation declared in + // dependsOn above — a thing this system produces. Before this it was simply + // invisible: absent from every digest bucket, so a channel of untranscribed + // videos read as fully digested. + if (!isVideoTranscribed(files)) return "blocked"; + const { fresh: cuesFresh } = await isCuesJsonFresh(videoDir); + if (!cuesFresh) { + // NO CURRENT NORMALIZED TRANSCRIPT. Digesting now would either fail for + // want of one or describe superseded text and then look fresh forever, so + // the video is held back — `deferred` is the classification with the + // matching meaning: not attempted, not broken, not counted as reachable + // work. digestVideo reports the same condition as `skipped` rather than a + // failure, for the same reason. + // + // THIS DOES NOT RESOLVE ITSELF, and an earlier version of this comment + // said it did. Measured over the whole corpus (79,219 video dirs): + // + // 1,942 have NO cues.json at all ← reason "missing" + // 47 have one that is superseded ← reason "stale" + // + // so the superseded case this branch was written for is 2.4% of what it + // actually catches. The missing case is permanent: transcribeOne is the + // ONLY automatic caller of normalizeTranscript, and a channel with + // `handling: "youtube"` fetches subtitles with --skip-download and so + // never runs it — 1,683 of the 1,942 are piratesoftware alone. It went + // unnoticed because buildIndex treats cues.json as a CACHE and silently + // re-parses the raw VTT when it is absent, so the published site is + // correct and only this lane, which has no such fallback, can see it. + // + // The fix is the normalize pass, run deliberately: normalizeChannel- + // Transcripts (controller/normalizeAll.ts), wired to a button on the + // digest stage card next to this count. Both reasons are fixed by it, + // which is why they share one classification — see CuesFreshReason. + return "deferred"; + } + const record = await loadDigest(videoDir); + // A digest SHARED from a duplicate cluster's canonical member counts as + // done. The canonical member's own freshness drives regeneration and the + // share is re-applied from it (isSharedFrom's contract) — so re-deriving it + // here would undo ~11% of the sweep's saving. Same rule as the snapshot's + // noDigest bucket, deliberately, because two definitions of "digested" is + // the exact failure this entry exists to stop. + if (record?.derivedFrom != null) return "present"; + // EVERY configured section must be fresh to count as done, matching + // countMissingDigests and the batch. What is new is that "not all of them" + // is no longer one answer. + // + // WHY `partial` IS NOT JUST A NICER WORD FOR `stale`. digestVideo does not + // regenerate a video, it regenerates SECTIONS: its `stale` list at + // digestVideo.ts:158 is `sections.filter(not fresh)`, and only those are + // generated. So a video with fresh chapters and no tags is a fraction of the + // cost of one with neither, and folding them together prices the work wrong + // in the direction that matters — the corpus is ~77,000 videos and the + // recorded surcharge for adding tags to an existing chapters pass is ~44% of + // a full pass, against 100% for a genuine re-generation. + // + // The case is not hypothetical: `sections` is a setting, and the pre-sweep + // decision is to turn tags on. The moment that happens every already-digested + // video in the corpus becomes part-done at once, and without this split it + // would read as `stale` — indistinguishable, on a stage card, from a + // PROMPT_VERSION bump that really did invalidate everything. + // + // The empty-sections case still reads `present` (0 of 0 fresh), exactly as + // the `every()` this replaces did, so a caller passing no sections is + // unchanged. + const fresh = sections.filter((section) => + isSectionFresh(record, section, freshness), + ).length; + if (fresh === sections.length) return "present"; + // No record at all cannot be part-done, and it is the one branch that must + // stay `missing`: it is what separates "never digested" from "digested and + // superseded" everywhere downstream. + if (!record) return "missing"; + return fresh > 0 ? "partial" : "stale"; + }, + async run(opts) { + // Lazy, at the point of running something: digestVideo drags in the prompt + // module, the markdown renderer and the transcript normalizer, and this + // module is imported by channelSnapshot on the editor's hot path. + const { digestVideo } = await import("../controller/digestVideo"); + const { sections } = opts.target as DigestTarget; + // OperationRunOptions already carries channelSlug and videoId, which is + // exactly what digestVideo takes — no reshaping of the controller. + const outcome = await digestVideo({ + paths: opts.paths, + channelSlug: opts.channelSlug, + videoId: opts.videoId, + sections, + // Unit-executor injection. On the primary both are unset and digestVideo + // resolves exactly as before; on an executor the injected config (baseUrl + // stripped — the executor localises its own endpoint) and the injected + // context are what keep the identity the primary's. + ...(opts.appConfig ? { config: opts.appConfig } : {}), + ...(opts.digestContext ? { context: opts.digestContext } : {}), + force: opts.force, + onLog: opts.onLog, + signal: opts.signal, + }); + // The outcome union already lines up almost exactly. + if (outcome.status === "wrote") return "done"; + if (outcome.status === "fresh") return "already-present"; + return "skipped"; + }, + inputs: (files) => transcriptUnitInputs(files, [DIGEST_FILENAME]), + outputs: [DIGEST_FILENAME], + applyResult: applyDigestResult, +}; + +type DigestTarget = { + target: DigestFreshnessTarget; + sections: DigestSectionKind[]; +}; + +// The digest operation has TWO lanes, and which one a run uses follows from the +// engine it is configured with rather than from a separate setting. This is the +// declaration; digestBatch consults it instead of re-testing the app id, so +// "which lane must stand aside for transcription" is stated once. +// +// The queue keys must never be shared: registry.ts runs each key at concurrency +// 1, so one key would idle the network lane while the GPU works — across a +// sweep measured in weeks. +export function digestLaneFor(appLane: DigestLane): Lane { + return appLane === "local-gpu" + ? // Competes with the transcription engine for the same VRAM. Yields. + { queueKey: DIGEST_LOCAL_QUEUE, contendsFor: "gpu" } + : // Metered and network-bound: it competes for nothing local, so yielding + // would park a lane that costs nothing to keep running. + { queueKey: DIGEST_REMOTE_QUEUE, contendsFor: "network" }; +} + +// Which resource a diarization run competes for, which follows from the engine +// it is configured with rather than from a separate setting — the same shape as +// digestLaneFor, and stated here so nothing has to re-test the engine id. +// +// The queue key does NOT change with the engine. Diarization serializes against +// itself either way, and giving the GPU variant its own key would only let two +// diarizations run at once — which is precisely what must not happen when each +// holds ~4.4 GB of an 8 GB card. +export function diarizationLaneFor(diarization: { + engine: DiarizationEngineId; + backend: DiarizationBackend; +}): Lane { + return diarization.engine === SORTFORMER_DIARIZATION_ENGINE && + diarization.backend === "vulkan" + ? // Competes with the transcription engine for the same VRAM. Yields. + { queueKey: BACKFILL_QUEUE, contendsFor: "gpu" } + : // sherpa-onnx is ONNX/CPU, and sortformer on the CPU backend is likewise + // only after cores. Contends for CPU, whose share the backfill lane's own + // weight already governs. + { queueKey: BACKFILL_QUEUE, contendsFor: "cpu" }; +} + +// Whether a lane must stand aside while transcription is working. One rule, so +// the digest lane and any future GPU operation cannot answer it differently. +export function laneYieldsToTranscription(lane: Lane): boolean { + return lane.contendsFor === "gpu"; +} + +// --------------------------------------------------------------------------- +// The operations this system knows about but does NOT dispatch. +// +// Registering them costs nothing and gets the dependency graph right: without a +// `transcription` entry, digest.dependsOn = ["transcription"] names nothing and +// no surface can say what a blocked video is waiting for. +// +// They stay externally dispatched DELIBERATELY, and the reasons are concrete +// rather than a lack of time: +// +// - A SELF-REFERENTIAL YIELD. transcriptionActivity() reports busy exactly +// when transcription is running, so a transcription lane that yielded to it +// would yield to itself. +// - NO WORKER-LEASE SURFACE. OperationRunOptions has no notion of leasing a +// worker, of a tier, of draining, or of a partial stop — all of which the +// worker pool provides and transcription requires. +// - A SCALAR AGAINST A POOL. backfillLimit() returns one number; the worker +// pool is per-worker with individual enable flags and priorities. +// - PER-CHANNEL VS CROSS-CHANNEL SCOPE. backfillBatch is one job per channel; +// autoRunner arbitrates across every channel at once, which is the whole +// point of its policy tree. +export type ExternalOperation = { + id: string; + label: string; + hint: string; + group: OperationGroup; + shortLabel: string; + costBasis: string; + lane: Lane; + dependsOn?: readonly string[]; + dispatch: "external"; + // The auto-queue runner that dispatches this, when one does. `dispatch` alone + // cannot answer it: download and transcription are `external` WITH a runner, + // and a future `transcode` would be `external` with none. A console that + // guessed from the id would hand transcode the transcription runner's + // controls — a live Start button over the wrong lane. + runner?: AutoQueueKind; +}; + +export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [ + { + id: "download", + label: "Download", + hint: "Fetching the media. Dispatched by the auto-download runner and the per-channel pipeline actions.", + group: "media", + shortLabel: "Download", + costBasis: "one fetch per video, over the network", + // Really one queue per platform (downloadQueueKey), not a single key. Named + // here as the shape rather than the exact key, because the catalog's job is + // the dependency graph, not dispatch. + lane: { queueKey: "download:<platform>", contendsFor: "network" }, + dispatch: "external", + runner: "download", + }, + { + id: "transcription", + label: "Transcription", + hint: "Turning audio into a transcript. Dispatched by the auto-transcribe runner across the worker pool.", + group: "transcript", + shortLabel: "Transcribe", + costBasis: "one pass over the audio per video, on a worker", + lane: { queueKey: TRANSCRIPTION_QUEUE, contendsFor: "gpu" }, + dependsOn: ["download"], + dispatch: "external", + runner: "transcription", + }, +]; + +// Every media-derived operation, dispatched here or not. The catalog — what a +// dependency id resolves against, and what a future scheduler enumerates. +export type OperationDescriptor = { + id: string; + label: string; + hint: string; + group: OperationGroup; + shortLabel: string; + costBasis: string; + lane: Lane; + dependsOn?: readonly string[]; + dispatch: OperationDispatch; + // See ExternalOperation.runner. Absent for every backfill kind: those are + // dispatched by the sweep and the arbiter, not by a runner. + runner?: AutoQueueKind; +}; + +export function operationCatalog(): OperationDescriptor[] { + return [ + ...EXTERNAL_OPERATIONS, + ...OPERATIONS.map((k) => ({ + id: k.id, + label: k.label, + hint: k.hint, + group: k.group, + shortLabel: k.shortLabel, + costBasis: k.costBasis, + lane: k.lane, + dependsOn: k.dependsOn, + dispatch: "backfill" as const, + })), + ]; +} + +// The label for a dependency id, from anywhere in the catalog. Returns the id +// itself for something unknown rather than throwing — a dangling dependency is +// already tolerated everywhere else here. +export function operationLabel(id: string): string { + return operationCatalog().find((o) => o.id === id)?.label ?? id; +} + +// The group an operation belongs to, or null for an id the catalog does not +// know. Null rather than a fallback group: a caller grouping by this must be +// able to tell "unknown" from "media", and silently filing a dangling id under +// the first group would put it on the wrong station. +export function operationGroup(id: string): OperationGroup | null { + return operationCatalog().find((o) => o.id === id)?.group ?? null; +} + +// What one video of an operation costs — its cost basis — in words. Empty string for an unknown +// id, so a surface can print it unconditionally without a placeholder. +export function operationShortLabel(id: string): string { + return operationCatalog().find((o) => o.id === id)?.shortLabel ?? id; +} + +export function operationCostBasis(id: string): string { + return operationCatalog().find((o) => o.id === id)?.costBasis ?? ""; +} + +// The label for a SET of operations — a lane card, a transit-line station, a +// column group. One group means that group's name; a mix means "Derived data". +// +// DERIVED, never hardcoded, which is the point: a lane that gains a kind from a +// different group degrades to the honest generic name instead of continuing to +// advertise a label that now describes two thirds of what it holds. +// The same set, named as a THING AN OPERATOR RUNS rather than as a stage on a +// line — "Run speaker work", "N videos are waiting on digests". +// +// Two labels for one group is not duplication: "Speakers" is a station on a +// transit line and has to be a noun at eyebrow width; "speaker work" is the +// object of a verb and has to survive being lower-cased into a sentence. The +// alternative — deriving one from the other — produces "Run speakers work", +// which is why this is declared. +export function groupActionLabel(group: OperationGroup): string { + switch (group) { + case "media": + return "downloads"; + case "transcript": + return "transcripts"; + case "digest": + return "digests"; + case "speakers": + return "speaker work"; + } +} + +export function operationsActionLabel(ids: ReadonlyArray<string>): string { + const groups = new Set<OperationGroup>(); + for (const id of ids) { + const group = operationGroup(id); + if (group) groups.add(group); + } + if (groups.size !== 1) return "derived data"; + return groupActionLabel([...groups][0]); +} + +export function operationsGroupLabel(ids: ReadonlyArray<string>): string { + const groups = new Set<OperationGroup>(); + for (const id of ids) { + const group = operationGroup(id); + if (group) groups.add(group); + } + if (groups.size !== 1) return "Derived data"; + return groupLabel([...groups][0]); +} + +// The digest operation's id, named once. Surfaces that read one specific +// operation off a snapshot (the digest stage card, the dashboard's coverage +// instrument) need this string, and a typo in it fails the way a missing +// snapshot entry does — silently, as "nothing to do". +export const DIGEST_OPERATION_ID = "digest"; + +// The diarization operation's id, named once for the same reason. The cleanup +// accounting needs it to ask a question no other surface asks: whether the lane +// that would release a held video is even running (allOperations drops a kind +// whose enabled() is false, so an ABSENT entry is the answer, not a zero). +export const DIARIZATION_OPERATION_ID = "diarization"; + +// One entry per backfill known to the system. +export const OPERATIONS: readonly Operation[] = [ + diarization, + attributionDiarized, + // Text-only LAST, deliberately. resolveBackfillLaneOperations preserves this order and + // backfillBatch walks the kinds in it, so on a video that has diarization the + // cheap, better lane gets there first and the text lane then finds a record it + // must not overwrite — one wasted classification instead of ~30 model calls. + attributionText, + // Digest is in the table, but on its own lane — so it is counted and + // classified by everything that reads this registry, and dispatched by none + // of it. See backfillQueueKinds. + digest, +]; + +export const OPERATION_BY_ID: Record<string, Operation> = + Object.fromEntries(OPERATIONS.map((k) => [k.id, k])); + +export function getOperation(id: string): Operation | undefined { + return OPERATION_BY_ID[id]; +} + +// Every registered kind whose feature is switched on, whatever lane it runs on. +// The catalog view: what EXISTS and is live, for a scheduler or a dependency +// lookup. Callers that mean "the backfill lane" want backfillLaneOperations below. +export function allOperations(settings: SiteSettings): Operation[] { + return OPERATIONS.filter((k) => k.tier === "lane" && k.enabled(settings)); +} + +// THE BACKFILL LANE's kinds: enabled, `lane` tier, and running on the shared +// BACKFILL_QUEUE. Three filters, and the third is new with the digest entry. +// +// The queue filter is a SAFETY RAIL, not a tidy-up. Everything downstream of +// this function — backfillBatch's dispatch, the channel Backfill card, the +// dashboard's backfill instrument, /actionable's backfill rows — treats these +// as "one lane, one job, one set of counters". Digest satisfies none of that: +// +// - DISPATCH. backfillBatch runs its kinds in one job on one queue under +// backfillLimit(). Handing it digest would SERIALIZE the GPU digest lane +// behind CPU diarization, when the entire reason they hold separate queue +// keys is that they currently overlap. +// - GUARDS. digest carries digestsPaused, the yield-to-transcription +// carve-out (with its CPU-worker exemption), spendCapUsd on the metered +// lane, the remoteEnabled fail-fast, shortest-first ordering, +// duplicate-cluster sharing and the engine probe() fail-fast. Those are +// measured decisions in digestBatch's limit(), and not one of them is +// expressible as backfillLimit()'s single scalar. +// - COUNTERS. Digest already has its own instrument, its own stage card and +// its own snapshot bucket. Folding it in here would double-count it against +// surfaces that are live on a 78,000-video corpus mid-sweep. +// +// So digest is CLASSIFIED and CATALOGUED through this registry and DISPATCHED +// through its own controller. Collapsing the two sets of counters and the two +// schedulers is the unified-rule-model work, not this function's job. A kind +// joins this list when its lane rule can be expressed here without losing a +// guard. +export function backfillLaneOperations(settings: SiteSettings): Operation[] { + return allOperations(settings).filter( + (k) => k.lane.queueKey === BACKFILL_QUEUE, + ); +} + +// The READ-SIDE twin of backfillLaneOperations: given a snapshot's per-kind map, +// return only the entries belonging to the shared backfill lane — KEYED, so a +// surface can say WHICH kind a number came from. backfillLaneEntriesOf below is this +// with the ids dropped, for the callers that only sum. +// +// The keyed form is what the corpus-wide backfill card needs. Summed, this lane +// reads "77,952 reachable · 77,134 need media" — both figures correct, and +// together meaningless: 99.5% of the first is attribution-text (one model call +// per transcript CHUNK) and all of the second is diarization (329 runs). Adding +// kinds gives a number in no unit at all, which is the mistake the header +// forbids one level up for `missing` vs `missing-input`. +// +// THIS EXISTS BECAUSE THE SNAPSHOT MAP STOPPED BEING THE LANE. It used to be +// written from backfillLaneOperations, so `Object.values(snapshot.backfill)` and "the +// backfill lane" were the same set by construction, and four surfaces summed it +// generically on that basis — the channel dashboard's backfill instrument, +// /actionable's two backfill functions and the widget's sync payload. The moment +// channelSnapshot writes an entry per CATALOG operation, that identity breaks: +// those four would silently absorb ~75,000 digest videos into a number that has +// only ever meant diarization plus attribution. +// +// FILTERED BY THE DECLARATION, NOT BY ID. `key !== "digest"` would fix today and +// leave the identical trap armed for the next operation registered on a lane of +// its own — which is the whole direction of the unified-operations work. The +// rule is the same one backfillLaneOperations applies on the write side, asked of the +// registry: does this kind run on BACKFILL_QUEUE? +// +// An id the catalog does not know is EXCLUDED. A snapshot is a file on disk that +// may have been written by an older build and may name a kind that has since +// been renamed or removed; there is no lane declaration to check it against, so +// it cannot be asserted to belong to this one. Deliberately not filtered on +// `enabled(settings)` — these are counts already written to disk, and a feature +// switched off after a snapshot was taken does not retroactively unmake the work +// it recorded. +export function backfillLaneOperationEntriesOf<T>( + backfill: Record<string, T> | undefined | null, +): [string, T][] { + if (!backfill) return []; + return Object.entries(backfill).filter( + ([id]) => getOperation(id)?.lane.queueKey === BACKFILL_QUEUE, + ); +} + +// The same set with the ids dropped, for the callers that only ever sum. Defined +// in terms of the above rather than beside it: the filter and every word of the +// rule above it must stay in ONE place, or the next surface that wants per-kind +// detail copies a `key !== "digest"` in and re-arms the trap. +export function backfillLaneEntriesOf<T>( + backfill: Record<string, T> | undefined | null, +): T[] { + return backfillLaneOperationEntriesOf(backfill).map(([, entry]) => entry); +} + +// Resolve a caller-supplied list of kind ids against the registry. An empty or +// absent list means "every enabled lane kind" — the sweep's scope default. +// Unknown ids are dropped rather than throwing: a settings file may name a kind +// from a newer build, and a stale scope must not wedge the lane. +export function resolveBackfillLaneOperations( + settings: SiteSettings, + ids: readonly string[] | undefined, +): Operation[] { + const lane = backfillLaneOperations(settings); + const selected = + !ids || ids.length === 0 + ? lane + : lane.filter((k) => new Set(ids).has(k.id)); + return orderByDependencies(selected); +} + +// Order kinds so a prerequisite is attempted before anything that declares it. +// +// WHY THIS IS WORTH DOING AT ALL. backfillBatch walks the kinds in the order it +// is given, all the way through the video list, before starting the next kind. +// So with `attribution-diarized` ahead of `diarization`, a video diarized +// during a pass becomes eligible for attribution only on whatever LATER pass +// happens to find the sidecar on disk. Ordering by the declaration collapses +// that into one pass, and costs a topological sort over three entries. +// +// STABLE, and that is load-bearing rather than tidiness. OPERATIONS puts +// attribution-text LAST on purpose (see the comment there): on a video that has +// diarization, the better lane must get there first so the text lane finds a +// record it must not overwrite — one wasted classification instead of ~30 model +// calls. Kahn's algorithm with a queue seeded and drained in declaration order +// preserves every ordering the declarations do not contradict, so that decision +// survives. +export function orderByDependencies(kinds: Operation[]): Operation[] { + const byId = new Map(kinds.map((k) => [k.id, k])); + // Only dependencies that are actually IN this selection constrain anything. A + // kind that names a disabled or unselected prerequisite is not held back — + // it will report `blocked` per video, which is the honest answer, rather than + // being silently dropped from the run. + const remaining = new Map( + kinds.map((k) => [ + k.id, + (k.dependsOn ?? []).filter((d) => byId.has(d) && d !== k.id).length, + ]), + ); + const dependants = new Map<string, string[]>(); + for (const k of kinds) { + for (const d of k.dependsOn ?? []) { + if (!byId.has(d) || d === k.id) continue; + const list = dependants.get(d); + if (list) list.push(k.id); + else dependants.set(d, [k.id]); + } + } + const out: Operation[] = []; + const emitted = new Set<string>(); + // Repeatedly take the FIRST still-unemitted kind in declaration order whose + // prerequisites are all out. Quadratic in the number of kinds, which is three. + for (;;) { + const next = kinds.find( + (k) => !emitted.has(k.id) && (remaining.get(k.id) ?? 0) === 0, + ); + if (!next) break; + emitted.add(next.id); + out.push(next); + for (const id of dependants.get(next.id) ?? []) { + remaining.set(id, (remaining.get(id) ?? 1) - 1); + } + } + // A CYCLE leaves entries unemitted. Append them in declaration order rather + // than throwing or dropping them: a mis-declared dependency should degrade to + // the old behaviour (run in table order), never wedge the lane or silently + // stop a backfill from running at all. + for (const k of kinds) if (!emitted.has(k.id)) out.push(k); + return out; +} + +// Per-kind counts, the shape every indicator reads. `missing` and `missingInput` +// are never summed — see the header. +export type OperationCounts = { + missing: number; + stale: number; + missingInput: number; + // Work the kind is refusing to attempt under the current configuration (the + // diarization duration cap). A THIRD number, alongside the other two that are + // never summed. Snapshots written before this field existed do not carry it, + // so every read site needs `?? 0` — `.toLocaleString()` on undefined throws. + deferred: number; + // Waiting on a prerequisite kind's output. A FOURTH number, and the same rule + // applies: never summed with the others, and `?? 0` at every read site, + // because every snapshot currently on disk predates it. + // + // This number should FALL on its own as the prerequisite lane works, which is + // the whole difference from missingInput — that one only falls if an operator + // turns re-download on. + blocked: number; + // Reachable work that is PART DONE. Unlike deferred and blocked, this one IS + // summed into reachableOperationWork — it is work the lane can do today. It is + // split out of `stale` because the two cost different amounts and want + // different decisions: see the digest entry's state(). + // + // `?? 0` at every read site, like deferred and blocked before it. Every + // snapshot currently on disk predates this field. + partial: number; +}; + +export function emptyOperationCounts(): OperationCounts { + return { + missing: 0, + stale: 0, + partial: 0, + missingInput: 0, + deferred: 0, + blocked: 0, + }; +} + +// Fold one classification into a counts record. Central so no surface invents +// its own accounting: `present` and `not-applicable` add to nothing, which is +// what makes these counts a WORK LIST rather than a coverage measure. +export function addOperationState( + counts: OperationCounts, + state: OperationClassification, +): void { + if (state === "missing") counts.missing++; + else if (state === "stale") counts.stale++; + else if (state === "partial") counts.partial++; + else if (state === "missing-input") counts.missingInput++; + else if (state === "deferred") counts.deferred++; + else if (state === "blocked") counts.blocked++; +} + +// What the lane can act on WITHOUT re-acquiring media. The number every "how +// much is left?" surface should lead with. +// +// DELIBERATELY UNCHANGED by the addition of `deferred`, and unchanged again by +// `blocked`. This function is the guard: adding a state to the union raises no +// TypeScript error here (the exhaustiveness check lives on the DISPATCH +// decision, in backfillBatch's candidateAction, which is the branch that can do +// harm), so the only thing keeping capped and blocked videos out of the work +// total is that they are not added here. If a future state belongs in the +// total, it goes in on purpose. +// +// A blocked video is emphatically not reachable work: there is nothing this +// lane can do about it this pass. Counting it would make a corpus with one +// diarization and 73,000 waiting attributions report 73,000 jobs ready to run. +// +// `partial` IS in the total, and that is the on-purpose case the paragraph +// above reserves. A part-done video is work the lane can pick up right now and +// the run will write to it; leaving it out would make a corpus mid-tags-backfill +// report less work than it has. Splitting it from `stale` is about what the +// work COSTS, not about whether it is reachable — so the sum is unchanged from +// what it would have been before the split, which is the property that keeps +// this a refinement rather than a behaviour change. +// +// `?? 0` because every snapshot on disk predates the field, and undefined would +// poison the sum to NaN rather than merely under-report. +export function reachableOperationWork(counts: OperationCounts): number { + return counts.missing + counts.stale + (counts.partial ?? 0); +} + +// What a channel snapshot stores per kind: the three counts, plus the ids of the +// REACHABLE work only. +// +// The asymmetry is deliberate. A stage card has to list what it would act on, so +// those ids have to be somewhere the render path can read without walking the +// corpus (there is a guard test forbidding exactly that). But `missingInput` is +// ~76,000 videos corpus-wide, and writing that list into all 66 snapshots would +// put tens of megabytes of ids on disk to say a number we already have. So: ids +// for the actionable half, a count for the other. +export type OperationSnapshotEntry = OperationCounts & { + // Exactly the reachable set — missing + stale + partial — sorted. Never + // includes missing-input, deferred or blocked. Kept equal to + // reachableOperationWork(entry) by a test, because a policy leaf hands this + // list out as work while the cards render the count. + ids: string[]; + // How many videos this operation has an OPINION about: everything it did not + // classify not-applicable. The denominator, and deliberately not part of + // OperationCounts — those are a work list, and mixing a coverage measure into + // them is what would let a surface add "done" to "to do". + // + // It is stored rather than derived because `present` is the one classification + // addOperationState throws away, so nothing downstream can reconstruct the + // total from the counts alone. With it, present = eligible - (every work + // count), which is what presentOperationWork below computes. + // + // Optional: every snapshot written before this field lacks it, and a reader + // that cannot tell how many videos were considered must say so rather than + // divide by a zero it invented. + eligible?: number; +}; + +// How many videos this operation is DONE with, derived from the stored +// denominator minus every work state. Returns null when the snapshot predates +// `eligible`, because the honest answer there is "unknown" — a 0 would render as +// "nothing digested" on a fully digested channel. +export function presentOperationWork( + entry: OperationSnapshotEntry, +): number | null { + if (entry.eligible == null) return null; + return Math.max( + 0, + entry.eligible - + (entry.missing + + entry.stale + + (entry.partial ?? 0) + + entry.missingInput + + (entry.deferred ?? 0) + + (entry.blocked ?? 0)), + ); +} diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -242,7 +242,7 @@ export type SiteSettings = { // corpus-wide text-only pass is worth 25-55 GPU-days at all. export type AttributionSettings = { // Master switch. Off means the backfill registry reports no attribution work - // at all — the feature gate every BackfillKind has. + // at all — the feature gate every Operation has. enabled: boolean; // Which digest app runs the naming. Attribution IS a digest-app workload — // constrained JSON decoding over transcript text — so it reuses that registry diff --git a/common/lib/sweepPlan.ts b/common/lib/sweepPlan.ts @@ -31,7 +31,7 @@ import type { AutoQueueOrder } from "../jobs/autoQueuePolicy"; // One operation's outstanding work on one channel. // // `reachable` and `missingInput` stay apart here as they do everywhere else — -// see lib/backfillKinds.ts's header. On the measured corpus they are four orders +// see lib/operations.ts's header. On the measured corpus they are four orders // of magnitude apart on diarization, and one "remaining" number would say the // same thing about a lane that is finished and a lane that cannot start. export type SweepKindCounts = { @@ -89,7 +89,7 @@ export type SweepPlanEntry = { // Fold the per-kind counts down to one row per channel, ordered. // // `kindIds` empty means EVERY kind on the payload — the same rule -// resolveBackfillKinds applies to an absent scope, and the same rule +// resolveBackfillLaneOperations applies to an absent scope, and the same rule // startBackfillSweep persists. It is deliberately not "no kinds": an unscoped // sweep is the corpus-wide one, and rendering it as an empty plan would say the // opposite of what arming it would do. diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -34,7 +34,7 @@ export type VideoFiles = { // the expensive one: the channel snapshot reads this per video across 77,000 // of them, so a per-video re-listing is the difference between an indicator // that is free and one that doubles the report's I/O. See - // lib/backfillKinds.ts, which classifies missing vs missing-input from it. + // lib/operations.ts, which classifies missing vs missing-input from it. entries: string[]; }; diff --git a/common/lib/workers.ts b/common/lib/workers.ts @@ -99,9 +99,9 @@ export type Worker = { llm?: LlmWorkerConfig; }; -// The contended-resource half of the tag vocabulary — BackfillLane.contendsFor's +// The contended-resource half of the tag vocabulary — Lane.contendsFor's // three values, restated here because this module must stay client-safe and the -// lane type lives in backfillKinds.ts, whose import graph reaches controllers. +// lane type lives in operations.ts, whose import graph reaches controllers. // A drift between the two lists is caught by a test, not by the type system. export const WORKER_RESOURCE_TAGS = ["gpu", "cpu", "network"] as const; diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts @@ -1,9 +1,9 @@ import type { Paths } from "yt-dlp-transcript-common/lib/paths"; import { cache } from "react"; import { - laneEntriesOf, - reachableBackfillWork, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + backfillLaneEntriesOf, + reachableOperationWork, +} from "yt-dlp-transcript-common/lib/operations"; import type { ChannelBrief } from "yt-dlp-transcript-common/controller/channels"; import { getChannelBriefs, @@ -158,15 +158,15 @@ export function actionableDigestWarningsCount(row: ActionableRow): number { // the reachable count corpus-wide, so filtering on it would put every channel in // the list forever. That is not a hypothetical: it is the documented reason // /api/widget/actionable refuses to filter on `noDigest`. -// laneEntriesOf, not Object.values: the snapshot map is every catalog operation +// backfillLaneEntriesOf, not Object.values: the snapshot map is every catalog operation // now, and digest is one of them. These two functions decide whether a channel // appears in the BACKFILL section at all, so folding a ~75,000-video operation // that runs on another queue into them would put every channel in the list // forever — the same trap /api/widget/actionable documents for `noDigest`, hit // from the other direction. export function actionableBackfillCount(row: ActionableRow): number { - return laneEntriesOf(row.snapshot?.backfill).reduce( - (n, e) => n + reachableBackfillWork(e), + return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( + (n, e) => n + reachableOperationWork(e), 0, ); } @@ -174,7 +174,7 @@ export function actionableBackfillCount(row: ActionableRow): number { export function actionableBackfillMissingInputCount( row: ActionableRow, ): number { - return laneEntriesOf(row.snapshot?.backfill).reduce( + return backfillLaneEntriesOf(row.snapshot?.backfill).reduce( (n, e) => n + e.missingInput, 0, ); diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx @@ -4,9 +4,9 @@ import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { formatBytes } from "yt-dlp-transcript-common/lib/format"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { - laneBackfillKinds, + backfillLaneOperations, operationsActionLabel, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; import { actionableCleanExtraFormatsBytes, actionableCleanExtraFormatsCount, @@ -72,7 +72,7 @@ export default async function ActionablePage() { // enabled. "Backfill" is its queue key; on this install it stands for three // operations, and no button anyone presses should be named after a queue. const laneAction = `Run ${operationsActionLabel( - laneBackfillKinds(getSettings()).map((k) => k.id), + backfillLaneOperations(getSettings()).map((k) => k.id), )}`; const nothingPending = summary.undownloaded.length === 0 && diff --git a/editor/app/api/widget/sync/route.ts b/editor/app/api/widget/sync/route.ts @@ -5,14 +5,14 @@ import { digestWorkOf } from "yt-dlp-transcript-common/controller/channelSnapsho import { getChannelBriefs } from "../../../lib/requestCache"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { - laneBackfillKinds, - laneKindEntriesOf, + backfillLaneOperations, + backfillLaneOperationEntriesOf, operationCostBasis, operationLabel, operationsGroupLabel, - presentBackfillWork, - reachableBackfillWork, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + presentOperationWork, + reachableOperationWork, +} from "yt-dlp-transcript-common/lib/operations"; import { buildScheduleView } from "yt-dlp-transcript-common/jobs/syncScheduler"; import { readSchedulerState } from "yt-dlp-transcript-common/jobs/syncSchedulerState"; @@ -86,7 +86,7 @@ export type WidgetSyncPayload = { // protects is that this endpoint is POLLED: a per-channel breakdown would grow // with the corpus (67 channels and climbing) and put a corpus walk back on a // 15-second timer. A per-KIND breakdown cannot, because adding a kind means - // adding an entry to backfillKinds.ts. That distinction still holds, and it is + // adding an entry to operations.ts. That distinction still holds, and it is // the line to keep: bounded by the code, never by the data. backfill: { reachable: number; // missing + stale: what the lane can do now @@ -100,7 +100,7 @@ export type WidgetSyncPayload = { anyKind: boolean; // The lane's name in the operator's terms, derived on the server from the // GROUP its enabled kinds declare — "Speakers" today. The client must not - // resolve this itself: backfillKinds.ts reaches the filesystem. + // resolve this itself: operations.ts reaches the filesystem. // // "Backfill" is a queue key. Nobody arms, pauses or runs "a backfill"; the // word survives only on the two controls that genuinely act on the shared @@ -124,7 +124,7 @@ export type WidgetSyncPayload = { // across channels, a third of the snapshots on disk predate `eligible`, and // a partial sum is a denominator smaller than its own numerator. One // channel that cannot report voids the kind's whole figure — see - // presentBackfillWork, which returns null rather than 0 "because the honest + // presentOperationWork, which returns null rather than 0 "because the honest // answer there is unknown". kinds: { id: string; @@ -143,7 +143,7 @@ export type WidgetSyncPayload = { eligible: number | null; // How many it is DONE with. null = unknown. Never derived by subtracting // the work counts from `eligible` at a call site — that is exactly what - // presentBackfillWork does, once, with the right guard. + // presentOperationWork does, once, with the right guard. present: number | null; }[]; }; @@ -237,8 +237,8 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> { // catalog operation, digest included, and the widget's backfill strip has // only ever meant diarization plus attribution — the digest coverage figure // it shows beside this one is computed separately, from digestCountOf above. - for (const [id, entry] of laneKindEntriesOf(c.snapshot?.backfill)) { - backfillReachable += reachableBackfillWork(entry); + for (const [id, entry] of backfillLaneOperationEntriesOf(c.snapshot?.backfill)) { + backfillReachable += reachableOperationWork(entry); backfillNeedsMedia += entry.missingInput; const acc = backfillByKind.get(id) ?? { reachable: 0, @@ -248,7 +248,7 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> { eligible: 0 as number | null, present: 0 as number | null, }; - acc.reachable += reachableBackfillWork(entry); + acc.reachable += reachableOperationWork(entry); acc.needsMedia += entry.missingInput; // `?? 0` at every read site: every snapshot written before these fields // existed lacks them, and undefined poisons the sum to NaN. @@ -261,7 +261,7 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> { acc.eligible == null || entry.eligible == null ? null : acc.eligible + entry.eligible; - const done = presentBackfillWork(entry); + const done = presentOperationWork(entry); acc.present = acc.present == null || done == null ? null : acc.present + done; backfillByKind.set(id, acc); @@ -294,9 +294,9 @@ export async function buildWidgetSyncPayload(): Promise<WidgetSyncPayload> { videos, enabled: settings.backfill.enabled, sweeping: settings.backfill.sweepEnabled, - anyKind: laneBackfillKinds(settings).length > 0, + anyKind: backfillLaneOperations(settings).length > 0, groupLabel: operationsGroupLabel( - laneBackfillKinds(settings).map((k) => k.id), + backfillLaneOperations(settings).map((k) => k.id), ), kinds: [...backfillByKind] .map(([id, counts]) => ({ diff --git a/editor/app/api/worker/unit/route.ts b/editor/app/api/worker/unit/route.ts @@ -1,6 +1,6 @@ import { NextResponse } from "next/server"; import { authorizeWorkerRequest } from "yt-dlp-transcript-common/lib/workerToken"; -import { getBackfillKind } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { getOperation } from "yt-dlp-transcript-common/lib/operations"; import { startWorkerUnit, type StartWorkerUnitInput, @@ -36,7 +36,7 @@ export async function POST(request: Request) { { status: 400 }, ); } - if (!getBackfillKind(body.op)) { + if (!getOperation(body.op)) { return NextResponse.json( { error: `"${body.op}" is not a backfill kind this executor can run` }, { status: 400 }, diff --git a/editor/app/channels/[slug]/components/flow/SidingList.tsx b/editor/app/channels/[slug]/components/flow/SidingList.tsx @@ -32,7 +32,7 @@ const VISIBLE_SIDINGS = 2; // The hover card carries the sentence nobody remembers: what "deferred" is // waiting for, what a "blocked" video is blocked ON. Those come from the // registry (kind.deferredHint, dependsOn → operationLabel), resolved on the -// server, because BACKFILL_KIND_BY_ID reads the filesystem. +// server, because OPERATION_BY_ID reads the filesystem. export function SidingList({ slug, diff --git a/editor/app/channels/[slug]/components/stages/BackfillStage.tsx b/editor/app/channels/[slug]/components/stages/BackfillStage.tsx @@ -32,7 +32,7 @@ export type BackfillKindView = { // missing + stale: what the lane can do right now. reachableIds: string[]; // Videos whose input is gone. A COUNT only — the id list is corpus-sized and - // deliberately not stored in the snapshot (see BackfillSnapshotEntry). + // deliberately not stored in the snapshot (see OperationSnapshotEntry). missingInput: number; // Videos this kind refuses to attempt under the current configuration. A // THIRD number, never added to the other two: summing it would let a capped diff --git a/editor/app/channels/[slug]/components/stages/DigestStage.tsx b/editor/app/channels/[slug]/components/stages/DigestStage.tsx @@ -41,7 +41,7 @@ type Props = { // NOT self-clearing, which is why this card carries a button. Measured over // the corpus: 1,942 have no cues.json and 47 have a stale one, and nothing // automatic writes the missing ones — see the digest entry in - // common/lib/backfillKinds.ts. + // common/lib/operations.ts. deferred: number; // Part of the reachable count above, not an addition to it: videos that have // SOME configured section at the current identity and not the rest. Broken diff --git a/editor/app/channels/[slug]/lib/channelFlow.test.ts b/editor/app/channels/[slug]/lib/channelFlow.test.ts @@ -3,9 +3,9 @@ import assert from "node:assert/strict"; import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import type { - BackfillKind, - BackfillSnapshotEntry, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + Operation, + OperationSnapshotEntry, +} from "yt-dlp-transcript-common/lib/operations"; import { computeChannelFlow, type FlowStationId } from "./channelFlow"; import { computeStageStatuses, normalizeBuckets } from "./stageStatus"; @@ -26,22 +26,22 @@ function snapshotOf(patch: Partial<ChannelSnapshot> = {}): ChannelSnapshot { // A registry entry is a big object with three async methods on it; none of them // are reachable from computeChannelFlow, which only ever reads id/label/hints. -function kind(patch: Partial<BackfillKind> & { id: string }): BackfillKind { - return { label: patch.id, hint: "", ...patch } as BackfillKind; +function kind(patch: Partial<Operation> & { id: string }): Operation { + return { label: patch.id, hint: "", ...patch } as Operation; } // A snapshot entry AS WRITTEN TO DISK. The cast is the point of the helper: -// BackfillCounts declares deferred/blocked/partial required, but every snapshot +// OperationCounts declares deferred/blocked/partial required, but every snapshot // currently on disk predates them, which is why every read site carries `?? 0`. // Omitting them here is how these tests exercise the real files. -function entry(patch: Partial<BackfillSnapshotEntry>): BackfillSnapshotEntry { - return { ids: [], missing: 0, stale: 0, missingInput: 0, ...patch } as BackfillSnapshotEntry; +function entry(patch: Partial<OperationSnapshotEntry>): OperationSnapshotEntry { + return { ids: [], missing: 0, stale: 0, missingInput: 0, ...patch } as OperationSnapshotEntry; } function flowOf( snapshot: ChannelSnapshot, opts: { - backfillKinds?: BackfillKind[]; + laneOperations?: Operation[]; playlistCount?: number | null; transcodeApplies?: boolean; failedVideoIds?: string[]; @@ -63,7 +63,7 @@ function flowOf( failedVideoIds, failedTranscodingIds, transcodeApplies: opts.transcodeApplies ?? false, - backfillKinds: opts.backfillKinds ?? [kind({ id: "diarization" })], + laneOperations: opts.laneOperations ?? [kind({ id: "diarization" })], playlistCount: opts.playlistCount ?? null, }); } @@ -116,7 +116,7 @@ test("the lane station does NOT sum its operations — it reads the lead one", ( }, }), { - backfillKinds: [ + laneOperations: [ kind({ id: "diarization" }), kind({ id: "attribution-diarized" }), ], @@ -144,7 +144,7 @@ test("the lane station is named after its operations, not its queue key", () => // its kinds declare, so a lane holding a mix degrades to the generic name // rather than advertising one member's. const speakers = flowOf(snapshotOf(), { - backfillKinds: [ + laneOperations: [ kind({ id: "diarization" }), kind({ id: "attribution-text" }), ], @@ -152,12 +152,12 @@ test("the lane station is named after its operations, not its queue key", () => assert.equal(station(speakers, "speakers").label, "Speakers"); const mixed = flowOf(snapshotOf(), { - backfillKinds: [kind({ id: "diarization" }), kind({ id: "digest" })], + laneOperations: [kind({ id: "diarization" }), kind({ id: "digest" })], }); assert.equal(station(mixed, "speakers").label, "Derived data"); // Nothing enabled: the generic name, and no operations to state. - const off = flowOf(snapshotOf(), { backfillKinds: [] }); + const off = flowOf(snapshotOf(), { laneOperations: [] }); assert.equal(station(off, "speakers").label, "Derived data"); assert.deepEqual(station(off, "speakers").operations, []); }); @@ -168,7 +168,7 @@ test("an unknown `eligible` on the lead operation still renders unknown, not zer // station must say "—" rather than 0%. const flow = flowOf( snapshotOf({ backfill: { diarization: entry({ missing: 3 }) } }), - { backfillKinds: [kind({ id: "diarization" })] }, + { laneOperations: [kind({ id: "diarization" })] }, ); const lane = station(flow, "speakers"); assert.equal(lane.through, null); @@ -197,7 +197,7 @@ test("a lane that is switched off reads neutral, never ok and never amber", () = }, }); - const off = flowOf(snapshot, { backfillKinds: [] }); + const off = flowOf(snapshot, { laneOperations: [] }); assert.equal(station(off, "speakers").tone, "neutral"); // …and the work it recorded is not offered as something to press, because // nothing would run it. @@ -208,7 +208,7 @@ test("a lane that is switched off reads neutral, never ok and never amber", () = // disabled lane were ever offered as the next action again. assert.notEqual(off.next?.stage, "speakers"); - const on = flowOf(snapshot, { backfillKinds: [kind({ id: "diarization" })] }); + const on = flowOf(snapshot, { laneOperations: [kind({ id: "diarization" })] }); assert.equal(on.gaps.find((g) => g.to === "speakers")?.reachable, 7); }); diff --git a/editor/app/channels/[slug]/lib/channelFlow.ts b/editor/app/channels/[slug]/lib/channelFlow.ts @@ -6,15 +6,15 @@ import { } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { digestCountOf } from "yt-dlp-transcript-common/controller/channels"; import { - laneEntriesOf, + backfillLaneEntriesOf, operationCatalog, operationLabel, operationsActionLabel, operationsGroupLabel, - presentBackfillWork, - reachableBackfillWork, - type BackfillKind, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + presentOperationWork, + reachableOperationWork, + type Operation, +} from "yt-dlp-transcript-common/lib/operations"; import type { OperationBand } from "../../../components/pipelines/band"; import { buildChannelBands } from "../../../components/pipelines/buildBands"; import { @@ -156,7 +156,7 @@ export type ComputeChannelFlowInput = { transcodeApplies: boolean; // Enabled lane kinds. EMPTY MEANS THE LANE IS OFF, which is not the same as // finished — see the tone rule at the bottom of this file. - backfillKinds: BackfillKind[]; + laneOperations: Operation[]; // How many videos the channel's `playlist` file names, or null when there is // no playlist file. Read by the caller (countPlaylist) so this stays pure. playlistCount: number | null; @@ -229,7 +229,7 @@ export function computeChannelFlow( failedVideoIds, failedTranscodingIds, transcodeApplies, - backfillKinds, + laneOperations, playlistCount, } = input; @@ -248,15 +248,15 @@ export function computeChannelFlow( // has not been re-reported since the registry landed would read as done. const digestWork = digestWorkOf(snapshot); - // laneEntriesOf, never Object.values: the per-kind map now carries EVERY + // backfillLaneEntriesOf, never Object.values: the per-kind map now carries EVERY // catalog operation including digest (~75k videos on the live corpus), and // digest has its own station one step upstream. - const laneEntries = laneEntriesOf(snapshot.backfill); + const laneEntries = backfillLaneEntriesOf(snapshot.backfill); // An empty kind list means the operator switched the feature off. That is not // an empty work list in the "finished" sense, and the tone rule below says so. - const laneOff = backfillKinds.length === 0; + const laneOff = laneOperations.length === 0; - const laneKindIds = backfillKinds.map((k) => k.id); + const laneKindIds = laneOperations.map((k) => k.id); const operationsById = stationOperations(snapshot, laneKindIds); const opsFor = (...ids: string[]): StationOperation[] => ids @@ -269,7 +269,7 @@ export function computeChannelFlow( const laneReachable = laneOff ? 0 - : laneEntries.reduce((n, e) => n + reachableBackfillWork(e), 0); + : laneEntries.reduce((n, e) => n + reachableOperationWork(e), 0); const laneMissingInput = laneEntries.reduce((n, e) => n + e.missingInput, 0); // `?? 0` is load-bearing, not defensive: snapshots written before these fields // existed have neither, and .toLocaleString() on undefined throws in a render. @@ -467,7 +467,7 @@ export function computeChannelFlow( "deferred", digestWork.deferred, "digest", - deferredHintFor(backfillKinds, "digest"), + deferredHintFor(laneOperations, "digest"), ), ...siding( "digest warnings", @@ -489,13 +489,13 @@ export function computeChannelFlow( "deferred", laneDeferred, "speakers", - deferredHintFor(backfillKinds), + deferredHintFor(laneOperations), ), ...siding( "waiting on an earlier backfill", laneBlocked, "speakers", - dependsOnHint(backfillKinds), + dependsOnHint(laneOperations), ), ], }; @@ -587,7 +587,7 @@ function pickNext( // than hardcoded here. BackfillStage used to say "too long to diarize", which // was true only while diarization was the sole kind that could defer. function deferredHintFor( - kinds: BackfillKind[], + kinds: Operation[], onlyId?: string, ): string | undefined { const hints = kinds @@ -601,7 +601,7 @@ function deferredHintFor( // Which operations' output the lane's kinds are blocked on, by label — so a // blocked count says what it is waiting FOR rather than merely that it is stuck. -function dependsOnHint(kinds: BackfillKind[]): string | undefined { +function dependsOnHint(kinds: Operation[]): string | undefined { const labels = [ ...new Set(kinds.flatMap((k) => (k.dependsOn ?? []).map(operationLabel))), ]; diff --git a/editor/app/channels/[slug]/lib/stageOrder.test.ts b/editor/app/channels/[slug]/lib/stageOrder.test.ts @@ -18,7 +18,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { OPERATION_GROUP_ORDER } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { OPERATION_GROUP_ORDER } from "yt-dlp-transcript-common/lib/operations"; import { GROUP_STAGES, type StageId } from "./stageStatus"; // The expression from page.tsx, verbatim. Duplicated rather than exported and diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -6,11 +6,11 @@ import { } from "yt-dlp-transcript-common/controller/channelSnapshot"; import type { JobRecord } from "yt-dlp-transcript-common/jobs/registry"; import { - laneEntriesOf, + backfillLaneEntriesOf, operationsGroupLabel, - reachableBackfillWork, + reachableOperationWork, type OperationGroup, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; export type SnapshotBuckets = ChannelSnapshot["buckets"]; @@ -458,7 +458,7 @@ export function computeStageStatuses( // for work that cannot be done without an opt-in re-download — precisely the // trap /api/widget/actionable documents for `noDigest`. const backfillRunning = runningByStage.has("speakers"); - // laneEntriesOf, not Object.values. The snapshot's per-kind map carries every + // backfillLaneEntriesOf, not Object.values. The snapshot's per-kind map carries every // operation in the catalog now, including digest — which runs on its own queue // key, has its own stage card directly above, and would otherwise add ~75,000 // videos to this instrument on the measured corpus. The filter is by the kind's @@ -467,10 +467,10 @@ export function computeStageStatuses( // A disabled lane has NO work, whatever the snapshot recorded before it was // switched off — see `backfillEnabled` above. const backfillEntries = backfillEnabled - ? laneEntriesOf(snapshot.backfill) + ? backfillLaneEntriesOf(snapshot.backfill) : []; const backfillPending = backfillEntries.reduce( - (n, e) => n + reachableBackfillWork(e), + (n, e) => n + reachableOperationWork(e), 0, ); const backfillMissingInput = backfillEntries.reduce( diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -68,12 +68,12 @@ import { TranscribeStage } from "./components/stages/TranscribeStage"; import { DigestStage } from "./components/stages/DigestStage"; import { BackfillStage } from "./components/stages/BackfillStage"; import { - getBackfillKind, - laneBackfillKinds, + getOperation, + backfillLaneOperations, operationsActionLabel, operationsGroupLabel, OPERATION_GROUP_ORDER, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; import { ChannelLine } from "./components/flow/ChannelLine"; import { NextAction } from "./components/flow/NextAction"; import { AttentionStrip } from "./components/flow/AttentionStrip"; @@ -243,18 +243,18 @@ export default async function ChannelDetailPage({ config.handling === "transcribe" && !!config.audioFormat; // Enabled lane backfills. A settings read, no I/O. // - // laneBackfillKinds, still: this is the BACKFILL lane's list, and the snapshot + // backfillLaneOperations, still: this is the BACKFILL lane's list, and the snapshot // now carries an entry for every catalog operation including digest, which has // its own station and its own queue key. - const backfillKinds = laneBackfillKinds(settings); + const laneOperations = backfillLaneOperations(settings); const stages = computeStageStatuses({ snapshot, failedVideoIds, failedTranscodingIds, config, runningJobs, - backfillEnabled: backfillKinds.length > 0, - backfillKindIds: backfillKinds.map((k) => k.id), + backfillEnabled: laneOperations.length > 0, + backfillKindIds: laneOperations.map((k) => k.id), }); // THE MIDDLE COMES FROM THE REGISTRY, in OPERATION_GROUP_ORDER. Same array as @@ -304,7 +304,7 @@ export default async function ChannelDetailPage({ failedVideoIds, failedTranscodingIds, transcodeApplies, - backfillKinds, + laneOperations, playlistCount, }); @@ -445,7 +445,7 @@ export default async function ChannelDetailPage({ // reads the filesystem and calls controllers, so a client component // must never import it. Only ENABLED kinds are passed — a disabled // feature reports no backfill anywhere. - kinds={backfillKinds.map((kind) => { + kinds={laneOperations.map((kind) => { const entry = snapshot.backfill?.[kind.id]; return { id: kind.id, @@ -465,19 +465,19 @@ export default async function ChannelDetailPage({ deferredHint: kind.deferredHint, // Same reason: every snapshot on disk predates this field. blocked: entry?.blocked ?? 0, - // Resolved here, on the server, because BACKFILL_KIND_BY_ID + // Resolved here, on the server, because OPERATION_BY_ID // reads the filesystem and must never reach a client component. dependsOnLabels: (kind.dependsOn ?? []) - .map((id) => getBackfillKind(id)?.label) + .map((id) => getOperation(id)?.label) .filter((l): l is string => Boolean(l)), stale: entry?.stale ?? 0, }; })} allowRedownload={settings.backfill.allowRedownload} - anyEnabled={backfillKinds.length > 0} + anyEnabled={laneOperations.length > 0} // The lane's name comes from what it HOLDS, not from its queue key. - groupLabel={operationsGroupLabel(backfillKinds.map((k) => k.id))} - actionLabel={operationsActionLabel(backfillKinds.map((k) => k.id))} + groupLabel={operationsGroupLabel(laneOperations.map((k) => k.id))} + actionLabel={operationsActionLabel(laneOperations.map((k) => k.id))} /> ); case "cleanup": { diff --git a/editor/app/channels/components/ChannelsTable.tsx b/editor/app/channels/components/ChannelsTable.tsx @@ -27,7 +27,7 @@ export type ChannelRow = ChannelStat & { // A column heading for one pipeline. Comes off the operation registry on the // server (shortLabel, costBasis) rather than being abbreviated here — see -// backfillKinds.ts. +// operations.ts. export type PipelineColumn = { id: string; shortLabel: string; diff --git a/editor/app/channels/groupActions.ts b/editor/app/channels/groupActions.ts @@ -19,7 +19,7 @@ import { downloadMissingAction, syncAction } from "./[slug]/pipelineActions"; import { transcribeMissingAction } from "./[slug]/whisperActions"; import { backfillChannelAction } from "./[slug]/backfillActions"; import { runOperationChannelJob } from "yt-dlp-transcript-common/controller/operationJobs"; -import { DIGEST_KIND_ID } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { DIGEST_OPERATION_ID } from "yt-dlp-transcript-common/lib/operations"; // Run one pipeline stage over every channel in one of a site's groups. // @@ -70,7 +70,7 @@ const RUN_FOR: Record< runOperationChannelJob({ paths: getPaths(), channelSlug: slug, - operation: DIGEST_KIND_ID, + operation: DIGEST_OPERATION_ID, digest: { lane: "local" }, onDone: () => revalidatePath(`/channels/${slug}`), }), diff --git a/editor/app/channels/lib/channelGroupSections.test.ts b/editor/app/channels/lib/channelGroupSections.test.ts @@ -6,7 +6,7 @@ import type { } from "yt-dlp-transcript-common/controller/channels"; import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig"; import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; -import type { BackfillSnapshotEntry } from "yt-dlp-transcript-common/lib/backfillKinds"; +import type { OperationSnapshotEntry } from "yt-dlp-transcript-common/lib/operations"; import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings"; import type { Site } from "yt-dlp-transcript-common/lib/site"; import { normalizeBuckets } from "../[slug]/lib/stageStatus"; @@ -30,14 +30,14 @@ function snapshotOf(patch: Partial<ChannelSnapshot> = {}): ChannelSnapshot { }; } -function entry(patch: Partial<BackfillSnapshotEntry>): BackfillSnapshotEntry { +function entry(patch: Partial<OperationSnapshotEntry>): OperationSnapshotEntry { return { ids: [], missing: 0, stale: 0, missingInput: 0, ...patch, - } as BackfillSnapshotEntry; + } as OperationSnapshotEntry; } function channel( @@ -77,7 +77,7 @@ function siteOf(patch: Partial<Site> = {}): Site { }; } -// laneBackfillKinds / allBackfillKinds only read these two branches, so a +// backfillLaneOperations / allOperations only read these two branches, so a // partial cast exercises the real registry predicates. function settingsOf(opts: { diarization?: boolean; diff --git a/editor/app/channels/lib/channelGroupSections.ts b/editor/app/channels/lib/channelGroupSections.ts @@ -11,12 +11,12 @@ import { excludedDownloadIdSet, } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { - DIGEST_KIND_ID, - allBackfillKinds, - laneBackfillKinds, - laneEntriesOf, - reachableBackfillWork, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + DIGEST_OPERATION_ID, + allOperations, + backfillLaneOperations, + backfillLaneEntriesOf, + reachableOperationWork, +} from "yt-dlp-transcript-common/lib/operations"; import type { SiteSettings } from "yt-dlp-transcript-common/lib/settings"; import type { Site } from "yt-dlp-transcript-common/lib/site"; import { normalizeBuckets } from "../[slug]/lib/stageStatus"; @@ -85,9 +85,9 @@ export function laneOffFor( settings: SiteSettings, ): boolean { if (station === "digest") { - return !allBackfillKinds(settings).some((k) => k.id === DIGEST_KIND_ID); + return !allOperations(settings).some((k) => k.id === DIGEST_OPERATION_ID); } - if (station === "backfill") return laneBackfillKinds(settings).length === 0; + if (station === "backfill") return backfillLaneOperations(settings).length === 0; return false; } @@ -169,14 +169,14 @@ export function stationWorkFor( return { eligible: true, work: digestWorkOf(snapshot).reachable }; } - // Backfill. laneEntriesOf, NEVER Object.values: the per-kind map carries + // Backfill. backfillLaneEntriesOf, NEVER Object.values: the per-kind map carries // every catalog operation including digest, which has its own station right // beside this one. The lane filter is what keeps the two figures disjoint. if (!snapshot) return { eligible: true, work: null }; return { eligible: true, - work: laneEntriesOf(snapshot.backfill).reduce( - (n, e) => n + reachableBackfillWork(e), + work: backfillLaneEntriesOf(snapshot.backfill).reduce( + (n, e) => n + reachableOperationWork(e), 0, ), }; diff --git a/editor/app/channels/page.tsx b/editor/app/channels/page.tsx @@ -12,11 +12,11 @@ import { siteChannelSlugs, } from "yt-dlp-transcript-common/lib/site"; import { - allBackfillKinds, + allOperations, operationCatalog, OPERATION_GROUP_ORDER, type OperationGroup, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { buildChannelBands } from "../components/pipelines/buildBands"; import { EXTERNAL_BAND_IDS } from "../components/pipelines/buildBands"; @@ -124,7 +124,7 @@ export default async function ChannelsPage({ // function over the same data, which is what stops a channel figure and a // rail figure disagreeing about what "downloaded" means.) const { ids, columns } = pipelineColumns( - allBackfillKinds(getSettings()).map((k) => k.id), + allOperations(getSettings()).map((k) => k.id), ); // Keyed by slug rather than by index: listChannelStatsFromSnapshots happens // to map the briefs in order today, and pairing a channel's counts with diff --git a/editor/app/components/lanes/LaneDeck.tsx b/editor/app/components/lanes/LaneDeck.tsx @@ -404,7 +404,7 @@ export function LaneDeck({ const backfillAvailable = backfill?.anyKind ?? false; // Empty when the payload has not arrived, or when it predates `kinds` — which // drops the breakdown rather than rendering a row of zeros. - const backfillKinds = backfillAvailable ? (backfill?.kinds ?? []) : []; + const laneOperations = backfillAvailable ? (backfill?.kinds ?? []) : []; // The lane's name in the operator's terms, from the server. "Backfill" is a // queue key; it survives below only on the two controls that genuinely act on // the shared queue, and the card now says what is in it. @@ -494,9 +494,9 @@ export function LaneDeck({ // applies per channel. A single-kind corpus would otherwise be shown a // breakdown of itself, restating the figure one line lower. detail={ - backfillKinds.length > 1 ? ( + laneOperations.length > 1 ? ( <> - {backfillKinds.map((k) => ( + {laneOperations.map((k) => ( <span key={k.id} aria-label={`backfill lane kind ${k.id}`} @@ -532,7 +532,7 @@ export function LaneDeck({ // folding them together loses the one that is a problem. note={ <> - {backfillAvailable && backfillKinds.length > 1 && ( + {backfillAvailable && laneOperations.length > 1 && ( // THE ONE PLACE THE LANE IS STILL NAMED AS A LANE, and it earns it: // that these operations share a queue and a pause is a true, // load-bearing fact — the sweep and the pause are two controls, and @@ -544,8 +544,8 @@ export function LaneDeck({ aria-label="backfill lane members" className="block text-xs text-muted-foreground" > - {backfillKinds.length} operations share one backfill queue and one - pause: {backfillKinds.map((k) => k.label).join(", ")}. + {laneOperations.length} operations share one backfill queue and one + pause: {laneOperations.map((k) => k.label).join(", ")}. </span> )} {backfillSweeping && !laneEnabled && ( @@ -639,7 +639,7 @@ function BacklogLine({ // gated behind an opt-in re-download and `blocked` behind another kind finishing. // Four different units and four different things an operator would have to do, so // a total of any two of them means nothing. This is the same rule -// backfillKinds.ts states for `missing` vs `missing-input`, one level down. +// operations.ts states for `missing` vs `missing-input`, one level down. // // Each clause is omitted at 0, so a kind with nothing outstanding says so in // words rather than showing a row of zeros. diff --git a/editor/app/components/pipelines/band.ts b/editor/app/components/pipelines/band.ts @@ -218,7 +218,7 @@ export function bandHeadline(band: OperationBand): string { // // THIS IS THE ONE PLACE SUMMING ACROSS OPERATIONS IS LEGITIMATE, and it is worth // being precise about why, because every other surface is forbidden from doing -// it. The rule that forbids it (buildBands' header, laneEntriesOf's) is about a +// it. The rule that forbids it (buildBands' header, backfillLaneEntriesOf's) is about a // FIGURE IN NO UNIT: adding diarization's videos to attribution-text's videos // gives a number that is neither, because one unit of the first is an audio pass // and one unit of the second is ~1 model call per transcript chunk. diff --git a/editor/app/components/pipelines/buildBands.test.ts b/editor/app/components/pipelines/buildBands.test.ts @@ -2,7 +2,7 @@ import { test } from "node:test"; import assert from "node:assert/strict"; import { readFileSync } from "node:fs"; import type { ChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot"; -import type { BackfillSnapshotEntry } from "yt-dlp-transcript-common/lib/backfillKinds"; +import type { OperationSnapshotEntry } from "yt-dlp-transcript-common/lib/operations"; import { bandCoverage, buildChannelBands, @@ -23,7 +23,7 @@ function snapshotOf(patch: Partial<ChannelSnapshot> = {}): ChannelSnapshot { } as ChannelSnapshot; } -function entryOf(patch: Partial<BackfillSnapshotEntry>): BackfillSnapshotEntry { +function entryOf(patch: Partial<OperationSnapshotEntry>): OperationSnapshotEntry { return { missing: 0, stale: 0, @@ -33,7 +33,7 @@ function entryOf(patch: Partial<BackfillSnapshotEntry>): BackfillSnapshotEntry { blocked: 0, ids: [], ...patch, - } as BackfillSnapshotEntry; + } as OperationSnapshotEntry; } const bandOf = (bands: OperationBand[], id: string): OperationBand => { diff --git a/editor/app/components/pipelines/buildBands.ts b/editor/app/components/pipelines/buildBands.ts @@ -4,12 +4,12 @@ import { excludedDownloadIdSet, } from "yt-dlp-transcript-common/controller/channelSnapshot"; import { - DIGEST_KIND_ID, + DIGEST_OPERATION_ID, operationCostBasis, operationLabel, - presentBackfillWork, - reachableBackfillWork, -} from "yt-dlp-transcript-common/lib/backfillKinds"; + presentOperationWork, + reachableOperationWork, +} from "yt-dlp-transcript-common/lib/operations"; import { sumOrNull, type OperationBand } from "./band"; export type { OperationBand } from "./band"; @@ -64,20 +64,20 @@ function emptyBand(id: string, dispatched: boolean): OperationBand { function addRegistryEntry(band: OperationBand, snapshot: ChannelSnapshot): void { const entry = snapshot.backfill?.[band.id]; if (!entry) return; - band.reachable += reachableBackfillWork(entry); + band.reachable += reachableOperationWork(entry); band.missingInput += entry.missingInput; // `?? 0` at every read: snapshots written before these fields existed lack // them, and undefined poisons the sum to NaN. band.blocked += entry.blocked ?? 0; band.deferred += entry.deferred ?? 0; band.eligible = sumOrNull([band.eligible, entry.eligible ?? null]); - band.present = sumOrNull([band.present, presentBackfillWork(entry)]); + band.present = sumOrNull([band.present, presentOperationWork(entry)]); } export type BuildOperationBandsInput = { snapshots: ReadonlyArray<ChannelSnapshot | null>; // Registry operations to build a band for, in rail order. Comes from - // allBackfillKinds(), so a switched-off feature is simply absent — which is + // allOperations(), so a switched-off feature is simply absent — which is // the honest rendering: an empty work list because nobody enabled it is not // the same as being finished. operationIds: ReadonlyArray<string>; @@ -92,7 +92,7 @@ export type BuildOperationBandsInput = { // blocked behind diarization). Leaving transcription and download off it would // remove exactly the two lanes whose state explains the other four. // -// Their numbers do NOT come from BackfillKind.state() — they have no entry, +// Their numbers do NOT come from Operation.state() — they have no entry, // because EXTERNAL_OPERATIONS registers them for the dependency graph and not // for dispatch. They come from `totals` and the buckets, using the SAME // definitions the channel transit line already uses for its Download and @@ -176,7 +176,7 @@ export function buildOperationBands({ for (const id of operationIds) { const band = bands.get(id); if (!band) continue; - if (id === DIGEST_KIND_ID && !snapshot.backfill?.[id]) { + if (id === DIGEST_OPERATION_ID && !snapshot.backfill?.[id]) { // A snapshot written before digest joined the registry has no entry, and // digestWorkOf is the fallback that reads the same population off the // buckets with the transcript and cues-staleness gates applied. Without diff --git a/editor/app/jobs/actions.ts b/editor/app/jobs/actions.ts @@ -3,7 +3,7 @@ import { revalidatePath } from "next/cache"; import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings, writeSettings } from "yt-dlp-transcript-common/lib/settings"; -import { laneBackfillKinds } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { backfillLaneOperations } from "yt-dlp-transcript-common/lib/operations"; import { pruneJobLogs } from "yt-dlp-transcript-common/jobs/listJobs"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { readJobMeta } from "yt-dlp-transcript-common/jobs/jobMeta"; @@ -266,7 +266,7 @@ export async function startBackfillSweepAction( // VALIDATED HERE, because sanitizeBackfill does not. // // An unknown kind id survives a settings write and then matches nothing: - // resolveBackfillKinds filters the registry BY the list, so a scope naming + // resolveBackfillLaneOperations filters the registry BY the list, so a scope naming // one operation that has since been renamed arms a sweep that does exactly // no work while reporting itself armed. That is the worst kind of failure // this console can have — the operator reads a plan, clicks, and watches a @@ -275,7 +275,7 @@ export async function startBackfillSweepAction( // So: drop the unknowns, arm what is left, and SAY which were dropped. Not // a refusal — the remaining operations are still what the operator asked // for — and not silence either. - const known = new Set(laneBackfillKinds(getSettings()).map((k) => k.id)); + const known = new Set(backfillLaneOperations(getSettings()).map((k) => k.id)); const wanted = kindIds ?? []; const scope = wanted.filter((id) => known.has(id)); const dropped = wanted.filter((id) => !known.has(id)); diff --git a/editor/app/jobs/active/buildActiveJobs.ts b/editor/app/jobs/active/buildActiveJobs.ts @@ -22,7 +22,7 @@ import { } from "yt-dlp-transcript-common/controller/autoRunner"; import { getDigestSweepJobId } from "yt-dlp-transcript-common/controller/digestSweep"; import { getBackfillSweepJobId } from "yt-dlp-transcript-common/controller/backfillSweep"; -import { laneBackfillKinds } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { backfillLaneOperations } from "yt-dlp-transcript-common/lib/operations"; import { BACKFILL_QUEUE, DIGEST_LOCAL_QUEUE, @@ -335,7 +335,7 @@ function buildLanes(jobs: RunningJobsListItem[]): ActiveLaneView[] { // INVERTED on the wire: backfill's gate is `enabled` where digest's is // `paused`. Normalised here so both lanes mean the same thing downstream. gateHeld: !settings.backfill.enabled, - available: laneBackfillKinds(settings).length > 0, + available: backfillLaneOperations(settings).length > 0, inFlight: backfillInFlight, }), ); diff --git a/editor/app/operations/[id]/page.tsx b/editor/app/operations/[id]/page.tsx @@ -7,7 +7,7 @@ import { selectableBucketsForKind } from "yt-dlp-transcript-common/jobs/autoQueu import { operationCatalog, operationLabel, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; import type { AutoQueueKind } from "yt-dlp-transcript-common/jobs/autoQueueState"; import { getRegistry } from "yt-dlp-transcript-common/jobs/registry"; import { buildAutoQueueStatusPayload } from "../status"; diff --git a/editor/app/operations/components/SweepLane.tsx b/editor/app/operations/components/SweepLane.tsx @@ -80,7 +80,7 @@ export function SweepLane({ // read-only and show the armed scope, and the panel says how to change it. const armed = lane.sweeping; // A stored scope of `[]` means "every enabled operation" — the rule - // resolveBackfillKinds applies and the rule an unscoped sweep runs — so it + // resolveBackfillLaneOperations applies and the rule an unscoped sweep runs — so it // renders as every box ticked rather than none. const allKindIds = lane.operations.map((op) => op.id); const draftKinds = scopeKinds.length > 0 ? scopeKinds : allKindIds; diff --git a/editor/app/operations/components/SweepPlan.tsx b/editor/app/operations/components/SweepPlan.tsx @@ -80,7 +80,7 @@ export function SweepPlan({ return ordered.map((entry) => { const counts = byKind.get(entry.channelSlug); // An empty scope is EVERY operation — the same rule foldSweepPlan and - // resolveBackfillKinds apply, and the same rule an unscoped sweep runs. + // resolveBackfillLaneOperations apply, and the same rule an unscoped sweep runs. const inScope = scopeKinds; const selected = Object.entries(counts?.byKind ?? {}).filter( ([id]) => inScope.length === 0 || inScope.includes(id), diff --git a/editor/app/operations/components/SweepScope.tsx b/editor/app/operations/components/SweepScope.tsx @@ -30,7 +30,7 @@ export function SweepScope({ operations, selected, // Ids the STORED scope names that no operation answers to. sanitizeBackfill - // keeps an unknown id and resolveBackfillKinds then matches nothing with it, + // keeps an unknown id and resolveBackfillLaneOperations then matches nothing with it, // so a sweep armed on one runs forever doing nothing. Reported here rather // than swallowed. unknownScopeIds, diff --git a/editor/app/operations/components/railStates.ts b/editor/app/operations/components/railStates.ts @@ -65,7 +65,7 @@ export function railStates( // the same reasoning the old id-shaped sweepLaneIdFor used: everything that // is not digest is backfill. Both were true only because those are the only // two lanes registered TODAY. An operation on a third queue would still get a - // band (the rail is built from allBackfillKinds — every switched-on kind, + // band (the rail is built from allOperations — every switched-on kind, // whatever its queue) and would then have advertised a hold belonging to a // lane that would never dispatch it. // diff --git a/editor/app/operations/lanes.ts b/editor/app/operations/lanes.ts @@ -2,15 +2,15 @@ import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import type { AutoQueueOrder, AutoQueueReach } from "yt-dlp-transcript-common/jobs/autoQueuePolicy"; import { - allBackfillKinds, - DIGEST_KIND_ID, - laneBackfillKinds, + allOperations, + DIGEST_OPERATION_ID, + backfillLaneOperations, operationCostBasis, operationLabel, operationsActionLabel, operationsGroupLabel, type OperationDescriptor, -} from "yt-dlp-transcript-common/lib/backfillKinds"; +} from "yt-dlp-transcript-common/lib/operations"; import { buildSweepChannelCounts } from "yt-dlp-transcript-common/controller/sweepPreview"; import { loadSweepDates } from "yt-dlp-transcript-common/controller/sweepRecency"; import type { SweepChannelCounts } from "yt-dlp-transcript-common/lib/sweepPlan"; @@ -196,11 +196,11 @@ export async function buildAutoQueueLanes(): Promise<AutoQueueLanesPayload> { const paths = getPaths(); const settings = getSettings(); const briefs = await getChannelBriefs(paths); - // allBackfillKinds, not laneBackfillKinds: the rail is the CATALOG view — + // allOperations, not backfillLaneOperations: the rail is the CATALOG view — // every operation that is switched on, whatever queue it runs on. Filtering // to the shared backfill queue here would drop digest, which is the whole // reason the rail exists. - const kinds = allBackfillKinds(settings); + const kinds = allOperations(settings); const bands = buildOperationBands({ snapshots: briefs.map((c) => c.snapshot ?? null), @@ -222,12 +222,12 @@ export async function buildAutoQueueLanes(): Promise<AutoQueueLanesPayload> { // (That guard greps for the NAME, in any context — so it is not written here // even in prose. Dumb on purpose: a guard that skipped comments would be a // guard an offending call could hide behind.) - const laneKinds = laneBackfillKinds(settings); + const laneKinds = backfillLaneOperations(settings); const backfillKindIds = laneKinds.map((k) => k.id); // Digest is one operation on its own queue, always catalogued whether or not // the feature is switched on — the lane is always available (see below), so // its plan must be too. - const digestKindIds = [DIGEST_KIND_ID]; + const digestKindIds = [DIGEST_OPERATION_ID]; const planKindIds = [...new Set([...backfillKindIds, ...digestKindIds])]; // Dates for the ids both plans will order by. Cached for a minute — see // controller/sweepRecency.ts for the measured cost and why a stale date diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -43,7 +43,7 @@ type Props = { // it must never end up in the client bundle. Same reason as `apps`. digestApps: DigestAppDescriptor[]; // Known worker-tag vocabulary (operation ids + resources), built on the - // server for the same bundle reason — the catalog lives in backfillKinds.ts. + // server for the same bundle reason — the catalog lives in operations.ts. workerTags?: string[]; }; diff --git a/editor/app/settings/components/WorkersField.tsx b/editor/app/settings/components/WorkersField.tsx @@ -85,7 +85,7 @@ type Props = { apps: TranscriptionAppDescriptor[]; name: string; // The known tag vocabulary (operation ids + resource names), built on the - // server: the catalog lives in backfillKinds.ts, whose import graph must not + // server: the catalog lives in operations.ts, whose import graph must not // reach the client bundle. Used only to WARN about a tag the scheduler will // never match — an unknown tag saves fine. knownTags?: string[]; diff --git a/editor/app/settings/page.tsx b/editor/app/settings/page.tsx @@ -3,7 +3,7 @@ import { getPaths } from "yt-dlp-transcript-common/lib/paths"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { listTranscriptionApps } from "yt-dlp-transcript-common/lib/transcriptionApps"; import { listDigestApps } from "yt-dlp-transcript-common/lib/digestApps"; -import { operationCatalog } from "yt-dlp-transcript-common/lib/backfillKinds"; +import { operationCatalog } from "yt-dlp-transcript-common/lib/operations"; import { WORKER_RESOURCE_TAGS } from "yt-dlp-transcript-common/lib/workers"; import { readXSessionStatus } from "yt-dlp-transcript-common/social/xSessionBroker"; import { SettingsForm } from "./components/SettingsForm"; diff --git a/editor/e2e/channel-line.spec.ts b/editor/e2e/channel-line.spec.ts @@ -33,7 +33,7 @@ type Entry = Record<string, unknown>; // Settings with the diarization backfill on or off. The models are /dev/null // placeholders — nothing here runs the lane, it only has to be ENABLED so that -// laneBackfillKinds reports it. +// backfillLaneOperations reports it. function laneSettings(enabled: boolean) { return { adminTitle: "Test Admin",