// The single source of truth for what a job `kind` IS. Before this table, the // same per-kind knowledge was scattered across four places: the DRAINABLE_KINDS // set (registry.ts), the JOB_KIND_LABELS map (editor jobKindLabels.ts), the // switch in editor runJobSpec.ts, and the spec-presence replay check. Phase 1 // of the queue refactor consolidates the drainable + label data here; later // phases consume `defaultTier` (scheduler) and lean on `replayable`. // // Adding a new job kind should mean adding ONE entry here (plus its replay // handler in editor/app/jobs/jobReplayRegistry.ts if it is replayable). // How a kind picks its registry queueKey. Descriptive only — the real key is // still computed by the action that creates the job (channelQueueKey / // downloadQueueKey / ""). "parallel" means queueKey === "" (runs unserialized). export type QueueKeyStrategy = "parallel" | "platform" | "custom"; // Scheduler priority tiers. A higher-priority tier's queued jobs run before a // lower one's; FIFO within a tier. Consumed starting Phase 4 (the scheduler); // today every non-background job is effectively "foreground". export type SchedulerTier = "urgent" | "foreground" | "background"; export type JobKindMeta = { kind: string; // Human-readable label for the UI. Optional: a kind without one falls back to // its raw machine kind (preserving the prior JOB_KIND_LABELS fallback), so a // new kind is never invisible. label?: string; // The Drain (soft-cancel) button is only offered for running jobs of a // drainable kind — i.e. one whose controller honors the drain signal (stops // starting new sub-operations, lets in-flight ones finish). drainable: boolean; // Whether this kind's action attaches a replayable JobSpec (and thus has a // replay handler). Actual replayability still keys off spec PRESENCE on the // record at runtime — this flag just says the kind CAN be retried. replayable: boolean; // Descriptive: how the action derives its queueKey. Not consumed as logic. queueKeyStrategy: QueueKeyStrategy; // Fallback scheduler tier when a record is not explicitly background. Left // undefined today so Phase 4 maps it to "foreground" (unchanged behavior). defaultTier?: SchedulerTier; // Whether this kind OPENS OR WRITES A BIG FILE — the audio, a persisted // source container, the raw live-chat replay (lib/mediaTier.ts says which): // the files that may live on another drive (release 17, the media tier). // Declarative, and consumed by runManagedFunction, which refuses a media kind // for a channel whose media is not reachable (a relocated channel whose drive // is unmounted or stalled, one mid-move, a `legacy` one) — at enqueue and // again when its queue starts it — rather than letting it read an empty dir // as the truth or write into a tree being copied. See common/lib/channelMedia.ts. // // ABSENT MEANS FALSE, and that is deliberate rather than lazy. The guard is // opt-in so a kind that is not listed keeps exactly its current behavior, and // so that `refresh-report` — which is not in this table at all — still runs // and lets the snapshot generator itself report the reason it refused. A kind // that genuinely never opens a video dir (store playlist, clear markers) must // not be refused for a drive it does not read. // // THE TEST IS "DOES IT OPEN OR WRITE THE BIG FILE", NOT "IS IT BOOKKEEPING" // (the doctrine since release 17; before it, "does it open data/"). `sync` // and the metadata scan's cousins that DOWNLOAD write media; a transcription // reads it. A kind that reads only the text — a digest, normalize, the // availability checks, the metadata scan, a clip-window fetch — declares // `needsText` instead, and runs while the channel's media is moving, stalled // or unmounted. needsMedia?: boolean; // Whether this kind walks or writes the TEXT under `channels//data/` // (transcripts, cues, metadata, sidecars, `clips/`) and nothing else. The // text guard (`assertChannelTextReadable`) refuses it only where the text // itself cannot be read: a `legacy` channel (the retired whole-directory // layout, its text on the far drive), a `data/` that is not a directory, a // tier migration in flight. Against such a channel the walk would find // nothing and report a clean run over zero videos — or, for the // availability checks and the metadata scan, read every downloaded video as // missing and re-request the whole channel. Absent means false. needsText?: boolean; }; // One entry per kind known to the system. `label` is included only where the // old JOB_KIND_LABELS had one, so jobKindLabel() behavior is byte-identical. const JOB_KINDS: Record = { "auto-transcribe": { kind: "auto-transcribe", label: "Auto-transcribe", drainable: true, replayable: false, queueKeyStrategy: "parallel", needsMedia: true, }, "auto-download": { kind: "auto-download", label: "Auto-download runner", drainable: true, replayable: false, queueKeyStrategy: "parallel", needsMedia: true, }, "auto-download-unit": { kind: "auto-download-unit", label: "Auto-download", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: true, }, // THE OPERATOR'S "Clear hold" on /operations/download (release 17 slice // RL): one step that drops a held platform's hold, backoff and raised pace, // and says so in its log. Its own queue (`pacing:`), so a Sync // running on the platform's queue never makes the click wait. Touches no // channel and no file but the auto-queue state. "clear-platform-hold": { kind: "clear-platform-hold", label: "Clear rate-limit hold", drainable: false, replayable: false, queueKeyStrategy: "custom", }, // The two OPERATION lanes' runners. Same shape as the two above — one // long-lived job per lane on queueKey "", drainable, never replayable — with // one difference worth stating: their units make NO job record. A digest or a // diarization dispatched by the runner runs in-process inside the loop, so // there is no "auto-digest-unit" here to match auto-download-unit. That is // what lets a lane work per video where the manual verbs must work per // channel: the registry keeps 100 records and the log 500, and 55,956 digest // jobs would evict the history of the run that made them. "auto-digest": { kind: "auto-digest", label: "Auto-digest runner", drainable: true, replayable: false, queueKeyStrategy: "parallel", needsMedia: false, needsText: true, }, "auto-backfill": { kind: "auto-backfill", label: "Auto-backfill runner", drainable: true, replayable: false, queueKeyStrategy: "parallel", needsMedia: true, }, "whisper-all": { kind: "whisper-all", label: "Transcribe all", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, "whisper-bucket-downloaded-no-transcript": { kind: "whisper-bucket-downloaded-no-transcript", label: "Transcribe downloaded audio", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // Replace-auto-captions lane, transcribe half: whisper over videos whose only // transcript is a YouTube ASR VTT (the downloadedAutoSubsOnly bucket). Same // batch machinery as whisper-bucket-downloaded-no-transcript — drainable and // replayable, re-deriving the bucket's current members on replay. "whisper-bucket-auto-subs": { kind: "whisper-bucket-auto-subs", label: "Replace auto-captions", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // Delete the superseded English ASR VTTs kept as backups next to a finished // whisper transcript. Manual only — never auto-queued — and the single // irreversible step in the lane, so it is deliberately NOT drainable (it is a // fast file sweep) but IS replayable. "purge-superseded-auto-subs": { kind: "purge-superseded-auto-subs", label: "Purge superseded auto-captions", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, // AI digest sweep, local (ollama) lane — the one that carries the corpus. Both // digest kinds are drainable (the batch honors the drain signal: it stops // pulling new videos and lets the in-flight one finish) and replayable, since // a channel-scoped sweep is exactly the kind of thing an operator re-launches. "digest-channel-local": { kind: "digest-channel-local", label: "Digest channel (local)", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, // Same batch, metered lane. Off unless settings.digest.remoteEnabled is true, // and it lands on its own queue key so it runs CONCURRENTLY with the local lane // rather than behind it. "digest-channel-remote": { kind: "digest-channel-remote", label: "Digest channel (metered)", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, // Copy a duplicate cluster's canonical digest onto its aligned mirrors. A fast // file operation gated by the timestamp-alignment check, so it is not drainable // but is replayable. "digest-share-cluster": { kind: "digest-share-cluster", label: "Share cluster digest", drainable: false, replayable: false, queueKeyStrategy: "parallel", needsMedia: false, needsText: true, }, // Write the compact transcript.cues.json sidecar next to every raw transcript // that lacks a current one — corpus-wide from the Pool on /sites, or one // channel from the digest stage card. It was an UNREGISTERED kind string // until now: passed to runManagedFunction with no entry here, so /jobs showed // the raw machine kind. // // Not drainable: the walk honours the cancel signal (which is what stops it) // but has no drain-aware inner loop, and the unit of work is a single file // write, so "let the in-flight one finish" is already how it behaves. Not // replayable either — that would need a JobSpec and a jobReplayRegistry // handler, and the button is one click from the card that reports the count. // needsText (release 17; needsMedia before it): it walks `data//` for // every video and writes a sidecar into each — text, on the corpus disk. // Against an unreadable text tier (a `legacy` channel whose drive is // unmounted) it finds nothing, reports a clean 0/0/0/0 run and moves on — a // sweep that silently skips a channel. A stalled MEDIA drive does not stop it. "normalize-transcripts": { kind: "normalize-transcripts", label: "Normalize transcripts", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, "redownload-incomplete-bucket": { kind: "redownload-incomplete-bucket", label: "Re-download truncated transcripts", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, "download-from-playlist": { kind: "download-from-playlist", label: "Download from playlist", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: true, }, "download-missing": { kind: "download-missing", label: "Download missing", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: true, }, "download-missing-subs": { kind: "download-missing-subs", label: "Download missing subs", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, "import-one": { kind: "import-one", label: "Import video", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, // CHOSEN FILES OF ONE archive.org ITEM, imported one at a time on // archive.org's own queue (controller/archiveOrgImport.ts). Drainable: a // drain lets the file in flight finish and starts no more; re-running the // same command resumes, skipping what landed. "import-archive-org": { kind: "import-archive-org", label: "Import from archive.org", drainable: true, replayable: false, queueKeyStrategy: "platform", needsMedia: true, }, // AN ODYSEE OR BITCHUTE CHANNEL'S LISTING, diffed against what is held // (release 19 A6, controller/remoteListing.ts): one flat-playlist read on the // platform's queue, nothing written. Reads data/ (the text tier) for the // held ids, so the text guard covers it. "remote-listing": { kind: "remote-listing", label: "Remote listing", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // ONE WINDOW of a video's source media, fetched into data//clips/ for a // tool that asked for it by name (umtool's clip bench). On its platform's // CLIP queue (`clips:`, lib/queueKeys.ts clipWindowQueueKey — // release 19, A5), one window at a time per platform, never behind the // platform's long downloads; replayable because the request is a few numbers // and a reason, which is exactly what a JobSpec holds. "fetch-window": { kind: "fetch-window", label: "Fetch window", drainable: false, replayable: true, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // A LIST OF CLIP WINDOWS, one platform's, as one paced job // (controller/fetchWindows.ts): a site's missing evidence, a umtool // manifest's timeline, an explicit list. One job per platform's clip queue // (`clips:`, as fetch-window), so YouTube and Rumble run side by // side, each paced on its own. Drainable: it // stops between windows and a re-run fetches only what is still missing, // which is also why it is replayable. It spans channels and so starts with no // channelSlug — the per-channel text check `needsText` stands for is made by // the controller, once per channel, before that channel's first window. "fetch-windows": { kind: "fetch-windows", label: "Fetch windows", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, "redownload-archive": { kind: "redownload-archive", label: "Archive source video", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, "retry-bucket": { kind: "retry-bucket", label: "Retry", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: true, }, "clean-audio-transcribed": { kind: "clean-audio-transcribed", label: "Clean audio", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // Speaker-diarization backfill over a channel's retained audio. The capture // lane's catch-all: it picks up everything the post-transcribe hook missed // (videos transcribed before the feature, or with the inline hook off — which // is the recommended way to run a large batch). Drainable, because it is // CPU-hours of work an operator will want to stop without losing what it has // already written, and replayable, because it re-derives its work-list from // disk on every run and so replays correctly with nothing remembered. "diarize-channel": { kind: "diarize-channel", label: "Diarize speakers", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // One channel through the backfill lane. Drainable (the batch stops taking new // videos and lets the in-flight one finish) and replayable, because it // re-derives its work-list from disk on every run and so replays correctly // with nothing remembered — the same contract diarize-channel has. "backfill-channel": { kind: "backfill-channel", label: "Backfill channel", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // The corpus-wide corrupt-media scan and its per-channel twin. Registered // properly, unlike check-availability / refresh-report / detect-duplicates, // which have no entry here and fall back to their raw machine kind in the UI — // that absence is a gap, not a pattern worth copying. "scan-media": { kind: "scan-media", label: "Scan media for corruption", // Reports only; there is nothing in flight to let finish. drainable: false, replayable: false, // Local disk work with nothing to serialize against — see the queueKey "" // escape hatch refresh-report and detect-duplicates use. queueKeyStrategy: "parallel", defaultTier: "background", needsMedia: true, }, "scan-media-channel": { kind: "scan-media-channel", label: "Scan channel media", drainable: false, replayable: false, queueKeyStrategy: "parallel", defaultTier: "background", needsMedia: true, }, "check-kept-deleted": { kind: "check-kept-deleted", label: "Check kept videos", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, "persist-kept": { kind: "persist-kept", label: "Persist kept videos", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // Persist an explicit list of videos (controller/persistVideos.ts): one job // per channel, on the channel's download queue as persist-kept is. Drainable: // it stops between videos and leaves the rest for a re-run. "persist-videos": { kind: "persist-videos", label: "Persist videos", drainable: true, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, "backup-saved-videos": { kind: "backup-saved-videos", label: "Back up saved videos", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "verify-saved-video-backup": { kind: "verify-saved-video-backup", label: "Verify saved-video backup", drainable: false, replayable: false, queueKeyStrategy: "custom", }, // MOVE A CHANNEL'S MEDIA TO ANOTHER DRIVE, AND BACK // (plans/relocate-channel-media.md). // // `needsMedia: false` is written out rather than omitted, and this is the one // entry where the explicit `false` earns its line: this kind is what FIXES an // unreachable channel. Guard 1 in runManagedFunction refuses a media kind for // a channel whose media it cannot reach — so a kind that declared `true` here // would be refused precisely when the operator needs it (a `back` run after // remounting, or a retry of an interrupted move), and the only action that can // clear the condition would be the one the condition blocks. // // Not drainable: the work is one rsync child, and stopping it is a cancel — // which the AbortSignal already does, leaving the source untouched and the // partial copy resumable. There is no "stop starting new sub-operations" to // honor. Not replayable: a replay carries no direction and no root, and // re-running a move against a channel that has since moved is not a retry. // // Queue key is relocationQueueKey() (set by the one enqueue both actions // share), so EVERY relocation in the process serializes against every other // one: the registry caps a key at concurrency 1 and caps nothing across keys, // and a bulk move's jobs all write to the same destination volume and all run // their space check when they START. Serializing against the channel's own // bookkeeping jobs is not what the key buys — both actions refuse a channel // that has running or queued jobs before they enqueue, which refuses rather // than waits. "relocate-channel-media": { kind: "relocate-channel-media", label: "Relocate channel media", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: false, }, // FETCHED CLIP WINDOWS, BY AGE. `data//clips/` had no garbage collection // at all: the retention sweep is pointer-driven and never sees a window, and // the cleanup lanes are about `audio.*`. This is the only thing that removes // one. // // `needsText` (release 17): `clips/` is on the corpus disk and is never // tiered, so the media drive does not concern it. The guard still matters: // it DELETES, and against an unreadable text tier (a `legacy` channel whose // drive is unmounted) every `readdir` of `data/` throws and the walk would // report a clean eviction of zero bytes. "evict-clips": { kind: "evict-clips", label: "Evict fetched windows", drainable: false, replayable: false, queueKeyStrategy: "parallel", needsMedia: false, needsText: true, }, // THE SAVED-VIDEO STORE, ONTO A LOCATION AND BACK. Same mechanism as the // channel move (relocateDir.ts is literally the same code) over one directory // that belongs to no channel — so no `channelSlug`, and `needsMedia: false` // for the relocation's reason: this kind is what FIXES a store on a drive // that is not there. // // ON THE RELOCATION QUEUE KEY, with the channel move and the re-point. All // three rewrite symlinks under the same roots and all three run their space // check when they START; the registry caps a key at concurrency 1, so one at // a time across the three kinds is the whole point. "relocate-saved-videos": { kind: "relocate-saved-videos", label: "Move the saved-video store", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: false, }, // THE SAME LINKS, WITHOUT THE BYTES. A re-point rewrites every channel // symlink on one storage location plus the location's root, for the case the // relocation above cannot help with: the media never moved, the DISK did, and // every `config.mediaDir` on it now names a mountpoint that is not there. // // `needsMedia: false` for the relocation's reason, and more sharply: every // channel this job touches is BY DEFINITION unreachable when it starts — // that is the condition it exists to clear — so `true` here would refuse the // one action that can fix it. // // Not drainable: there is no "stop starting new sub-operations" to honor. The // work is n unlinks, n symlinks and n small writes, and a failure rolls the // whole ledger back rather than leaving a partial run to resume. Not // replayable: a replay carries no new root, and re-running against a location // that has since been re-pointed is not a retry. // // THE QUEUE KEY IS relocationQueueKey(), SHARED WITH THE MOVE. The registry // caps a key at concurrency 1, so a re-point can never run beside a // relocation that is rewriting the very links it is about to rewrite — and // only one re-point runs at a time. "repoint-storage-location": { kind: "repoint-storage-location", label: "Re-point storage location", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: false, }, // Social-post ingest for a `sourceKind: "social"` channel. Drainable (the // fetcher stops paging on the drain signal and keeps what it already has) and // replayable. queueKeyForUrl() routes x.com / bsky.app to // `platform:x.com` / `platform:bsky.app`, so per-platform serialization and // the existing 429 backoff come free. "fetch-posts": { kind: "fetch-posts", label: "Fetch posts", drainable: true, replayable: true, queueKeyStrategy: "platform", }, // Forum thread pages the operator saved from a browser, imported into a // forum-thread channel (controller/importForumPages.ts). Offline — nothing is // fetched — but on the PLATFORM queue so it never writes the channel's posts // beside a fetch writing the same shards. Not drainable (one parse and one // write); not replayable (the files named are the operator's, and may be // gone). "import-forum-pages": { kind: "import-forum-pages", label: "Import saved forum pages", drainable: false, replayable: false, queueKeyStrategy: "platform", }, // A screenshot and the attached media of specific archived posts (and the X // Article a post links to), into the channel's `posts-media/` // (controller/capturePosts.ts). On the PLATFORM queue, like fetch-posts: // each post is a page load and a download against the same source, so it // serialises with the fetch and shares its backoff. // Drainable (it stops between posts) and replayable (the ids are in the // spec; posts already captured are skipped on a re-run). "capture-posts": { kind: "capture-posts", label: "Capture posts", drainable: true, replayable: true, queueKeyStrategy: "platform", }, // The posts analogue of the video availability check: which archived posts // have since been deleted at the source. "check-post-availability": { kind: "check-post-availability", label: "Check deleted posts", drainable: true, replayable: true, queueKeyStrategy: "platform", }, // needsMedia: syncPaged does not merely refresh a playlist — it reads the // channel's data dir to decide what is already there and downloads into // `data//`. Against an unmounted drive its listing is empty, which means // "nothing is downloaded", which means download everything. sync: { kind: "sync", label: "Sync", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: true, }, // The metadata scan (ytdlp/metadataScan.ts). On the platform download queue, // because it contends for the same thing a download does — the source's // patience. // // `needsText` (release 17; `needsMedia` before it): its whole target set is // "listed, minus what is already on disk", and "on disk" is the video dirs in // `data/` — text. Against an unreadable text tier (a `legacy` channel whose // drive is unmounted) it would read every downloaded video as unfetched and // re-request the entire channel. It writes no media, so a stalled media // drive does not hold it. "metadata-scan": { kind: "metadata-scan", label: "Metadata scan", drainable: true, replayable: true, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // A podcast channel's records completed from its RSS feed // (controller/feedMetadataBackfill.ts): one fetch of the feed, then a write // of title, date, description and duration into each metadata.info.json that // lacks them. On the platform download queue, like the metadata scan, so the // fetch takes its turn with the host and the writes never land beside a // download rewriting the same file. `needsText`: it reads and writes the // text tier only, and against an unreadable one would report a clean run // over zero records. Not drainable (one request and a few small writes); not // replayable (a re-run is the same click, and finds the completed records // complete). "feed-metadata": { kind: "feed-metadata", label: "Backfill feed metadata", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // ONE video's metadata.info.json re-read from its source // (controller/refreshVideoMetadataJob.ts): a metadata-only yt-dlp pass, no // subtitles, no media, the rewrite recorded in metadata.history.json. On the // platform download queue, like the metadata scan, for the same reason. // `needsText`: it writes one text file and opens no big one. Not drainable // (one spawn); replayable (the spec carries the video id, and a Retry // re-resolves the target and re-asks the cooldown). "refresh-metadata": { kind: "refresh-metadata", label: "Refresh metadata", drainable: false, replayable: true, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // THE HUB'S AND THE HOMEPAGE'S BUILD AND DEPLOY (release 13 slice W1). They // ran from /sites — the hub since release 7, the homepage since release 11 — // with no entry here, so /jobs showed their raw machine kinds. The labels // were the lanes' own titles on /sites; since release 18 those buttons are // the Publish panel's and run `publish-*` stages, and these kinds are only // read back from the history. // // Queue: BUILD_QUEUE for a build, DEPLOY_QUEUE for a deploy or a // build-and-deploy (the actions' `queueKey`), hence "custom". Neither // drainable (one child process; stopping it is a cancel) nor replayable (no // JobSpec: a replayed deploy would not know its preview branch). No // `needsMedia`: they read the index and the export output, never a // channel's `data/`, and carry no channelSlug for the guard to check. "build-hub": { kind: "build-hub", label: "Build hub", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "deploy-hub": { kind: "deploy-hub", label: "Deploy hub", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-deploy-hub": { kind: "build-deploy-hub", label: "Build & deploy hub", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-homepage": { kind: "build-homepage", label: "Build homepage", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "deploy-homepage": { kind: "deploy-homepage", label: "Deploy homepage", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-deploy-homepage": { kind: "build-deploy-homepage", label: "Build & deploy homepage", drainable: false, replayable: false, queueKeyStrategy: "custom", }, // THE PUBLISH STAGES (release 18): one job per stage on the `publish` queue, // each a child process (`archilyzer stage `, publish/ // stageRun.ts) the editor spawns through runManagedCommand — so Cancel kills // it, and there is nothing to drain. Replayable: the spec's params ARE the // StageRequest (publish/publishStages.ts), and a replay re-asks the stage's // preconditions on disk. The labels are the stages' own (publish/stages.ts // STAGES[kind].label; jobKinds.test.ts holds them equal). No channelSlug and // no `needsMedia`: a stage reads the index and the bundles, and the update- // index child guards its own channels (the index build's hold). "publish-update-index": { kind: "publish-update-index", label: "Update the index", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-build-site": { kind: "publish-build-site", label: "Build site", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-deploy-site": { kind: "publish-deploy-site", label: "Deploy site", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-build-hub": { kind: "publish-build-hub", label: "Build hub", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-deploy-hub": { kind: "publish-deploy-hub", label: "Deploy hub", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-build-homepage": { kind: "publish-build-homepage", label: "Build homepage", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "publish-deploy-homepage": { kind: "publish-deploy-homepage", label: "Deploy homepage", drainable: false, replayable: true, queueKeyStrategy: "custom", }, // THE PUBLISH LANE'S RUNNER (publish/publishRunner.ts): one long-lived job on // queueKey "", like the four auto-queue runners. Drainable — a drain lets the // stage in flight finish and dispatches no more — and never replayable: a // loop is not a unit of work. "auto-publish": { kind: "auto-publish", label: "Auto-publish runner", drainable: true, replayable: false, queueKeyStrategy: "parallel", }, // THE KINDS RELEASE 18 NO LONGER CREATES. Their archived metas still read, // so they stay known, labels exactly as they were: the six the hub and the // homepage have carried since release 13 above, and these seven with NONE // (/jobs has always shown them by their raw kind, and e2e reads that text). "build-index": { kind: "build-index", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-stats": { kind: "build-stats", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-export": { kind: "build-export", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-deploy": { kind: "build-deploy", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-all": { kind: "build-all", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "build-deploy-all": { kind: "build-deploy-all", drainable: false, replayable: false, queueKeyStrategy: "custom", }, "deploy-export": { kind: "deploy-export", drainable: false, replayable: false, queueKeyStrategy: "custom", }, // A REPORT SITE'S EVIDENCE MEDIA (publish/reportMedia.ts): every clip its // published reports cite, cut from the media on disk, and every cited post // capture copied, into the site's report-media cache before its build. // `needsMedia`: it opens saved containers and audio, which may be on another // drive. It spans channels and runs with no `channelSlug`, so the guard in // runManagedFunction does not ask; the step itself reports a span on an // unreachable channel as unreachable rather than missing. Its own queue // (`reports-prepare`): one at a time, behind nothing — not the download // queues, not the build queue. Replayable: the spec is the site. "reports-prepare": { kind: "reports-prepare", label: "Prepare report media", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // A REPORT SITE'S EXPORTS (publish/reportExports.ts): each published report // as report.html, report.pdf, report.md and an evidence pack, into the // site's report-exports staging, for compose to publish. It reads the // reports' records (text) and the prepared media cache — never a channel's // big files — so `needsText`; like prepare it spans channels with no // `channelSlug`, and compose's own check reports an unreadable channel. On // prepare's queue (`reports-prepare`): it reads the cache prepare writes and // prunes. Replayable: the spec is the site (and the report and formats). "reports-export": { kind: "reports-export", label: "Export reports", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: false, needsText: true, }, // THE PER-VIDEO WRITERS THAT WERE NOT IN THIS TABLE (release 16 slice RM). // Each runs with a channelSlug and writes under `data//` — a single // video's transcription (the video page's two Transcribe buttons, and its // worker path), its download, its audio transcode, and the availability // checks (`availability.json` per video; the quick check diffs the playlist // against what is on disk) — and each, being absent, was never refused for // a channel whose media is moving or unreachable. Labels stay absent so /jobs // shows them exactly as before. "whisper-video": { kind: "whisper-video", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, "transcribe-one": { kind: "transcribe-one", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, // ONE ARBITRARY FILE (or a window of it) through a local worker, for a quote // check — `pnpm ops transcribe` (controller/transcribeFile.ts). No channel: // it reads the file it was given and writes only to a scratch dir and the // caller's `out`, never into the corpus, so neither guard applies. Parallel // (queueKey ""): the worker pool serialises it against every other // transcription. One engine run — cancelled, not drained; not replayable, // since the file it names may be gone by the time anyone retries. "transcribe-file": { kind: "transcribe-file", label: "Transcribe file", drainable: false, replayable: false, queueKeyStrategy: "parallel", }, "download-one-pipeline": { kind: "download-one-pipeline", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, "transcode-audio": { kind: "transcode-audio", drainable: false, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, "check-availability": { kind: "check-availability", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, "quick-availability-check": { kind: "quick-availability-check", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, "check-maybe-missing": { kind: "check-maybe-missing", drainable: false, replayable: false, queueKeyStrategy: "platform", needsMedia: false, needsText: true, }, // Replayable kinds that never had a JOB_KIND_LABELS entry: label omitted so // jobKindLabel() keeps falling back to the raw kind (unchanged behavior). "store-playlist": { kind: "store-playlist", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "clear-failed-transcriptions": { kind: "clear-failed-transcriptions", drainable: false, replayable: true, queueKeyStrategy: "custom", }, "clean-extra-audio-formats": { kind: "clean-extra-audio-formats", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, "remove-wrong-format-audio": { kind: "remove-wrong-format-audio", drainable: false, replayable: true, queueKeyStrategy: "custom", needsMedia: true, }, // LOCAL MEDIA ATTACHED TO HELD VIDEOS (release 21 D1, // controller/attachMedia.ts): each video's file copied out of a local // archive (a directory, a zip read in place, a 7z) into the saved-video // store. Media: it writes the big file. Drainable: it stops between videos, // and a re-run resumes. An INGEST kind: it writes `saved-video.json` beside // each record and, with `createRecords`, whole new records the index reads. // On the channel's own queue (`channel:`): nothing is fetched, so no // platform queue is owed a turn. Not replayable: the archive it names is a // path on some drive, and a retry is a re-run. "attach-media": { kind: "attach-media", label: "Attach local media", drainable: true, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, // PLAYABLE COPIES AND THEIR TORRENTS (release 21 D3, // controller/preparePlayable.ts): each saved container remuxed losslessly // into a browser's container, and one torrent per copy. Media: it reads the // saved container and writes copies as big. Drainable: it stops between // videos. An INGEST kind: the playable manifest it writes is what a site's // build reads for its torrents (D4). One queue machine-wide (`playable`): // a remux is disk-bound on the archive drive, and two at once only thrash // it. Heavy: each remux takes release 19's heavy slot (`pnpm heavy`'s gate), // one video at a time, so builds and e2e interleave with a long run. "prepare-playable": { kind: "prepare-playable", label: "Prepare playable copies", drainable: true, replayable: false, queueKeyStrategy: "custom", needsMedia: true, }, }; export function getJobKind(kind: string): JobKindMeta | undefined { return JOB_KINDS[kind]; } // Human-readable label, falling back to the raw kind for unknown/label-less // kinds (a new kind is never invisible). export function jobKindLabel(kind: string): string { return JOB_KINDS[kind]?.label ?? kind; } // THE PUBLISH STAGES' JOB KINDS (release 18): `publish-`. export function isPublishStageJobKind(kind: string | undefined): boolean { return typeof kind === "string" && kind.startsWith("publish-") && JOB_KINDS[kind] !== undefined; } // THE KINDS WHOSE `done` CAN CHANGE WHAT THE INDEX READS (release 18): what // makes the index stale, and a site's channels "changed", to the publish // status (publish/publishState.ts). The plan's words are "the drainable // kinds", and every drainable kind that runs per channel is here // (jobKinds.test.ts holds that) — plus the per-video and one-shot writers of // the same text and posts that are not drainable: the auto-download lane's // unit, a single import, transcription or download, the availability checks, // the forum import, the feed backfill and the cues sweep. The lane RUNNERS are // not: their meta is `running` for as long as the lane is, and their units // either have their own job (auto-download-unit) or none at all // (transcription, digest and backfill units — publishState.ts reads the // channel's report regeneration for those). const INGEST_KINDS: ReadonlySet = new Set([ "whisper-all", "whisper-bucket-downloaded-no-transcript", "whisper-bucket-auto-subs", "purge-superseded-auto-subs", "digest-channel-local", "digest-channel-remote", "digest-share-cluster", "normalize-transcripts", "redownload-incomplete-bucket", "download-from-playlist", "download-missing", "download-missing-subs", "import-one", "import-archive-org", "redownload-archive", "retry-bucket", "diarize-channel", "backfill-channel", "persist-videos", "persist-kept", "fetch-posts", "import-forum-pages", "capture-posts", "check-post-availability", "sync", "metadata-scan", "feed-metadata", "auto-download-unit", "whisper-video", "transcribe-one", "download-one-pipeline", "check-availability", "quick-availability-check", "check-maybe-missing", "attach-media", "prepare-playable", ]); export function isIngestKind(kind: string | undefined): boolean { return typeof kind === "string" && INGEST_KINDS.has(kind); } // Every registered kind (for the tests that hold one table to another). export function jobKindIds(): string[] { return Object.keys(JOB_KINDS); } export function isDrainableKind(kind: string): boolean { return JOB_KINDS[kind]?.drainable ?? false; } // Whether a kind's work opens or writes a big file. Absent = false: the media // guard is opt-in, so an unlisted (or unknown) kind behaves exactly as it did // before the guard existed. export function kindNeedsMedia(kind: string): boolean { return JOB_KINDS[kind]?.needsMedia ?? false; } // Whether a kind reads the text tier only (and so asks the text guard instead // of the media one). Absent = false. export function kindNeedsText(kind: string): boolean { return JOB_KINDS[kind]?.needsText ?? false; }