import type { ChannelBrief, ChannelStat } from "../controller/channels"; import { isSocialChannel } from "../lib/channelConfig"; import { FALLBACK_GROUP, resolveChannelGroupId, sortGroups, type ChannelGroup, } from "../lib/channelGroups"; import { digestWorkOf, excludedDownloadIdSet, } from "../controller/channelSnapshot"; import { DIGEST_OPERATION_ID, allOperations, backfillLaneOperations, backfillLaneEntriesOf, operationsGroupLabel, reachableOperationWork, } from "../lib/operations"; import { defaultChannelPriority, isChannelPaused, } from "../lib/channelPriority"; import type { SiteSettings } from "../lib/settings"; import type { Site } from "../lib/site"; import { normalizeBuckets } from "./pipeline/stageStatus"; // Groups a site's channels into the sections /channels renders, and totals each // section's pipeline work off the SAME snapshot readers the channel page's // transit line uses — so a group figure can never disagree with the per-channel // one. // // No "use server", no node:fs: imported by both the page (server) and the group // actions, and unit-tested with plain tsx --test. The actions reuse // stationWorkFor below, which is what keeps a button from acting on a different // set than the number printed on it. // "speakers" is the backfill LANE's station: every operation sharing // BACKFILL_QUEUE, run over a group by the lane runner. The id names what the // station is about (the operations), not the queue — the queue key and the job // kind behind it still say backfill, on purpose. export type StationId = "sync" | "download" | "transcribe" | "digest" | "speakers"; export type StationWork = { // Slugs this operation applies to at all. A channel whose report says it has // nothing to do IS in here — it is skipped at click time, where the read is // fresh, and dropping it here would make the "+" floor unreadable. eligible: string[]; // The sum of ONE reader across `eligible`. // // NEVER summed with another station's total. Five separate figures on five // separate controls, for the reason channelSnapshot.ts documents at length: a // single "remaining" number puts every channel permanently at the top of // every list. total: number; // Eligible channels with no snapshot at all. `total` is a FLOOR while this is // non-empty, which the UI marks with a trailing "+". unknown: string[]; // digest/speakers only: the lane is switched off in settings. A snapshot's // counts outlive the feature being switched off, so "off" is rendered instead // of 0 (which reads as finished) or — (which reads as unknown). laneOff?: boolean; }; // What the /channels client reads of a section: everything but the stats, // which it only counts and joins by slug. The page projects to this before the // sections cross to the browser, so no ChannelConfig (url, dataDir, yt-dlp // args, cookie settings) rides along with them. export type ChannelGroupSectionView = Omit & { channels: Array<{ slug: string }>; }; export type ChannelGroupSection = { group: ChannelGroup; // Membership order asc (absent sorts last), then slug. channels: ChannelStat[]; sync: StationWork; download: StationWork; transcribe: StationWork; digest: StationWork; speakers: StationWork; // What the speakers station is called, derived from the operations enabled on // the backfill lane exactly as the channel page's stage title is — "Speakers" // once a speaker operation is on, "Derived data" when none is (the default // test settings), so the button never claims work its lane is not doing. speakersLabel: string; }; // Whether a whole station is unrunnable because its lane is switched off. // // The digest kind declares `enabled: () => true` — it has no master switch, and // its pause is honoured at DISPATCH precisely so a paused lane still reports // what is outstanding. So digest is never "off" today; the question is asked of // the registry rather than hardcoded so that if digest ever gains a real switch, // this figure stops lying on its own. export function laneOffFor( station: StationId, settings: SiteSettings, ): boolean { if (station === "digest") { return !allOperations(settings).some((k) => k.id === DIGEST_OPERATION_ID); } if (station === "speakers") return backfillLaneOperations(settings).length === 0; return false; } // The two id lists the transcribe station counts and queues, after the same // download exclusions every other station honours. Disjoint by construction: // an ASR VTT makes a video "transcribed" for `downloadedNoTranscript`, so a // video is in at most one of them. They are two lists because they are two // batches — `runWhisperBatch`'s default scan drains the first, its // replace-auto-captions mode needs an ASR track per id for the second. export function transcribeStationIds( snapshot: NonNullable, ): { missing: string[]; autoSubs: string[] } { const excluded = excludedDownloadIdSet(snapshot); const buckets = normalizeBuckets(snapshot.buckets); return { missing: buckets.downloadedNoTranscript.filter((id) => !excluded.has(id)), autoSubs: buckets.downloadedAutoSubsOnly.filter((id) => !excluded.has(id)), }; } export type StationChannelWork = { // Whether the operation applies to this channel at all. eligible: boolean; // How much work its report says there is, or null when it cannot say (no // snapshot). Null is NOT zero: the figures above it are floors, and a click // must not skip a channel that never reported. work: number | null; // Why the channel is not eligible, for the skip readout. reason?: string; }; // The single per-channel reader every group figure and every group button goes // through. One derivation, so the label and the fan-out can never disagree. export function stationWorkFor( station: StationId, // The slug is part of the question now: whether an operation applies to a // channel is answered by the corpus-wide priority document as well as by the // channel's own config, and that document is keyed by slug. brief: Pick, settings: SiteSettings, ): StationChannelWork { const { slug, config, snapshot } = brief; if (laneOffFor(station, settings)) { return { eligible: false, work: 0, reason: "the lane is switched off" }; } const social = isSocialChannel(config); if (station === "sync") { // The same predicate "Sync every channel" applies. No figure: syncAction // decides per channel whether it is due, so there is no count to promise. if (!config.url) return { eligible: false, work: 0, reason: "no url" }; // THE PAUSED SECTION. The tier document is asked for the `sync` OPERATION // — `isChannelPaused(model, slug, "sync")` — which is precisely what the // deleted `excludeFromSync` flag meant, read the other way round, and is // now the only thing asked: S5 deleted the flag and migrated the 15 // channels that carried it. // `?? defaultChannelPriority()` for the same reason `isGateHeld` reaches // its key with optional chaining: this function is handed partial settings // objects by unit tests and by any caller that has not been through // `getSettings`, and an absent document means today's behaviour. const priority = settings.channelPriority ?? defaultChannelPriority(); if (isChannelPaused(priority, slug, "sync")) { return { eligible: false, work: 0, reason: "paused for sync" }; } return { eligible: true, work: 0 }; } if (social) { return { eligible: false, work: 0, reason: "social account" }; } if (station === "download") { if (!config.url) return { eligible: false, work: 0, reason: "no url" }; if (!snapshot) return { eligible: true, work: null }; const excluded = excludedDownloadIdSet(snapshot); return { eligible: true, work: (snapshot.undownloadedIds ?? []).filter((id) => !excluded.has(id)) .length, }; } if (station === "transcribe") { // HANDLING DOES NOT DECIDE THIS. Buckets are decided by FILES // (channelSnapshot.ts), and a `youtube`-handling channel whose video came // down with no captions at all is in `downloadedNoTranscript` exactly like a // `transcribe` one — the runner drains that bucket for every channel. What // the station counts is what its button queues: those, plus the videos whose // only transcript is YouTube's auto-captions and whose audio is on disk // (`downloadedAutoSubsOnly`, the channel page's replace-auto-captions half). // `transcribeStationIds` is the one fold; the group action runs off it too. if (!snapshot) return { eligible: true, work: null }; const { missing, autoSubs } = transcribeStationIds(snapshot); return { eligible: true, work: missing.length + autoSubs.length }; } if (station === "digest") { // Nothing to digest without transcripts — but a channel that has never // reported cannot claim it has none, so it stays eligible and unknown. if (!snapshot) return { eligible: true, work: null }; if (snapshot.totals.transcribed <= 0) { return { eligible: false, work: 0, reason: "no transcripts yet" }; } // digestWorkOf — the operation registry's entry is the one definition of // "digested". The `noDigest` bucket it replaced had no cues-staleness gate // and no transcript gate, so it called deferred and blocked videos done. return { eligible: true, work: digestWorkOf(snapshot).reachable }; } // Speakers — the backfill lane. backfillLaneEntriesOf, NEVER Object.values: the per-kind map carries // every catalog operation including digest, which has its own station right // beside this one. The lane filter is what keeps the two figures disjoint. if (!snapshot) return { eligible: true, work: null }; return { eligible: true, work: backfillLaneEntriesOf(snapshot.backfill).reduce( (n, e) => n + reachableOperationWork(e), 0, ), }; } // The groups a site's channels actually fall into, in render order. Mirrors the // bucketing MCP's list_channels does (handleListChannels), including its // stray-bucket fallback, so the two never disagree about where a channel lives. function bucketBySection( site: Site, stats: ReadonlyArray, ): { group: ChannelGroup; channels: ChannelStat[] }[] { const hasGroups = site.groups.length > 0; const order = new Map(); const groupOf = new Map(); for (const m of site.channels) { if (typeof m.order === "number") order.set(m.slug, m.order); groupOf.set(m.slug, resolveGroupIdFor(site, m.groupId, hasGroups)); } const byGroup = new Map(); for (const c of stats) { // A stat with no membership row cannot happen for a site-scoped list, but // fold it onto the default rather than dropping the channel. const gid = groupOf.get(c.slug) ?? resolveGroupIdFor(site, undefined, hasGroups); const bucket = byGroup.get(gid) ?? []; bucket.push(c); byGroup.set(gid, bucket); } const ordered = hasGroups ? sortGroups(site.groups) : [FALLBACK_GROUP]; const knownIds = new Set(ordered.map((g) => g.id)); const sections = ordered .filter((g) => (byGroup.get(g.id)?.length ?? 0) > 0) .map((group) => ({ group, channels: byGroup.get(group.id) ?? [] })); // Defensive: an id resolveChannelGroupId could not fold onto a real group // (e.g. a defaultGroupId naming a group that is no longer configured) gets a // synthetic section rather than having its channels silently disappear. for (const [gid, channels] of byGroup) { if (knownIds.has(gid)) continue; sections.push({ group: { ...FALLBACK_GROUP, id: gid }, channels }); } for (const s of sections) { s.channels.sort((a, b) => { const ao = order.get(a.slug) ?? Number.POSITIVE_INFINITY; const bo = order.get(b.slug) ?? Number.POSITIVE_INFINITY; if (ao !== bo) return ao - bo; return a.slug.localeCompare(b.slug); }); } return sections; } function resolveGroupIdFor( site: Site, groupId: string | undefined, hasGroups: boolean, ): string { return hasGroups ? resolveChannelGroupId(groupId, site.groups, site.defaultGroupId) : FALLBACK_GROUP.id; } export function buildChannelGroupSections( site: Site, stats: ReadonlyArray, briefs: ReadonlyArray, settings: SiteSettings, ): ChannelGroupSection[] { const briefBySlug = new Map(briefs.map((b) => [b.slug, b])); return bucketBySection(site, stats).map(({ group, channels }) => { const members = channels .map((c) => briefBySlug.get(c.slug)) .filter((b): b is ChannelBrief => !!b); const station = (id: StationId): StationWork => { const work: StationWork = { eligible: [], total: 0, unknown: [] }; if (laneOffFor(id, settings)) { work.laneOff = true; return work; } for (const b of members) { const w = stationWorkFor(id, b, settings); if (!w.eligible) continue; work.eligible.push(b.slug); if (w.work === null) work.unknown.push(b.slug); else work.total += w.work; } return work; }; return { group, channels, sync: station("sync"), download: station("download"), transcribe: station("transcribe"), digest: station("digest"), speakers: station("speakers"), speakersLabel: operationsGroupLabel( backfillLaneOperations(settings).map((k) => k.id), ), }; }); } // The server-action half of the same bucketing: which slugs are in this group. // Takes no snapshots, so a group button can never act on a different set than // the rows it sits above — and re-reading it at click time keeps it correct if // membership changed since the page rendered. export function slugsInGroup(site: Site, groupId: string): string[] { const hasGroups = site.groups.length > 0; const out: string[] = []; for (const m of site.channels) { if (resolveGroupIdFor(site, m.groupId, hasGroups) === groupId) { out.push(m.slug); } } return out; }