Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 4d1226bb1b5cf9ddac31dcdb12c9260ff02386fa
parent 8a955cde3e0169e4de307391227b6ec7c87e4059
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 02:43:44 -0400

Merge branch 'feat/filtered-channel-followups'

Diffstat:
Mcommon/controller/autoRunner.test.ts | 148+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/autoRunner.ts | 269++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Mcommon/controller/channelSets.test.ts | 39+++++++++++++++++++++++++++++++++++++++
Mcommon/controller/channelSets.ts | 16+++++++++++++++-
Mcommon/controller/channelSnapshot.test.ts | 323+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/channelSnapshot.ts | 106+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++----
Acommon/controller/metadataScanJob.ts | 80+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/controller/metadataScanStore.ts | 46++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/jobs/autoQueuePolicy.test.ts | 10+++++++---
Mcommon/jobs/autoQueuePolicy.ts | 25++++++++++++++++++++++++-
Mcommon/lib/channelConfig.ts | 31+++++++++++++++++++++++++++++++
Mcommon/lib/downloadFilters.test.ts | 89+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/downloadFilters.ts | 76++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mcommon/lib/downloadOutcome.ts | 18++++++++++++++++--
Mcommon/lib/operations.test.ts | 8+++++++-
Mcommon/lib/operations.ts | 22++++++++++++++++++----
Mcommon/lib/pauseGates.test.ts | 13++++++++-----
Mcommon/views/pipeline/channelFlow.ts | 11+++++++++++
Mcommon/views/pipeline/stageStatus.ts | 2++
Acommon/ytdlp/downloadOneManaged.test.ts | 140+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/ytdlp/downloadOneManaged.ts | 255++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Mcommon/ytdlp/runYtdlp.ts | 24+++++++++++++++++++++++-
Meditor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx | 21+++++++++++++++++++++
Meditor/app/channels/[slug]/page.tsx | 2++
Meditor/app/channels/[slug]/pipelineActions.ts | 27+++++++++++----------------
Meditor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx | 5+++++
Meditor/app/channels/components/ChannelForm.tsx | 21+++++++++++++++++++++
Meditor/app/channels/components/channelConfigToForm.ts | 9+++++++++
Meditor/app/channels/components/parseChannelForm.ts | 22++++++++++++++++++++++
Meditor/app/operations/components/InFlightList.tsx | 42++++++++++++++++++++++++++++++++----------
Meditor/e2e/auto-queue.spec.ts | 121++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Aeditor/e2e/chat-only.spec.ts | 379+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/bin/fake-ytdlp.mjs | 44++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/title-filter.spec.ts | 108++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mplans/FACTS.md | 174+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
35 files changed, 2646 insertions(+), 80 deletions(-)

diff --git a/common/controller/autoRunner.test.ts b/common/controller/autoRunner.test.ts @@ -7,6 +7,7 @@ import { focusHoldLine, laneDispatchRoot, makeFocusHoldReporter, + pickMetadataScanChannel, priorityContextFor, resetPriorityContextForTest, } from "./autoRunner"; @@ -21,6 +22,7 @@ import { LANES } from "../lib/autoQueueTypes"; import type { AutoQueueGroup, AutoQueuePolicy, + ChannelWork, } from "../jobs/autoQueuePolicy"; import type { Paths } from "../lib/paths"; import type { SiteSettings } from "../lib/settings"; @@ -331,3 +333,149 @@ test("the context is cached on the document, and a changed document re-resolves" "only-here", ]); }); + +// --------------------------------------------------------------------------- +// The download lane's metadata-scan pre-pick. +// +// It runs before every video pick on the lane, so its rule is the one thing +// standing between a filtered channel and 1,800 prefetch-reject-discard cycles +// against a source that is counting them. + +function work(over: Partial<ChannelWork> & { slug: string }): ChannelWork { + return { platform: "youtube", buckets: {}, ...over }; +} + +const noSkip = { + scanned: new Set<string>(), + platformSkip: new Set<string>(), + platformOf: () => "youtube", +}; + +test("a filtered channel with an unscanned backlog is the pick", () => { + const pick = pickMetadataScanChannel( + [ + work({ slug: "plain", scanUnscanned: 500, filtered: false }), + work({ slug: "filtered", scanUnscanned: 12, filtered: true }), + ], + noSkip, + ); + assert.deepEqual(pick, { slug: "filtered", targets: 12 }); +}); + +test("no filter, no scan — however big the backlog", () => { + // The scan's whole output is a verdict the filter makes. Without a filter it + // would fetch a title per listed video, decide nothing, and spend the + // source's patience doing it. + assert.equal( + pickMetadataScanChannel( + [work({ slug: "plain", scanUnscanned: 9999, filtered: false })], + noSkip, + ), + null, + ); +}); + +test("an empty backlog is not work", () => { + assert.equal( + pickMetadataScanChannel( + [work({ slug: "filtered", scanUnscanned: 0, filtered: true })], + noSkip, + ), + null, + ); + // And a projection with no scan fields at all — every lane but this one — + // offers nothing rather than throwing or defaulting to "yes". + assert.equal( + pickMetadataScanChannel([work({ slug: "filtered" })], noSkip), + null, + ); +}); + +test("once per runner: a channel already scanned this run is skipped", () => { + // A scan that left the backlog above zero was stopped by something — a soft + // block, a cooldown, a batch that ended early. Re-offering it three seconds + // later would hammer the source that just refused us. + assert.equal( + pickMetadataScanChannel( + [work({ slug: "filtered", scanUnscanned: 40, filtered: true })], + { ...noSkip, scanned: new Set(["filtered"]) }, + ), + null, + ); +}); + +test("a busy or cooling platform offers no scan either", () => { + // The same gate the downloads honour: the scan is one more thing asking that + // source for titles, and the per-platform cap is one at a time. + assert.equal( + pickMetadataScanChannel( + [work({ slug: "filtered", scanUnscanned: 40, filtered: true })], + { ...noSkip, platformSkip: new Set(["youtube"]) }, + ), + null, + ); + // A DIFFERENT platform is unaffected — the gate is per source, not global. + assert.deepEqual( + pickMetadataScanChannel( + [ + work({ slug: "yt", scanUnscanned: 40, filtered: true }), + work({ slug: "od", scanUnscanned: 7, filtered: true, platform: "odysee" }), + ], + { + scanned: new Set<string>(), + platformSkip: new Set(["youtube"]), + platformOf: (slug) => (slug === "od" ? "odysee" : "youtube"), + }, + ), + { slug: "od", targets: 7 }, + ); +}); + +test("the first candidate in list order wins, because that order is the priority", () => { + assert.deepEqual( + pickMetadataScanChannel( + [ + work({ slug: "first", scanUnscanned: 1, filtered: true }), + work({ slug: "second", scanUnscanned: 900, filtered: true }), + ], + noSkip, + ), + { slug: "first", targets: 1 }, + ); +}); + +test("a focus holds the scan the same way it holds every download pick", () => { + // The compiled tree makes a focus hold downloads by strict descent, and the + // pre-pick does not go through that tree. Without this, "only jeralyzer is + // moving" is false the moment another channel has titles to read — on the + // same network the focus is trying to have to itself. + const channels = [ + work({ slug: "other", scanUnscanned: 50, filtered: true }), + work({ slug: "focused", scanUnscanned: 4, filtered: true }), + ]; + assert.deepEqual( + pickMetadataScanChannel(channels, { + ...noSkip, + focus: { slugs: new Set(["focused"]), holding: true }, + }), + { slug: "focused", targets: 4 }, + ); + // Focus exhausted — strict descent moves on, and so does the scan. + assert.deepEqual( + pickMetadataScanChannel(channels, { + ...noSkip, + focus: { slugs: new Set(["focused"]), holding: false }, + }), + { slug: "other", targets: 50 }, + ); + // A holding focus with no scan work of its own offers nothing, rather than + // falling through to the channel it is holding back. + assert.equal( + pickMetadataScanChannel( + [work({ slug: "other", scanUnscanned: 50, filtered: true })], + { ...noSkip, focus: { slugs: new Set(["focused"]), holding: true } }, + ), + null, + ); +}); + diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts @@ -7,6 +7,8 @@ import { getSettings, type SiteSettings } from "../lib/settings"; import { diskGate } from "../lib/diskSpace"; import { formatBytes } from "../lib/format"; import { detectPlatform } from "../lib/platform"; +import { compileDownloadFilter } from "../lib/downloadFilters"; +import { runMetadataScanJob } from "./metadataScanJob"; import { getWorkerPool } from "../jobs/workerPool"; import { getRegistry } from "../jobs/registry"; import { runManagedFunction } from "../jobs/streamCommand"; @@ -160,6 +162,14 @@ const CHANNEL_LIST_TTL_MS = 30_000; // grow it unbounded. FIFO eviction; the snapshot catches up within ~1s anyway. const COMPLETED_CAP = 5000; +// THE DOWNLOAD LANE'S CHANNEL-SCOPED UNIT. A metadata scan has no video id, so +// it gets a synthetic one under a prefix nothing else can produce (a video id +// never contains a space) and a leaf id that exists in no tree — the pick is +// not made by `selectNextWork` and never reaches one. Both are exported so the +// console and the tests name the same strings the runner writes. +export const METADATA_SCAN_UNIT_PREFIX = "metadata-scan "; +export const METADATA_SCAN_LEAF_ID = "metadata-scan"; + // --- Live status (for /api/auto-queue/status), per kind -------------------- export type AutoRunnerInFlight = { @@ -167,6 +177,13 @@ export type AutoRunnerInFlight = { leafId: string; channelSlug: string; startedAt: number; + // A SENTENCE FOR A UNIT THAT IS NOT A VIDEO. `videoId` is the map key and + // therefore always set, but the download lane's metadata-scan unit is + // channel-scoped: its key is a synthetic `metadata-scan <slug>` that names no + // video directory. A console must not link it as one, so a unit carrying a + // note renders the note instead of a video link. Absent for every ordinary + // unit, which is how the surfaces stay unchanged. + note?: string; }; // Why the runner is up but dispatching nothing. Every one of these was already @@ -675,7 +692,21 @@ async function buildChannelWork( for (const id of ids) if (!owner.has(id)) owner.set(id, slug); } } - channels.push({ slug, platform, buckets, operations: ops }); + // THE CHANNEL-SCOPED HALF, off the snapshot already read. The download + // lane's pre-pick uses these two to decide whether to run a metadata scan + // for this channel before it downloads anything from it; every other lane + // ignores them. Both come out of `snap`/`config` with no extra I/O — the + // scan backlog is a number the snapshot already publishes, and the filter + // is the config listChannelConfigs already parsed. + const config = meta[i].config; + channels.push({ + slug, + platform, + buckets, + operations: ops, + scanUnscanned: snap.metadataScan?.unscanned ?? 0, + filtered: compileDownloadFilter(config.downloadFilter) !== null, + }); } return { channels, owner }; } @@ -1013,7 +1044,74 @@ function eligibleSlots(): number { // One unit of work selected by the policy, carried from the runPool `next` // (selection + reservation) into `run` (execution + release). -type Picked = { pick: WorkPick; channelSlug: string; unitPlatform: string | null }; +// WHICH CHANNEL GETS A METADATA SCAN THIS TICK, or none — the download lane's +// pre-pick, as a pure function over the projection the runner already built. +// +// Pure because the rule is the whole feature and everything else about it is +// plumbing. Three conditions, and each one is a bug if it goes: +// +// - A FILTER. Without one a scan settles nothing: it would fetch a title per +// listed video, change no verdict, and spend the source's patience to do +// it. `filtered` is "the channel's downloadFilter compiles", which is the +// same predicate `settledByTitleFilterIds` opens the store for. +// - A BACKLOG. `scanUnscanned` is the snapshot's own `metadataScan.unscanned`, +// defined to mean exactly what `metadataScanTargets()` will fetch, so a +// channel offered here is a channel the scan has work for. +// - ONCE PER RUNNER. A channel whose scan left the backlog above zero — a +// soft block, a cooldown, a batch that stopped early — is not retried by +// this runner. Re-offering it on the next three-second tick would hammer +// the source that just refused us; the operator's Run and the next restart +// both still work. +// +// The platform gate is the same one the downloads honour: a busy or cooling +// platform offers no scan either, because a scan is one more thing asking that +// source for titles. +// +// A FOCUS HOLDS THE SCAN TOO. The compiled priority tree makes a focus hold +// every download pick by strict descent (see laneDispatchRoot), and the pre-pick +// does not go through that tree — so without this, "only jeralyzer is moving" +// would be false the moment another channel had titles to read, on the same +// network the focus is trying to have to itself. The rule mirrors what strict +// descent does rather than re-implementing it: while the focus still has +// pending work, only a focused channel is offered a scan; once it is exhausted +// the lane is free and so is the scan. +// +// FIRST IN `channels` ORDER, which is `metaCache` order, which is the priority +// order every other pick on this lane already uses. +export function pickMetadataScanChannel( + channels: ReadonlyArray<ChannelWork>, + opts: { + scanned: ReadonlySet<string>; + platformSkip: ReadonlySet<string>; + platformOf: (slug: string) => string; + // The resolved focus, when one is holding the lane. Absent (or with + // `holding: false`) means every channel is eligible, which is the default + // and the byte-identical path for a corpus with no focus configured. + focus?: { slugs: ReadonlySet<string>; holding: boolean }; + }, +): { slug: string; targets: number } | null { + const held = opts.focus?.holding === true; + for (const c of channels) { + const targets = c.scanUnscanned ?? 0; + if (!c.filtered || targets <= 0) continue; + if (held && !opts.focus!.slugs.has(c.slug)) continue; + if (opts.scanned.has(c.slug)) continue; + if (opts.platformSkip.has(opts.platformOf(c.slug))) continue; + return { slug: c.slug, targets }; + } + return null; +} + +type Picked = { + pick: WorkPick; + channelSlug: string; + unitPlatform: string | null; + // THE CHANNEL-SCOPED UNIT. Set only by the download lane's metadata-scan + // pre-pick: `pick.videoId` is then a synthetic key naming no video and + // `pick.path` is empty, so no node's active count moves for it. `run()` + // branches on this and nothing else. + scan?: { slug: string; targets: number }; +}; async function runLoop( kind: AutoQueueKind, @@ -1111,6 +1209,13 @@ async function runLoop( // "unknown" for unrecognized hosts). Unused for transcription. const platformInFlight = new Map<string, number>(); const PER_PLATFORM_CAP = 1; + // Channels this runner has already scanned (or tried to). See the pre-pick in + // next() for why a scan is never retried inside one runner's lifetime. + const scannedThisRun = new Set<string>(); + // Platforms this tick may not touch — busy or cooling down. Hoisted out of the + // download branch because the metadata-scan pre-pick honours it too: a + // platform in a 429 cooldown must not be asked for 1,800 titles either. + const platformSkip = new Set<string>(); const platformKey = (slug: string, slugToPlatform: Map<string, string>) => slugToPlatform.get(slug) ?? "unknown"; @@ -1472,11 +1577,14 @@ async function runLoop( // Say — ONCE per transition — whether a focus is holding this lane. Read off // the `prio-*` leaf ids in the map just built, so it costs one pass over // keys and no new read, and it is skipped entirely while no focus resolves. - reportFocusHold( + // Computed ONCE and reused by the scan pre-pick below, which has to honour + // the same hold: `holding` is `focusPending > 0`, i.e. exactly "strict + // descent is still inside the focus group". + const focus = ctx.focusSlugs.length > 0 ? focusSummary(ctx.model, ctx.focusSlugs, pending) - : null, - ); + : null; + reportFocusHold(focus); // THE RUNNER JOB'S OWN BAR, on the metrics the per-channel jobs already // use — so a lane's runner row reads like the manual verb's row rather than // like an opaque long-lived loop. `target` moves as the corpus does (this @@ -1489,6 +1597,7 @@ async function runLoop( // download (busy) OR is in a rate-limit/network backoff window (cooling // down), so a busy/throttled platform yields to the next-priority free one. let anyCooling = false; + platformSkip.clear(); if (kind === "download") { // Merge in any cooldown a manual sync/import wrote to the shared state // (read-modify-write from outside the runner) since our last persist. @@ -1504,29 +1613,104 @@ async function runLoop( } catch { // Best-effort: a transient read failure just skips this iteration's merge. } - const skip = new Set<string>(); for (const [pf, n] of platformInFlight) { - if (n >= PER_PLATFORM_CAP) skip.add(pf); + if (n >= PER_PLATFORM_CAP) platformSkip.add(pf); } for (const pf of Object.keys(kindState.platformBackoff)) { if (isCoolingDown(kindState.platformBackoff, pf, now)) { - skip.add(pf); + platformSkip.add(pf); // A platform skipped for a cooldown is a different answer to "why is // it idle?" than one skipped for being busy, so the two are tracked // apart rather than both reading as "capped". anyCooling = true; } } - if (skip.size > 0) { + if (platformSkip.size > 0) { for (const leafId of Object.keys(pending)) { pending[leafId] = pending[leafId].filter((id) => { const slug = owner.get(id); - return !(slug && skip.has(platformKey(slug, slugToPlatform))); + return !(slug && platformSkip.has(platformKey(slug, slugToPlatform))); }); } } } + // ── THE SCAN COMES BEFORE THE DOWNLOADS ──────────────────────────────── + // + // A filtered channel with unscanned listed videos is a channel whose + // download queue is WRONG, not merely incomplete: every non-matching video + // in it will be prefetched, rejected and discarded one at a time, at one + // yt-dlp invocation each, against a source whose patience is the scarce + // resource. The scan answers the same question for the whole channel in one + // batch. So it is not another kind of work competing with downloads — it is + // the thing that decides what the downloads ARE, and it goes first. + // + // ONE PER CHANNEL PER RUNNER, and never a second while one is in flight. + // `scannedThisRun` is the session's memory: a channel whose scan left the + // backlog above zero (a soft block, a cooldown, a batch that stopped early) + // is NOT retried by this runner — the operator's Run button and the next + // restart both still work, and re-offering it on the next three-second tick + // would hammer the very source that just refused us. + // + // It takes a platform slot exactly as a download unit does, so a scan and a + // download never hit one source at once, and it inherits the platform + // cooldown for free: a platform in `platformSkip` offers no scan either. + if (kind === "download") { + const scanCandidate = pickMetadataScanChannel(channels, { + scanned: scannedThisRun, + platformSkip, + platformOf: (slug) => platformKey(slug, slugToPlatform), + ...(focus + ? { + focus: { + slugs: new Set(ctx.focusSlugs), + holding: focus.holding, + }, + } + : {}), + }); + if (scanCandidate) { + const { slug, targets } = scanCandidate; + const unitPlatform = platformKey(slug, slugToPlatform); + scannedThisRun.add(slug); + const videoId = `${METADATA_SCAN_UNIT_PREFIX}${slug}`; + const note = `scanning ${targets} listed video${ + targets === 1 ? "" : "s" + } for ${slug}`; + live.idleReason = null; + // No `path`, so no node's active count moves: the scan is not a leaf's + // work and must not consume a leaf's or a group's maxWorkers. + live.inFlight.set(videoId, { + videoId, + leafId: METADATA_SCAN_LEAF_ID, + channelSlug: slug, + startedAt: Date.now(), + note, + }); + platformInFlight.set( + unitPlatform, + (platformInFlight.get(unitPlatform) ?? 0) + 1, + ); + // IN THE PICK LOG LIKE ANY OTHER UNIT. The log is "what did this lane + // do", and a tick that ran a scan and no download would otherwise read + // as a tick that did nothing at all. + recordPick(kindState, { + at: Date.now(), + leafId: METADATA_SCAN_LEAF_ID, + videoId, + channelSlug: slug, + }); + persist(); + onLog(`Auto-download: ${note}.`); + return { + pick: { leafId: METADATA_SCAN_LEAF_ID, videoId, path: [] }, + channelSlug: slug, + unitPlatform, + scan: { slug, targets }, + }; + } + } + const pick = selectNextWork(root, pending, runtime, live.active); if (!pick) { // Attribute the idleness. Nothing pending at all is a different situation @@ -1682,6 +1866,17 @@ async function runLoop( result = { outcome: "skipped" }; return; } + if (picked.scan) { + result = await launchMetadataScan({ + paths, + slug: picked.scan.slug, + targets: picked.scan.targets, + tracker, + onLog, + onChildJob: (jid) => childJobIds.set(pick.videoId, jid), + }); + return; + } result = isOperationLane(kind) ? await runOperationPick(picked, runSignal) : await launchUnit({ @@ -1776,6 +1971,55 @@ async function runLoop( ); } +// ONE METADATA SCAN, as a child job on the channel's platform download queue. +// +// The SAME job the operator's Run button dispatches — kind, queue key, replay +// spec and the shared per-platform cooldown all come from +// controller/metadataScanJob.ts — so the registry serializes it against a +// manual sync on that platform exactly as it serializes a download unit, and +// there is one definition of what a scan IS. +// +// `background: true` for the same reason a download unit is: a clicked action +// preempts queued units without interrupting a running one. +// +// A SCAN IS NEVER A FAILURE FOR BACKOFF PURPOSES. runMetadataScan does not +// throw on a rate limit — it stops, records the shared cooldown through its own +// onPlatformBackoff, and keeps what it flushed. So this returns "skipped" or +// "transcribed" and never a `failureClass`: the cooldown is already recorded by +// the time we get here, and a second backoff entry from the runner would +// double-count one refusal. +async function launchMetadataScan(args: { + paths: Paths; + slug: string; + targets: number; + tracker: ReturnType<typeof makeTaskTracker>; + onLog: (line: string) => void; + onChildJob?: (jobId: string) => void; +}): Promise<UnitResult> { + const config = await readChannelConfig(args.paths, args.slug); + if (!config) return { outcome: "skipped" }; + const res = await runMetadataScanJob({ + paths: args.paths, + slug: args.slug, + channelConfig: config, + background: true, + }); + if (!res.ok) { + args.onLog( + `Auto-download: could not enqueue the metadata scan for ${args.slug}: ${res.error}`, + ); + return { outcome: "skipped" }; + } + args.onChildJob?.(res.jobId); + const term = await res.done; + if (term.status === "cancelled") return { outcome: "skipped" }; + if (term.status === "failed") { + args.onLog(`Auto-download: the metadata scan for ${args.slug} failed.`); + return { outcome: "skipped" }; + } + return { outcome: "transcribed" }; +} + type LaunchArgs = { kind: AutoQueueKind; paths: Paths; @@ -1978,6 +2222,11 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> { // Cancelled (runner stopped, or dropped while still queued) → not a failure. if (term.status === "cancelled") return { outcome: "skipped" }; if (unitStatus === "skipped-filtered") return { outcome: "skipped" }; + // The filter declined the media and the live chat was fetched instead. A + // SUCCESS for the lane — the unit did the work it was picked for, the video + // leaves chatOnlyPending, and a platform whose chat pass came back clean has + // its cooldown cleared exactly as a download would clear it. + if (unitStatus === "chat-only") return { outcome: "transcribed" }; // Complete-but-malformed source: terminal and kept on disk. Treat as skipped // (not failed) so it doesn't drive backoff and isn't re-picked for download. if (unitStatus === "corrupt-full-source") return { outcome: "skipped" }; diff --git a/common/controller/channelSets.test.ts b/common/controller/channelSets.test.ts @@ -21,11 +21,13 @@ function derive(input: { roster: ReadonlyArray<string>; listed: ReadonlyArray<string>; onDisk: ReadonlyArray<string>; + settled?: ReadonlyArray<string>; }) { return deriveChannelSets({ roster: rosterOf(input.roster), listedIds: new Set(input.listed), onDiskIds: new Set(input.onDisk), + ...(input.settled ? { settledIds: new Set(input.settled) } : {}), }); } @@ -111,3 +113,40 @@ test("an empty channel derives empty sets rather than throwing", () => { orphaned: [], }); }); + +test("a settled video is neither never-fetched nor undownloaded", () => { + // BOTH SETS ARE DEFINED BY THE ABSENCE OF A DIRECTORY, and a video the + // download filter settled has none — a title-filter rejection deletes its own + // prefetch dir. So without subtracting the settled set, a filtered-out video + // that later leaves the listing reads as "we were told about this, never got + // it, and it is gone", which is the loudest alarm this system raises and the + // sweep prints a recovery prompt for it. + const sets = derive({ + roster: ["keep", "drop"], + listed: ["keep"], + onDisk: [], + settled: ["drop"], + }); + assert.deepEqual(sets.missingNeverFetched, []); + assert.deepEqual(sets.undownloaded, ["keep"]); + // And the control: the same corpus with nothing settled still raises it, so + // the subtraction has not made the category unreachable. + const unsettled = derive({ + roster: ["keep", "drop"], + listed: ["keep"], + onDisk: [], + }); + assert.deepEqual(unsettled.missingNeverFetched, ["drop"]); +}); + +test("an omitted settled set changes nothing, for every channel without a filter", () => { + const withEmpty = derive({ + roster: ["a", "b"], + listed: ["a"], + onDisk: [], + settled: [], + }); + const without = derive({ roster: ["a", "b"], listed: ["a"], onDisk: [] }); + assert.deepEqual(withEmpty, without); +}); + diff --git a/common/controller/channelSets.ts b/common/controller/channelSets.ts @@ -37,15 +37,28 @@ export function deriveChannelSets(input: { roster: Roster; listedIds: ReadonlySet<string>; onDiskIds: ReadonlySet<string>; + // Ids this channel's download filter has SETTLED — the operator's own "not + // this one". Optional, and empty for every channel without a filter. + // + // WHY IT BELONGS HERE. Both sets below are defined by the ABSENCE of a + // directory, and since a title-filter rejection stopped leaving its prefetch + // dir behind, a settled video has none. Without this, a filtered-out video + // that later left the listing read as `missingNeverFetched` — "we were told + // about this, never got it, and now it is gone" — which is the highest-value + // alarm this system raises, pointed at a video the operator asked us not to + // fetch. `undownloaded` has the same shape of wrongness: it is a work list, + // and settled work is not work. + settledIds?: ReadonlySet<string>; }): ChannelSets { const { roster, listedIds, onDiskIds } = input; + const settledIds = input.settledIds ?? new Set<string>(); const rosterIds = Object.keys(roster.entries); const listed: string[] = []; const undownloaded: string[] = []; for (const id of listedIds) { listed.push(id); - if (!onDiskIds.has(id)) undownloaded.push(id); + if (!onDiskIds.has(id) && !settledIds.has(id)) undownloaded.push(id); } // missingDownloaded is derived from the DISK set, not from `roster ∩ disk`, @@ -64,6 +77,7 @@ export function deriveChannelSets(input: { const missingNeverFetched: string[] = []; for (const id of rosterIds) { + if (settledIds.has(id)) continue; if (!listedIds.has(id) && !onDiskIds.has(id)) missingNeverFetched.push(id); } diff --git a/common/controller/channelSnapshot.test.ts b/common/controller/channelSnapshot.test.ts @@ -358,3 +358,326 @@ test("generateChannelSnapshot refuses an unreachable channel rather than writing await rm(dir, { recursive: true, force: true }); } }); + +// --------------------------------------------------------------------------- +// The filtered channel, end to end through generateChannelSnapshot. +// +// It needs nothing but a directory: a config, a playlist and the metadata-scan +// store. That matters here because the whole point of the two facts below is +// that they are derived from the STORE and not from anything on a video's +// disk — a claim only a run with no video dirs at all can actually pin. + +type FilterFixture = { + filter?: Record<string, unknown>; + listed: string[]; + scanned?: Record< + string, + { title?: string; liveStatus?: string; noLiveChat?: boolean } + >; + // Ids the channel has EVER been seen to contain. Seeded so the + // missingNeverFetched rule — "we were told about this, never got it, and it + // is gone" — can be exercised: it is derived from the roster, not the disk. + roster?: string[]; +}; + +async function filteredChannel( + fixture: FilterFixture, +): Promise<{ dir: string; paths: Paths; channelDir: string }> { + const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-filter-")); + const paths = { + channelsDir: path.join(dir, "channels"), + transcriptsDir: dir, + savedVideosDir: path.join(dir, "saved-videos"), + } as Paths; + const channelDir = path.join(paths.channelsDir, "alpha"); + await mkdir(path.join(channelDir, "data"), { recursive: true }); + await writeFile( + path.join(channelDir, "config.json"), + JSON.stringify({ + handling: "youtube", + url: "https://www.youtube.com/@alpha/videos", + ...(fixture.filter ? { downloadFilter: fixture.filter } : {}), + }), + ); + await writeFile( + path.join(channelDir, "playlist"), + fixture.listed + .map((id) => `https://www.youtube.com/watch?v=${id}`) + .join("\n"), + ); + const entries: Record<string, unknown> = {}; + for (const [id, e] of Object.entries(fixture.scanned ?? {})) { + entries[id] = { + title: e.title ?? `Synthetic ${id}`, + description: "", + uploadDate: "20240101", + ...(e.liveStatus ? { liveStatus: e.liveStatus } : {}), + ...(e.noLiveChat ? { noLiveChat: true } : {}), + scannedAt: "2026-01-01T00:00:00.000Z", + }; + } + if (fixture.roster) { + await writeFile( + path.join(channelDir, "roster.json"), + JSON.stringify({ + version: 1, + entries: Object.fromEntries( + fixture.roster.map((id) => [ + id, + { + url: `https://www.youtube.com/watch?v=${id}`, + firstSeenAt: "2026-01-01T00:00:00.000Z", + }, + ]), + ), + }), + ); + } + await writeFile( + path.join(channelDir, "metadata-scan.json"), + JSON.stringify({ version: 1, entries, errors: {}, lastRun: null }), + ); + return { dir, paths, channelDir }; +} + +test("a settled video is in skippedByTitleFilter with NO directory of its own", async () => { + // THE BUCKET IS SOURCED FROM THE SCAN STORE, not from a download-outcome + // sidecar — which is what lets a title-filter rejection delete the prefetch + // directory it made (ytdlp/downloadOneManaged.ts, discardPrefetchDir) without + // the count moving. This fixture has no data/ dirs at all; if the bucket + // depended on one, it would be empty here. + const { dir, paths } = await filteredChannel({ + filter: { include: "keep" }, + listed: ["aaaa0000001", "bbbb0000002"], + scanned: { + aaaa0000001: { title: "keep this one" }, + bbbb0000002: { title: "drop this one" }, + }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); + // And it is out of the download lane's queue, which is the point of it. + assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); + // A settled id was never a video: it has no directory, so it was never in + // totals.videos either. + assert.equal(snap.totals.videos, 0); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a chat-only livestream is a corpus member, not a settled stub", async () => { + // THE RISK THIS PINS. A chat-only dir makes "metadata, no transcript" a + // LEGITIMATE shape — which is exactly what buildIndex and deriveChannelSets + // read as "this video was fetched". If the bucket rules are wrong, filtered + // livestreams either vanish from the report or reappear as download work + // forever. + const { dir, paths, channelDir } = await filteredChannel({ + filter: { include: "keep", rejectedLivestreams: "chat-only" }, + listed: ["aaaa0000001", "bbbb0000002", "cccc0000003"], + scanned: { + aaaa0000001: { title: "keep this one" }, + bbbb0000002: { title: "drop this one" }, + cccc0000003: { title: "a long stream", liveStatus: "was_live" }, + }, + }); + try { + // The chat has landed for the livestream: metadata + the raw chat, and + // nothing else on disk. + const videoDir = path.join(channelDir, "data", "cccc0000003"); + await mkdir(videoDir, { recursive: true }); + await writeFile( + path.join(videoDir, "metadata.info.json"), + JSON.stringify({ id: "cccc0000003", title: "a long stream" }), + ); + await writeFile( + path.join(videoDir, "transcript.live_chat.json"), + '{"action":{}}\n', + ); + + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]); + assert.deepEqual(snap.buckets.chatOnlyPending, []); + // NOT "we decided not to have it". + assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); + // Nothing will transcribe a video whose audio we chose not to fetch, and + // nothing will download it. + assert.ok(!snap.buckets.noTranscript.includes("cccc0000003")); + assert.ok(!snap.buckets.downloadedNoTranscript.includes("cccc0000003")); + assert.ok(!snap.undownloadedIds.includes("cccc0000003")); + assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); + // A video we deliberately have. `settledOnDisk` is subtracted from this and + // a chat-only dir must never be in it. + assert.equal(snap.totals.videos, 1); + assert.equal(snap.totals.downloaded, 0); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("before the chat lands it is chatOnlyPending — the download lane's bucket", async () => { + const { dir, paths } = await filteredChannel({ + filter: { include: "keep", rejectedLivestreams: "chat-only" }, + listed: ["aaaa0000001", "cccc0000003"], + scanned: { + aaaa0000001: { title: "keep this one" }, + cccc0000003: { title: "a long stream", liveStatus: "was_live" }, + }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.chatOnlyPending, ["cccc0000003"]); + assert.deepEqual(snap.buckets.chatOnly, []); + // It cannot ride in undownloadedIds: that list means "fetch the media", and + // fetching the media is the one thing this video must not have done to it. + assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); + assert.deepEqual(snap.buckets.skippedByTitleFilter, []); + // It IS the download lane's work, through the lane work list every lane + // reads its dispatch from. + assert.ok(snap.backfill?.download?.ids.includes("cccc0000003")); + // And LAST in it, behind the real download — the fold walks + // DOWNLOAD_BUCKETS in order. + const ids = snap.backfill?.download?.ids ?? []; + assert.equal(ids[ids.length - 1], "cccc0000003"); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("turning the mode off re-decides the channel with nothing to migrate", async () => { + // Same store, same disk, no filter field: the livestream is an ordinary + // settled rejection again and the chat-only buckets are empty. Nothing about + // the verdict was ever stored per video. + const { dir, paths } = await filteredChannel({ + filter: { include: "keep" }, + listed: ["cccc0000003"], + scanned: { + cccc0000003: { title: "a long stream", liveStatus: "was_live" }, + }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.chatOnly, []); + assert.deepEqual(snap.buckets.chatOnlyPending, []); + assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a settled video that left the listing is NOT reported as never fetched", async () => { + // THE LOUDEST ALARM THIS REPORT RAISES, pointed at the wrong video. Both + // missingNeverFetched and undownloaded are defined by the ABSENCE of a + // directory, and a settled video has none since a rejection stopped leaving + // its prefetch dir behind — so without subtracting the settled set, the sweep + // tells the operator to attempt a direct-link recovery of a video they + // configured us not to fetch. + const { dir, paths } = await filteredChannel({ + filter: { include: "keep" }, + // `bbbb0000002` is in the roster and NOT in the listing any more. + listed: ["aaaa0000001"], + roster: ["aaaa0000001", "bbbb0000002"], + scanned: { + aaaa0000001: { title: "keep this one" }, + bbbb0000002: { title: "drop this one" }, + }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual( + (snap.missingNeverFetched ?? []).map((m) => m.id), + [], + ); + // It is still settled, and still reported as such — it left the alarm, not + // the report. + assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("an unsettled video that left the listing still raises the alarm", async () => { + // The control for the test above: subtracting the settled set must not have + // made the category unreachable. + const { dir, paths } = await filteredChannel({ + filter: { include: "keep" }, + listed: [], + roster: ["aaaa0000001"], + scanned: { aaaa0000001: { title: "keep this one" } }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual( + (snap.missingNeverFetched ?? []).map((m) => m.id), + ["aaaa0000001"], + ); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a stream with no chat replay leaves chatOnlyPending for good", async () => { + // A clean pass that found nothing is an ANSWER: chat replay was off, and + // asking again tomorrow gets the same nothing. Without the flag the id sits + // in chatOnlyPending forever and every runner restart re-prefetches it to run + // a pass that will never return anything. + const { dir, paths } = await filteredChannel({ + filter: { include: "keep", rejectedLivestreams: "chat-only" }, + listed: ["cccc0000003"], + scanned: { + cccc0000003: { + title: "a long stream", + liveStatus: "was_live", + noLiveChat: true, + }, + }, + }); + try { + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.chatOnlyPending, []); + assert.deepEqual(snap.buckets.chatOnly, []); + // It settles as the ordinary filtered-out livestream it is. + assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]); + assert.ok(!snap.undownloadedIds.includes("cccc0000003")); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + +test("a chat already on disk stays a member after the mode is turned off", async () => { + // buildIndex has already published it as a chat track, and flipping the mode + // back to "skip" does not un-publish it. Deciding this off `chatOnlyIds` + // would make the same directory a settled STUB — and settledOnDisk is + // subtracted from totals.videos, so the site would serve a video the report + // had stopped counting. + const { dir, paths, channelDir } = await filteredChannel({ + filter: { include: "keep" }, + listed: ["cccc0000003"], + scanned: { + cccc0000003: { title: "a long stream", liveStatus: "was_live" }, + }, + }); + try { + const videoDir = path.join(channelDir, "data", "cccc0000003"); + await mkdir(videoDir, { recursive: true }); + await writeFile( + path.join(videoDir, "metadata.info.json"), + JSON.stringify({ id: "cccc0000003", title: "a long stream" }), + ); + await writeFile( + path.join(videoDir, "transcript.live_chat.json"), + '{"action":{}}\n', + ); + + const snap = await generateChannelSnapshot(paths, "alpha"); + assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]); + assert.deepEqual(snap.buckets.skippedByTitleFilter, []); + assert.equal(snap.totals.videos, 1); + // And nothing outstanding: the mode is off, so there is no chat to fetch. + assert.deepEqual(snap.buckets.chatOnlyPending, []); + } finally { + await rm(dir, { recursive: true, force: true }); + } +}); + diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -8,6 +8,7 @@ import { isVideoFetched, isVideoTranscribed, readVideoFiles, + LIVE_CHAT_FILENAME, VTT_FILENAME, type VideoFiles, } from "../lib/videoStatus"; @@ -48,6 +49,7 @@ import { import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server"; import { loadDownloadOutcome } from "../lib/downloadOutcome-server"; import { + chatOnlyIdsFrom, loadMetadataScan, metadataScanWanted, settledIdsFrom, @@ -213,6 +215,32 @@ export type ChannelSnapshot = { // re-appear as ordinary undownloaded videos on the next report. // Optional: older snapshots lack it; readers must default to []. skippedByTitleFilter: string[]; + // CHAT-ONLY videos whose chat is on disk. A rejected livestream on a channel + // whose `downloadFilter.rejectedLivestreams` is "chat-only": the media was + // never fetched, the live chat was, and the directory holds + // metadata.info.json, transcript.live_chat.json and its normalized + // live_chat.cues.json, beside the download-outcome.json and download.log + // every managed download leaves. No media, and no transcript. + // + // THE FILE ON DISK DECIDES, not the mode: an id whose chat has landed stays + // in this bucket after `rejectedLivestreams` is set back to "skip", + // because buildIndex has already published it and flipping a setting does + // not un-publish it. + // + // A MEMBER OF THIS CORPUS, not a stub. It is in totals.videos, it is in the + // LMDB index, and the site publishes it as a chat track with no captions. + // It is deliberately NOT in skippedByTitleFilter (that bucket means "we + // decided not to have it"), NOT in noTranscript or downloadedNoTranscript + // (nothing is going to transcribe a video whose audio we chose not to + // fetch), and NOT in undownloadedIds. Optional: older snapshots lack it; + // readers must default to [] — which normalizeBuckets does, like every + // other bucket declared required here and absent from older files. + chatOnly: string[]; + // The same population, chat NOT yet on disk — one of the download lane's + // buckets (DOWNLOAD_BUCKETS). It cannot ride in undownloadedIds: that list + // means "fetch the media", and fetching the media is the one thing this + // video must not have done to it. Absent from older files in the same way. + chatOnlyPending: string[]; // Videos whose transcript covers only a small fraction of the video's // duration — the audio download silently truncated (yt-dlp exited "ok") so // whisper transcribed just the first few minutes. The detection threshold @@ -638,9 +666,14 @@ async function readPlaylistUrls(file: string): Promise<string[]> { } } -function videoHasAnyArtifact(files: VideoFiles): boolean { - return files.hasYtVtt || files.hasWhisper || files.audioFiles.length > 0; -} +// WAS A PRIVATE COPY OF isVideoDownloaded, byte for byte, and this module +// already imported that one (it is what increments `downloaded` in the same +// walk). Two names for one predicate is how the settled short-circuit and the +// totals end up disagreeing about what an artifact is, so the copy is gone and +// the name stays as an alias: "does this dir hold anything we fetched?" reads +// better at the five call sites than "is it downloaded?", and it is now +// guaranteed to be the same question. +const videoHasAnyArtifact = isVideoDownloaded; export async function generateChannelSnapshot( paths: Paths, @@ -925,6 +958,11 @@ export async function generateChannelSnapshot( // whole channel on the next report, with no rescan and nothing to migrate. const metadataScanStore = await loadMetadataScan(paths, slug); const settledIds = settledIdsFrom(metadataScanStore, config); + // A STRICT SUBSET of the settled set: the rejections the operator asked for + // the live chat of (downloadFilter.rejectedLivestreams === "chat-only"). + // Settled means the MEDIA is not wanted, which is true of these too — what + // this adds is that the video is still a corpus member, as a chat track. + const chatOnlyIds = chatOnlyIdsFrom(metadataScanStore, config); const noTranscript: string[] = []; const downloadedNoTranscript: string[] = []; @@ -942,6 +980,12 @@ export async function generateChannelSnapshot( // their metadata before deciding). Tracked separately from the full settled // set only so totals.videos can subtract exactly the dirs it counted. const settledOnDisk: string[] = []; + // Chat-only videos whose chat is on disk — corpus members, counted in + // totals.videos, and out of every bucket that means "something is missing". + const chatOnly: string[] = []; + // Chat-only videos whose chat is NOT on disk yet: the download lane's work, + // and the reason this is a bucket rather than a derived count. + const chatOnlyPending: string[] = []; const incompleteTranscript: string[] = []; const shortAudio: string[] = []; const autoSubsOnly: string[] = []; @@ -1079,8 +1123,27 @@ export async function generateChannelSnapshot( // it from transcribedWithAudio and the cleanup estimates while it still // counted in totals.transcribed, i.e. a channel reporting more transcripts // than videos. Settlement only ever decides what NOT to fetch. + // SETTLED, and the two answers it can have. + // + // THE CHAT ON DISK DECIDES, NOT THE MODE. A dir holding + // transcript.live_chat.json is already published by buildIndex as a chat + // track with no captions — that is a fact about the corpus, and flipping + // `rejectedLivestreams` back to "skip" does not un-publish it. Asking + // `chatOnlyIds` here instead would make the same directory count as a + // settled STUB the moment the mode changed, and settledOnDisk is subtracted + // from totals.videos: the site would still serve the video while the report + // stopped counting it. So the file is the test, and the Diagnostics copy — + // "what is already on disk stays" — is true. + // + // Short-circuited for the same reason the settled branch is: everything + // under this line classifies a video by what is MISSING, and nothing is + // missing from a chat-only video. if (settledIds.has(id) && !videoHasAnyArtifact(files)) { - settledOnDisk.push(id); + if (files.entries.includes(LIVE_CHAT_FILENAME)) { + chatOnly.push(id); + } else { + settledOnDisk.push(id); + } continue; } if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id); @@ -1275,6 +1338,12 @@ export async function generateChannelSnapshot( urls.map((u) => extractVideoId(u)).filter((id): id is string => Boolean(id)), ), onDiskIds: new Set(videoDirNames), + // See deriveChannelSets: both sets it derives are defined by the ABSENCE of + // a directory, and a settled video has none since a rejection stopped + // leaving its prefetch dir behind. Without this, `missingNeverFetched` — + // the loudest alarm this report raises — would fire for videos the operator + // asked us not to fetch. + settledIds, }); const listedIdSet = new Set(sets.listed); @@ -1307,6 +1376,19 @@ export async function generateChannelSnapshot( ) { metadataScanUnscanned++; } + // CHAT ONLY: the media is settled (so this id never reaches + // undownloadedIds) but the CHAT may still be outstanding, and that is real + // download-lane work with no other home — the id has no artifact, so no + // artifact-derived bucket can carry it. Once the chat lands it drops out + // here and appears in `chatOnly` instead. + if (chatOnlyIds.has(dirId)) { + // Already a member (the chat is on disk) => nothing outstanding. The scan + // store's `noLiveChat` flag is the other way out of this list: a stream + // we asked and got nothing from is not in `chatOnlyIds` at all, so it + // settles as an ordinary rejection instead of sitting here forever. + if (!f?.entries.includes(LIVE_CHAT_FILENAME)) chatOnlyPending.push(dirId); + continue; + } // A settled video has no artifact and never will while the filter stands. // This is the line that makes the settlement STICK: undownloadedIds is the // auto-download runner's work queue, and it is derived from artifacts, so an @@ -1352,6 +1434,7 @@ export async function generateChannelSnapshot( const corruptSourceSet = new Set(corruptSource); const corruptFullSourceSet = new Set(corruptFullSource); + const chatOnlySet = new Set(chatOnly); const snapshotBuckets: ChannelSnapshot["buckets"] = { noTranscript: noTranscript.sort(), downloadedNoTranscript: downloadedNoTranscript.sort(), @@ -1374,13 +1457,26 @@ export async function generateChannelSnapshot( nonStandardVtt: nonStandardVtt.sort(), skippedByFilter: skippedByFilter.sort(), // The settled set minus anything already on disk — same rule as the - // short-circuit above, so the bucket and the classification cannot disagree. + // short-circuit above, so the bucket and the classification cannot disagree + // — and minus the chat-only subset, which has its own two buckets. A + // chat-only video IS settled (its media is not wanted) but reporting it as + // "skipped by the title filter" would say we decided not to have it, when + // we decided to have its chat. skippedByTitleFilter: [...settledIds] .filter((id) => { + // Minus the two chat buckets, whichever way an id got into them — a + // chat-only video IS settled (its media is not wanted) but reporting it + // as "skipped by the title filter" would say we decided not to have it, + // when we decided to have its chat. `chatOnlySet` is what the loop + // above actually classified (the chat is on disk); `chatOnlyIds` is the + // outstanding half. + if (chatOnlySet.has(id) || chatOnlyIds.has(id)) return false; const f = filesById.get(id); return !f || !videoHasAnyArtifact(f); }) .sort(), + chatOnly: chatOnly.sort(), + chatOnlyPending: chatOnlyPending.sort(), incompleteTranscript: incompleteTranscript.sort(), shortAudio: shortAudio.sort(), autoSubsOnly: autoSubsOnly.sort(), diff --git a/common/controller/metadataScanJob.ts b/common/controller/metadataScanJob.ts @@ -0,0 +1,80 @@ +// THE METADATA SCAN AS A JOB — one definition, two dispatchers. +// +// The scan itself is `ytdlp/metadataScan.ts`, which is pure work: it takes a +// channel and a config and reads titles. What it is NOT is a job — the kind, +// the queue key, the replay spec and the platform cooldown that surrounds it +// all lived in the editor's server action, which is the one place the auto +// runner cannot reach. +// +// So the JOB moves here and the editor action keeps what is genuinely its +// own: the cooldown pre-check it answers to a click with (a returned +// `{ ok: false, info: true }` sentence, not a queued job nobody watches) and +// its three `revalidatePath` calls. The runner passes neither. What both get +// from this module is identical: the same `kind`, the same per-platform queue +// key, the same replay spec, and the same `onPlatformBackoff` wiring into the +// SHARED cooldown the download lane already honours — which is what stops the +// runner and a manual sync from arguing about whether YouTube is angry. +// +// `needsMedia` comes from the kind (jobs/jobKinds.ts declares it TRUE for this +// one, because the scan's target set is "listed, minus what is already on +// disk"), so `runManagedFunction` refuses the job against an unmounted drive +// for either caller with nothing said here. + +import { detectPlatform } from "../lib/platform"; +import { downloadQueueKey, resolveQueueKey } from "../lib/queueKeys"; +import { recordDownloadBackoff } from "../jobs/downloadBackoff"; +import { + runManagedFunction, + type StreamActionResult, +} from "../jobs/streamCommand"; +import type { ChannelConfig } from "../lib/channelConfig"; +import type { Paths } from "../lib/paths"; +import { runMetadataScan } from "../ytdlp/metadataScan"; + +export const METADATA_SCAN_JOB_KIND = "metadata-scan"; + +export type MetadataScanJobOpts = { + paths: Paths; + slug: string; + channelConfig: ChannelConfig; + // Per-run override of the platform queue key (the editor threads one through + // from the pipeline action's own queue plumbing). + queueKey?: string; + // Run in the background so a clicked action preempts it in the queue. The + // runner passes true for the same reason its download units do. + background?: boolean; + // Called after a successful scan, inside the job. The editor revalidates its + // three paths here; the runner asks for a snapshot regen instead. + afterRun?: () => void | Promise<void>; +}; + +export async function runMetadataScanJob( + opts: MetadataScanJobOpts, +): Promise<StreamActionResult> { + const { paths, slug, channelConfig } = opts; + const platform = detectPlatform(channelConfig.url) ?? "unknown"; + return runManagedFunction({ + kind: METADATA_SCAN_JOB_KIND, + queueKey: resolveQueueKey(downloadQueueKey(channelConfig), opts.queueKey), + paths, + channelSlug: slug, + ...(opts.background ? { background: true } : {}), + spec: { + kind: METADATA_SCAN_JOB_KIND, + slug, + params: { queueKey: opts.queueKey }, + }, + fn: async (onLog, signal, setProgress) => { + await runMetadataScan({ + paths, + channelSlug: slug, + channelConfig, + onLog, + signal, + setProgress, + onPlatformBackoff: () => recordDownloadBackoff(platform, paths), + }); + await opts.afterRun?.(); + }, + }); +} diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts @@ -30,7 +30,11 @@ import path from "node:path"; import { readFile, rename, writeFile } from "node:fs/promises"; import type { Paths } from "../lib/paths"; import type { ChannelConfig } from "../lib/channelConfig"; -import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters"; +import { + compileDownloadFilter, + titleFilterRejects, + titleFilterWantsChat, +} from "../lib/downloadFilters"; export const METADATA_SCAN_FILENAME = "metadata-scan.json"; export const METADATA_SCAN_VERSION = 1; @@ -45,6 +49,14 @@ export type MetadataScanEntry = { uploadDate: string; liveStatus?: string; duration?: number; + // THE CHAT-ONLY DEAD END. Set when a chat-only pass ran cleanly and the + // source returned no live chat — replay was off for that stream, or it has + // since been dropped. Asking again tomorrow gets the same nothing, so this is + // an ANSWER and not a retry: `chatOnlyIdsFrom` drops the id, it leaves + // `chatOnlyPending` for good, and it settles as the ordinary filtered-out + // livestream it is. Deliberately NOT set by a pass that FAILED (a non-zero + // exit, a cooldown, an abort) — that one has to stay retryable. + noLiveChat?: boolean; scannedAt: string; }; @@ -103,6 +115,7 @@ function normalizeEntry(raw: unknown): MetadataScanEntry | null { if (typeof r.duration === "number" && Number.isFinite(r.duration)) { entry.duration = r.duration; } + if (r.noLiveChat === true) entry.noLiveChat = true; return entry; } @@ -224,7 +237,8 @@ export async function upsertMetadataScan( prev.description !== entry.description || prev.uploadDate !== entry.uploadDate || prev.liveStatus !== entry.liveStatus || - prev.duration !== entry.duration + prev.duration !== entry.duration || + prev.noLiveChat !== entry.noLiveChat ) { scan.entries[id] = entry; changed = true; @@ -282,6 +296,34 @@ export function settledIdsFrom( return settled; } +// THE CHAT-ONLY SUBSET OF THE SETTLED SET — a strict subset, never a second +// population. `titleFilterRejects` answers true for a chat-only video (see +// DownloadFilterVerdict), so every id here is also in `settledIdsFrom`'s +// answer; what this adds is which of them the operator wants the live chat of. +// +// Derived from the config every time, exactly like settledIdsFrom, so turning +// `rejectedLivestreams` off again re-decides the channel with no rescan and +// nothing stored per video to undo. +export function chatOnlyIdsFrom( + scan: MetadataScan, + config: Pick<ChannelConfig, "downloadFilter"> | null | undefined, +): Set<string> { + const ids = new Set<string>(); + const compiled = compileDownloadFilter(config?.downloadFilter); + // The mode only ever means something alongside a real filter, and a channel + // that is not chat-only pays one field read for this question. + if (!compiled || compiled.rejectedLivestreams !== "chat-only") return ids; + for (const [id, entry] of Object.entries(scan.entries)) { + // A stream we already asked and got nothing from is not chat-only work any + // more — it is an ordinary settled rejection. Without this it sits in + // chatOnlyPending forever and every runner restart re-prefetches it to run + // a pass that will never return anything. + if (entry.noLiveChat) continue; + if (titleFilterWantsChat(compiled, entry)) ids.add(id); + } + return ids; +} + // The loading form, for callers that have no scan in hand. export async function settledByTitleFilterIds( paths: Paths, diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts @@ -329,9 +329,13 @@ test("bucketsForKind: per-kind ordered bucket lists", () => { [...bucketsForKind("transcription")], ["downloadedNoTranscript", "failedListed"], ); + // `chatOnlyPending` is LAST: a chat fetch is the cheapest work on the lane and + // must never delay a real download. It is [] for every channel that has not + // set downloadFilter.rejectedLivestreams, so the order below is unchanged for + // all of them. assert.deepEqual( [...bucketsForKind("download")], - ["partialDownloads", "undownloadedIds"], + ["partialDownloads", "undownloadedIds", "chatOnlyPending"], ); }); @@ -427,7 +431,7 @@ test("the default union is unchanged by the opt-in buckets", () => { ); assert.deepEqual( [...bucketsForKind("download")], - ["partialDownloads", "undownloadedIds"], + ["partialDownloads", "undownloadedIds", "chatOnlyPending"], ); assert.deepEqual( [...defaultBucketsForPolicy("transcription", { replaceAutoSubs: false })], @@ -446,7 +450,7 @@ test("selectableBucketsForKind offers defaults plus the opt-in buckets", () => { ); assert.deepEqual( [...selectableBucketsForKind("download")], - ["partialDownloads", "undownloadedIds", "autoSubsOnly"], + ["partialDownloads", "undownloadedIds", "chatOnlyPending", "autoSubsOnly"], ); }); diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts @@ -79,7 +79,17 @@ export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder { // its internal priority. Single source of truth for the runner, the pending- // count helper, and the editor's bucket picker. export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const; -export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const; +// `chatOnlyPending` is LAST on purpose: it is the smallest and cheapest work on +// the lane (one --skip-download pass per video, no media), and putting it ahead +// of real downloads would let a chat backlog delay the corpus. Empty for every +// channel that has not set `downloadFilter.rejectedLivestreams` — which is +// every channel that predates the field — so the fold, the lane's work list and +// the pick order are byte-identical for them. +export const DOWNLOAD_BUCKETS = [ + "partialDownloads", + "undownloadedIds", + "chatOnlyPending", +] as const; // Buckets a runner will NOT draw from unless asked. Replacing YouTube's // auto-captions with our own transcript costs an audio download plus a @@ -303,6 +313,19 @@ export type ChannelWork = { // Optional: every projection written before operations existed omits it, and // a leaf naming an operation simply finds nothing. operations?: Record<string, string[]>; + // THE CHANNEL-SCOPED WORK A LEAF CANNOT NAME. The metadata scan is a fact + // about the CHANNEL, not about a video — there is no id to put in a bucket, + // because the whole point of the scan is that these videos have no directory + // — so the download lane's pre-pick reads it off the projection here rather + // than through `pick()`. Both optional: a projection that omits them offers + // no scan, which is what every caller but the download runner wants. + // + // `scanUnscanned` is the snapshot's own `metadataScan.unscanned`, which is + // defined to mean exactly what metadataScanTargets() will fetch. + // `filtered` is "this channel has a download filter that compiles" — without + // one a scan settles nothing and is pure cost against the source. + scanUnscanned?: number; + filtered?: boolean; }; // `retainLeaves(pending, root, "buckets" | "operations")` USED TO LIVE HERE, and diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts @@ -67,8 +67,31 @@ export type DownloadFilterConfig = { // `{ includeLivestreams: true }` alone rejects plain uploads and passes // livestreams. See titleFilterRejects. includeLivestreams?: boolean; + // WHAT TO DO WITH A LIVESTREAM THE FILTER REJECTED. Absent = "skip", which + // is every channel that predates this field and is byte-for-byte what they + // did before it existed. + // + // "chat-only" is the middle answer that did not exist: a multi-hour stream + // whose TITLE says nothing about the subject is usually not worth its audio, + // but its live chat is text, it is small, and it is the only record of what + // the room said. So the video is not downloaded, its chat is, and it joins + // the corpus as a chat track with no captions — NOT as a downloaded video. + // See classifyAgainstFilter for why this is a third VERDICT rather than a + // flag read at download time. + rejectedLivestreams?: RejectedLivestreamMode; }; +// Absent behaves as "skip". Spelled as a union rather than a boolean because a +// third answer (keeping the audio at a lower quality, say) is a plausible next +// one and a boolean would have to be migrated to make room for it. +export type RejectedLivestreamMode = "skip" | "chat-only"; + +export function isRejectedLivestreamMode( + v: unknown, +): v is RejectedLivestreamMode { + return v === "skip" || v === "chat-only"; +} + export type ChannelConfig = { handling: ChannelHandling; // Omitted = "video" (every channel that predates the posts corpus). @@ -365,11 +388,19 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null { const include = typeof df.include === "string" ? df.include.trim() : ""; const exclude = typeof df.exclude === "string" ? df.exclude.trim() : ""; const includeLivestreams = df.includeLivestreams === true; + // ONLY ALONGSIDE A REAL FILTER, and only when it is not the default. The + // mode says what to do with a REJECTED livestream, so with nothing to + // reject it names a decision that can never be taken — storing it would put + // a setting on the Configure form that does nothing and explains nothing. + const rejectedLivestreams = isRejectedLivestreamMode(df.rejectedLivestreams) + ? df.rejectedLivestreams + : "skip"; if (include || exclude || includeLivestreams) { config.downloadFilter = { ...(include ? { include } : {}), ...(exclude ? { exclude } : {}), ...(includeLivestreams ? { includeLivestreams: true } : {}), + ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}), }; } } diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts @@ -9,6 +9,7 @@ import { downloadFilterText, evaluateDownloadFilters, titleFilterRejects, + titleFilterWantsChat, type DownloadFilterContext, } from "./downloadFilters"; import type { RawMetadata } from "./transcripts-server"; @@ -409,3 +410,91 @@ test("the matched description is capped", () => { true, ); }); + +// --- rejectedLivestreams: the chat-only tier -------------------------------- +// +// The whole design risk of this feature is one sentence: "chat-only" is a +// REJECTION with an instruction attached, not a fourth way to pass. Everything +// that asks "is this video's media wanted?" must keep answering no for it, or +// the downloader fetches the very media the operator said not to. + +test("chat-only is a rejection, and titleFilterRejects still says so", () => { + const f = compileDownloadFilter({ + include: "guest", + rejectedLivestreams: "chat-only", + }); + assert.ok(f); + const stream = liveMeta("was_live"); + assert.equal(classifyAgainstFilter(f, stream), "chat-only"); + // THE LOAD-BEARING ONE. settledIdsFrom and therefore undownloadedIds are + // built on this: a chat-only video is settled exactly like any other + // rejection, so the download queue never offers its media. + assert.equal(titleFilterRejects(f, stream), true); + assert.equal(titleFilterWantsChat(f, stream), true); +}); + +test("it applies to livestreams ONLY, and only when configured", () => { + const f = compileDownloadFilter({ + include: "guest", + rejectedLivestreams: "chat-only", + }); + assert.ok(f); + // A rejected plain upload is a plain rejection: there is no chat to keep. + assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected"); + assert.equal(titleFilterWantsChat(f, liveMeta("not_live")), false); + // A video the filter WANTS is unaffected either way. + assert.equal( + classifyAgainstFilter(f, liveMeta("was_live", "a guest appears")), + "text", + ); + + // The default, and every channel written before the field: absent = skip. + const plain = compileDownloadFilter({ include: "guest" }); + assert.ok(plain); + assert.equal(plain.rejectedLivestreams, "skip"); + assert.equal(classifyAgainstFilter(plain, liveMeta("was_live")), "rejected"); + assert.equal(titleFilterWantsChat(plain, liveMeta("was_live")), false); +}); + +test("an EXCLUDE-matched livestream is chat-only too, because the mode is about the rejection", () => { + // The setting says what to do with a rejected livestream, not which selector + // did the rejecting. One rule, in one place, so the two cannot diverge. + const f = compileDownloadFilter({ + exclude: "rerun", + rejectedLivestreams: "chat-only", + }); + assert.ok(f); + assert.equal( + classifyAgainstFilter(f, liveMeta("was_live", "rerun of last night")), + "chat-only", + ); + // Exclude-only is still "everything else passes" — the mode adds no filtering. + assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text"); +}); + +test("the mode alone is not a filter, and is never the reason a channel is filtered", () => { + // Nothing rejecting means nothing for the mode to say, so it does not make a + // channel filtered — which is also what keeps the download lane from offering + // a metadata scan to a channel that has no filter at all. + assert.equal(compileDownloadFilter({ rejectedLivestreams: "chat-only" }), null); +}); + +test("evaluateDownloadFilters marks the decision, and it is still a skip", () => { + const decision = evaluateDownloadFilters( + ctx({ + metadata: meta({ title: "Synthetic plainvid0001", live_status: "was_live" }), + channelConfig: { + ...BASE_CONFIG, + downloadFilter: { include: "guest", rejectedLivestreams: "chat-only" }, + }, + }), + ); + assert.ok(decision); + // `skip` is what every media-fetching path reads, and it is unchanged: the + // downloader is the one caller that asks the next question. + assert.equal(decision.skip, true); + assert.equal(decision.filter, "titleFilter"); + assert.equal(decision.chatOnly, true); + assert.match(decision.reason, /live chat only/); +}); + diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts @@ -10,7 +10,11 @@ // Adding a filter = append one entry to FILTERS. Each filter sees the same // context (metadata + channel config + resolved settings). -import type { ChannelConfig, DownloadFilterConfig } from "./channelConfig"; +import type { + ChannelConfig, + DownloadFilterConfig, + RejectedLivestreamMode, +} from "./channelConfig"; import type { RawMetadata } from "./transcripts-server"; export type DownloadFilterSettings = { @@ -32,6 +36,11 @@ export type DownloadFilterDecision = { skip: boolean; filter: string; reason: string; + // The video's MEDIA is skipped and its live chat is wanted — see + // DownloadFilterConfig.rejectedLivestreams. `skip` is still true: everything + // that decides whether to fetch the media reads that and is unchanged. The + // downloader is the one caller that asks the next question. + chatOnly?: boolean; }; type DownloadFilter = { @@ -80,6 +89,9 @@ export type CompiledDownloadFilter = { include: RegExp | null; exclude: RegExp | null; includeLivestreams: boolean; + // See DownloadFilterConfig.rejectedLivestreams. "skip" is the default and is + // what every channel written before the field did. + rejectedLivestreams: RejectedLivestreamMode; }; // Compile a channel's filter. Returns null when there is no filter AND when a @@ -96,11 +108,17 @@ export function compileDownloadFilter( const exclude = filter?.exclude?.trim() ?? ""; const includeLivestreams = filter?.includeLivestreams === true; if (!include && !exclude && !includeLivestreams) return null; + // The mode is NOT a positive selector and deliberately does not make a + // channel "filtered" on its own: it says what to do with a rejection, and + // with nothing rejecting there is nothing for it to say. + const rejectedLivestreams: RejectedLivestreamMode = + filter?.rejectedLivestreams === "chat-only" ? "chat-only" : "skip"; try { return { include: include ? new RegExp(include, "i") : null, exclude: exclude ? new RegExp(exclude, "i") : null, includeLivestreams, + rejectedLivestreams, }; } catch { return null; @@ -192,7 +210,18 @@ export function downloadFilterPatternProblem(pattern: string): string | null { // Why a video passed, or that it didn't. "livestream" exists so the UI can say // how many videos a channel is keeping for a reason other than their name. -export type DownloadFilterVerdict = "text" | "livestream" | "rejected"; +// +// "chat-only" IS A REJECTION with an instruction attached, not a fourth way to +// pass. The video is not wanted; its live chat is. Everything that asks "is +// this video wanted?" — `titleFilterRejects`, and therefore the settled set and +// the download queue — must keep answering yes-it-is-rejected for it, or the +// downloader would fetch the media the operator asked it not to. Only the +// callers that ask the NEXT question ("and then what?") look for this value. +export type DownloadFilterVerdict = + | "text" + | "livestream" + | "rejected" + | "chat-only"; // PURE, and the SINGLE matcher. Both the download-time registry entry below and // the derived settled set (controller/metadataScanStore.ts) call exactly this, @@ -217,19 +246,50 @@ export function classifyAgainstFilter( meta: FilterableVideo, ): DownloadFilterVerdict { const text = downloadFilterText(meta); - if (compiled.exclude && compiled.exclude.test(text)) return "rejected"; + if (compiled.exclude && compiled.exclude.test(text)) { + return rejection(compiled, meta); + } const hasPositive = Boolean(compiled.include || compiled.includeLivestreams); if (!hasPositive) return "text"; if (compiled.include && compiled.include.test(text)) return "text"; if (compiled.includeLivestreams && isLivestream(meta)) return "livestream"; - return "rejected"; + return rejection(compiled, meta); } +// HOW a rejection is spelled — the ONE place, so `exclude`-matched and +// unmatched livestreams cannot get different answers. The operator's setting is +// about what to do with a rejected livestream, not about which selector did the +// rejecting. +function rejection( + compiled: CompiledDownloadFilter, + meta: FilterableVideo, +): DownloadFilterVerdict { + return compiled.rejectedLivestreams === "chat-only" && isLivestream(meta) + ? "chat-only" + : "rejected"; +} + +// IS THIS VIDEO'S MEDIA UNWANTED? True for "chat-only" too — see +// DownloadFilterVerdict. This is what `settledByTitleFilterIds` and therefore +// `undownloadedIds` are built on, so a chat-only video is settled exactly like +// any other rejection and the download queue never offers its media. export function titleFilterRejects( compiled: CompiledDownloadFilter, meta: FilterableVideo, ): boolean { - return classifyAgainstFilter(compiled, meta) === "rejected"; + const verdict = classifyAgainstFilter(compiled, meta); + return verdict === "rejected" || verdict === "chat-only"; +} + +// Is this one the operator wants the CHAT of? A separate question from the one +// above, asked by exactly the two places that act on the answer: the downloader, +// which fetches the chat instead of skipping, and the snapshot, which puts the +// video in its own bucket rather than in `skippedByTitleFilter`. +export function titleFilterWantsChat( + compiled: CompiledDownloadFilter, + meta: FilterableVideo, +): boolean { + return classifyAgainstFilter(compiled, meta) === "chat-only"; } // Why a rejection happened, for the log and the outcome record. @@ -288,10 +348,14 @@ const titleFilter: DownloadFilter = { const m = ctx.metadata; if (!m) return null; // fail open — see above if (!titleFilterRejects(compiled, m)) return null; + const chatOnly = titleFilterWantsChat(compiled, m); return { skip: true, filter: "titleFilter", - reason: titleFilterReason(compiled, m), + reason: chatOnly + ? `${titleFilterReason(compiled, m)} — fetching its live chat only` + : titleFilterReason(compiled, m), + ...(chatOnly ? { chatOnly: true } : {}), }; }, }; diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts @@ -27,7 +27,16 @@ export type DownloadOutcomeStatus = // The app-level filter pass (e.g. skip-live) declined to download this video. // Not a failure and not archived — the next sync/download-missing retries it // once the filter no longer matches (e.g. a live stream becomes a VOD). - | "skipped-filtered"; + | "skipped-filtered" + // The download filter rejected this livestream and the channel's + // `rejectedLivestreams` mode is "chat-only": the MEDIA was not fetched, the + // live chat was. The directory holds metadata.info.json, + // transcript.live_chat.json and its live_chat.cues.json (plus this sidecar + // and download.log) — no media and no transcript — so the video is indexable + // as a chat track with no captions while every "is it downloaded?" predicate + // — all of which test whisper/VTT/audio — still answers no. Terminal on the operator's + // terms, like skipped-filtered, and equally re-decided by editing the filter. + | "chat-only"; export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus> = [ "ok", @@ -39,6 +48,7 @@ export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus "corrupt-full-source", "failed-short-audio", "skipped-filtered", + "chat-only", ]; export type DownloadAttemptKind = @@ -52,7 +62,11 @@ export type DownloadAttemptKind = // A cookie re-run of a metadata prefetch that failed with an auth/age error // (cookie mode "always"/"when-required" with a cookie value configured). // Recorded with n: 0 alongside the failed prefetch it retries. - | "metadata-prefetch-auth-retry"; + | "metadata-prefetch-auth-retry" + // The live-chat pass a "chat-only" filter verdict runs INSTEAD of a download: + // --skip-download --write-subs --sub-langs live_chat, no media, no archive + // line. Recorded with n: 1, since it is the only real attempt there is. + | "live-chat-only"; export type AudioCheckProbeVerdict = "clean" | "partial" | "malformed"; diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts @@ -1245,10 +1245,16 @@ test("`runner` names the auto-queue runner, and only for the two that have one", const runners = new Map(operationCatalog().map((o) => [o.id, o.runner])); assert.equal(runners.get("download"), "download"); assert.equal(runners.get("transcription"), "transcription"); + // THE THIRD ONE IS CHANNEL-SCOPED, and that is the point of it being here + // rather than inferred: the download runner dispatches the metadata scan as a + // channel-scoped unit before it picks any video, so the scan's console — and + // its pause — are the download lane's. + assert.equal(runners.get("metadata-scan"), "download"); // Everything the sweep dispatches must leave it unset — a backfill kind with // a runner would render a runner console over a lane no runner feeds. + const dispatched = new Set(["download", "transcription", "metadata-scan"]); for (const op of operationCatalog()) { - if (op.id === "download" || op.id === "transcription") continue; + if (dispatched.has(op.id)) continue; assert.equal(op.runner, undefined, `${op.id} declares a runner`); } // Sync included, and it is the interesting one: the sync scheduler's diff --git a/common/lib/operations.ts b/common/lib/operations.ts @@ -1360,10 +1360,23 @@ export const SYNC_OPERATION: OperationDescriptor = { // download: it is one metadata request per listed video, and the source counts // them the same way it counts a download's. // -// NO `runner`, deliberately: nothing auto-dispatches this yet. The operator -// presses Run. The obvious follow-up is for the download lane to run it for a -// channel whose filter has unscanned listed videos, which is exactly the -// backlog this declares — see plans/FACTS.md. +// THE DOWNLOAD LANE DISPATCHES IT, and that is what `runner` says. The runner +// checks, before every pick, whether a filtered channel has unscanned listed +// videos — exactly the backlog this entry declares — and runs the scan for it +// as one channel-scoped unit on the platform download queue, holding the same +// per-platform slot a download unit would. +// +// It goes FIRST because it decides what the downloads are. Every unscanned +// non-match on a filtered channel is otherwise prefetched, rejected and +// discarded one yt-dlp invocation at a time, against a source whose patience is +// the scarce resource; one batch answers the whole channel. +// +// WHAT NAMING THE RUNNER ALSO CHANGES: `pauseLaneFor` asks `runner` first, so +// this operation's console is the download lane's and the download pause now +// holds the AUTO-dispatch. The operator's own Run button is unaffected — it +// checks the platform cooldown and nothing else — which keeps the original +// point of the exemption intact: you can still scan, while downloads are +// paused, precisely to decide what the lane should fetch when it resumes. export const METADATA_SCAN_OPERATION: OperationDescriptor = { id: "metadata-scan", label: "Metadata scan", @@ -1375,6 +1388,7 @@ export const METADATA_SCAN_OPERATION: OperationDescriptor = { dispatch: "external", scope: "channel", trigger: "backlog", + runner: "download", }; export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [ diff --git a/common/lib/pauseGates.test.ts b/common/lib/pauseGates.test.ts @@ -167,11 +167,14 @@ test("pauseLaneFor answers for every catalog id", () => { // Catalogued, and deliberately gateless: the sync scheduler's own `enabled` // is its switch, and its queue key is neither sweep's. sync: null, - // Catalogued and gateless for a different reason: the metadata scan rides - // the platform download queue but is NOT held by the download gate. It - // fetches no media, and the operator runs it precisely to decide what a - // paused download lane should fetch when it resumes. - "metadata-scan": null, + // THE DOWNLOAD LANE'S, because the download runner is what dispatches it. + // It used to be null, on the reasoning that a scan fetches no media and the + // operator runs it precisely to decide what a paused lane should fetch when + // it resumes. That reasoning is intact and now belongs to the RUN BUTTON, + // which checks the platform cooldown and no gate at all: pausing downloads + // stops the lane from auto-dispatching a scan, and stops nothing the + // operator does by hand. + "metadata-scan": "download", }; const ids = operationCatalog().map((o) => o.id); assert.equal(ids.length, 8); diff --git a/common/views/pipeline/channelFlow.ts b/common/views/pipeline/channelFlow.ts @@ -413,6 +413,17 @@ export function computeChannelFlow( "diagnostics", "Declined as currently live or upcoming; retried on a later sync.", ), + // THE GAP IS REAL AND IT IS THIS STATION'S. A chat-only video is not + // settled out of the listing (it is not in skippedByTitleFilter), so it + // counts as expected here — and until its chat lands it has no directory, + // so it is missing from totals.videos. That is exactly the gap a siding + // exists to name. + ...siding( + "chat only, not fetched", + buckets.chatOnlyPending.length, + "diagnostics", + "Livestreams the filter rejected on a channel set to keep the chat. The lane fetches the live chat last — after every real download — because it costs no media.", + ), ...siding( "need cookies", diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts @@ -39,6 +39,8 @@ export function normalizeBuckets( nonStandardVtt: raw?.nonStandardVtt ?? [], skippedByFilter: raw?.skippedByFilter ?? [], skippedByTitleFilter: raw?.skippedByTitleFilter ?? [], + chatOnly: raw?.chatOnly ?? [], + chatOnlyPending: raw?.chatOnlyPending ?? [], incompleteTranscript: raw?.incompleteTranscript ?? [], shortAudio: raw?.shortAudio ?? [], autoSubsOnly: raw?.autoSubsOnly ?? [], diff --git a/common/ytdlp/downloadOneManaged.test.ts b/common/ytdlp/downloadOneManaged.test.ts @@ -0,0 +1,140 @@ +import { test } from "node:test"; +import assert from "node:assert/strict"; +import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises"; +import { tmpdir } from "node:os"; +import path from "node:path"; +import { __discardPrefetchDirForTest as discardPrefetchDir } from "./downloadOneManaged"; + +// THE DISCARD IS A RECURSIVE DELETE, so the only test worth having is the one +// that asks what it REFUSES. +// +// It used to be a deny-list naming isVideoDownloaded, `clips/` and +// `saved-video.json`, which is a shape that cannot be right: every file nobody +// thought of is deleted by default. The cases below are the ones that were +// actually reachable — a chat-only corpus member is the worst of them, because +// the chat-only branch calls this whenever its chat pass came back with no +// file, so a failed re-fetch would have deleted the chat that was already +// there. +// +// Each case is one directory, one call, one question: did it survive? + +async function dirWith( + files: Record<string, string>, + dirs: string[] = [], +): Promise<{ root: string; videoDir: string }> { + const root = await mkdtemp(path.join(tmpdir(), "ttb-discard-")); + const videoDir = path.join(root, "data", "vid0000001"); + await mkdir(videoDir, { recursive: true }); + for (const [name, body] of Object.entries(files)) { + await writeFile(path.join(videoDir, name), body); + } + for (const name of dirs) await mkdir(path.join(videoDir, name)); + return { root, videoDir }; +} + +const noop = () => {}; + +async function discardOf( + files: Record<string, string>, + dirs: string[] = [], +): Promise<{ removed: boolean; survivors: string[] }> { + const { root, videoDir } = await dirWith(files, dirs); + try { + const removed = await discardPrefetchDir(videoDir, noop); + const survivors = await readdir(videoDir).catch(() => null); + return { removed, survivors: survivors ?? [] }; + } finally { + await rm(root, { recursive: true, force: true }); + } +} + +test("a directory holding nothing but this pass's own files is removed", async () => { + // Exactly what a rejected prefetch leaves: the info json it wrote, the + // managed download's log, and the outcome sidecar an EARLIER attempt on the + // same id left behind (including one a pre-rule rejection wrote). + const { removed, survivors } = await discardOf({ + "metadata.info.json": "{}", + "download.log": "yt-dlp\n", + "download-outcome.json": '{"status":"skipped-filtered"}', + }); + assert.equal(removed, true); + assert.deepEqual(survivors, []); +}); + +test("an empty directory is removed, and a missing one is not an error", async () => { + assert.equal((await discardOf({})).removed, true); + const root = await mkdtemp(path.join(tmpdir(), "ttb-discard-")); + try { + // Nothing to list means nothing to delete. An `rm -rf` on a path we could + // not read is the last thing to do about an unreadable path. + assert.equal( + await discardPrefetchDir(path.join(root, "data", "nope"), noop), + false, + ); + } finally { + await rm(root, { recursive: true, force: true }); + } +}); + +// Every one of these is a file the deny-list did not name and would have +// deleted. The assertion is the same each time: the directory is still there, +// with its contents intact. +const PROTECTED: Array<[string, Record<string, string>, string[]?]> = [ + [ + "a chat-only corpus member", + { + "metadata.info.json": "{}", + "transcript.live_chat.json": '{"replayChatItemAction":{}}\n', + "live_chat.cues.json": '{"cues":[]}', + "download-outcome.json": '{"status":"chat-only"}', + }, + ], + [ + "a downloaded video", + { "metadata.info.json": "{}", "audio.mp3": "bytes" }, + ], + [ + "a video whose only transcript is ours", + { "metadata.info.json": "{}", "transcript.json": "{}" }, + ], + [ + "a video whose only transcript is a foreign VTT", + { "metadata.info.json": "{}", "transcript.es.vtt": "WEBVTT\n" }, + ], + [ + "a resumable partial download", + { "metadata.info.json": "{}", "audio.mp3.part": "half" }, + ], + [ + "a diarization sidecar", + { "metadata.info.json": "{}", "diarization.json": "{}" }, + ], + [ + "a digest sidecar", + { "metadata.info.json": "{}", "digest.json": "{}" }, + ], + [ + "a saved-video pointer", + { "metadata.info.json": "{}", "saved-video.json": "{}" }, + ], + [ + "a clip window another tool asked for", + { "metadata.info.json": "{}" }, + ["clips"], + ], + [ + "anything at all that nobody has thought of yet", + { "metadata.info.json": "{}", "something-new.json": "{}" }, + ], +]; + +for (const [what, files, dirs] of PROTECTED) { + test(`${what} keeps its directory`, async () => { + const { removed, survivors } = await discardOf(files, dirs ?? []); + assert.equal(removed, false, `${what} was discarded`); + assert.deepEqual( + survivors.sort(), + [...Object.keys(files), ...(dirs ?? [])].sort(), + ); + }); +} diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts @@ -1,6 +1,6 @@ import path from "node:path"; import { appendFile, mkdir, readdir, readFile, rm } from "node:fs/promises"; -import { createWriteStream, type WriteStream } from "node:fs"; +import { createWriteStream, type Dirent, type WriteStream } from "node:fs"; import { execa } from "execa"; import { AUTH_RETRY_CLASSES, @@ -46,6 +46,8 @@ import { isLivestreamMetadata, } from "../lib/transcripts-server"; import { evaluateDownloadFilters } from "../lib/downloadFilters"; +import { normalizeLiveChat } from "../controller/normalizeLiveChat"; +import { LIVE_CHAT_FILENAME } from "../lib/videoStatus"; import { upsertMetadataScan, type MetadataScanEntry, @@ -321,6 +323,98 @@ function transcribeHandlingArgsForAudioCheck( // the body lives in ytdlp/channelArgs.ts so the clip-window fetch shares it. const channelConfigArgs = channelExtraArgs; +// A TITLE-FILTER REJECTION MUST NOT LEAVE A VIDEO DIRECTORY BEHIND. +// +// The prefetch writes `data/<id>/metadata.info.json` BEFORE the filters get to +// look at it — that is the whole point of the split — so by the time the +// operator's own "not this one" is known, the directory exists. And a directory +// holding a metadata.info.json is not a neutral leftover: `buildIndex.ts` +// admits ANY such dir to the LMDB index and to the published site (transcript +// or not), and `deriveChannelSets` reads the dir NAME as "ever fetched", which +// is what takes a vanished video out of `missingNeverFetched`. The metadata +// scan creates none of them for exactly these two reasons +// (controller/metadataScanStore.ts); a rejection that ran through the +// downloader must not create them either, or the same channel gets both +// answers depending on which path reached the video first. +// +// THE BUCKET DOES NOT COME FROM THIS DIRECTORY. `skippedByTitleFilter` is +// derived in the snapshot from the metadata-scan store against the channel's +// CURRENT filter (channelSnapshot.ts, settledIdsFrom) — the entry recorded +// beside this call — so removing the dir costs the count nothing. The retryable +// `skippedByFilter` bucket IS read off `download-outcome.json`, which is why +// every OTHER filter (skip-live) still writes one: that skip says "we will try +// again", and a video with no directory and no outcome would silently leave it. +// +// ── IT IS AN ALLOW-LIST, AND IT HAS TO BE ──────────────────────────────────── +// +// This function deletes a directory recursively, so the question it must answer +// is "is EVERYTHING in here mine?", not "is anything in here one of the few +// things I thought to check for". The deny-list it replaced named +// isVideoDownloaded, `clips/` and `saved-video.json` — and would therefore have +// deleted a chat-only corpus member (`transcript.live_chat.json` + +// `live_chat.cues.json`), a foreign-language `transcript.es.vtt`, a resumable +// `audio.mp3.part`, a `diarization.json` or a `digest.json`. That is not a +// hypothetical: the chat-only branch below calls this whenever its chat pass +// did NOT come back with a file, so a chat re-fetch that failed, was aborted +// mid-write or hit a cooldown would have deleted the chat that was already +// there, and an `importVideoAction` on an existing chat-only id would have done +// the same. +// +// So: the dir goes only when every entry is something THIS pass wrote or could +// have found from a previous run of itself. Anything else — any file, any +// subdirectory, anything a future feature adds — keeps it, and the caller falls +// back to writing the outcome sidecar exactly as it always did. Being wrong in +// this direction costs one metadata stub; being wrong in the other costs bytes +// nobody can enumerate. +const PREFETCH_OWN_FILES: ReadonlySet<string> = new Set([ + // What the prefetch pass itself writes: --write-info-json under the + // `infojson:` output template (outputArgsForUrl). + "metadata.info.json", + // The managed download's own per-video log, opened before the prefetch runs. + "download.log", + // A sidecar from an EARLIER managed attempt on this same id — including the + // one a previous rejection wrote before this rule existed. It records what + // happened, never what is on disk, so it is ours to drop with the rest. + "download-outcome.json", +]); + +async function discardPrefetchDir( + videoDir: string, + onLog: (line: string) => void, +): Promise<boolean> { + let entries: Dirent[]; + try { + entries = await readdir(videoDir, { withFileTypes: true }); + } catch { + // No directory at all — the legacy single-call path never made one — or it + // is unreadable. Either way there is nothing of ours to remove, and a + // `rm -rf` on a path we could not list is the last thing to do about it. + return false; + } + for (const entry of entries) { + // A DIRECTORY IS NEVER OURS. `clips/` is the one that exists today (media + // another tool asked this editor for, invisible to every video-dir + // enumerator); the rule is shaped so the next one needs no edit here. + if (!entry.isFile() || !PREFETCH_OWN_FILES.has(entry.name)) return false; + } + try { + await rm(videoDir, { recursive: true, force: true }); + return true; + } catch (err) { + onLog( + `Could not remove the prefetch directory ${videoDir}: ${(err as Error).message}\n`, + ); + return false; + } +} + +// Exported for the unit test ONLY. The rule above is a delete, and the +// difference between the allow-list and the deny-list it replaced is invisible +// to every end-to-end path that does not happen to have the right file on disk +// — which is exactly the shape of bug that reaches production. See +// downloadOneManaged.test.ts. +export const __discardPrefetchDirForTest = discardPrefetchDir; + // The one-invocation runner moved to ytdlp/runOneYtdlp.ts (the clip-window // fetch needs the same log tee, stderr tail and archive scrape). This wrapper // keeps the ManagedDownloadOpts-shaped call sites below unchanged. @@ -336,6 +430,57 @@ function runOneYtdlp( ); } +// ONE EXTRA yt-dlp PASS THAT FETCHES A LIVESTREAM'S CHAT AND NOTHING ELSE. +// +// Reached only from the filter branch below, for a channel whose +// `rejectedLivestreams` is "chat-only". The media is not wanted; the chat is. +// +// WHAT THE ARGUMENT ORDER IS FOR. yt-dlp takes the LAST occurrence of an +// option, so the refusals go AFTER the channel's own extra args — a channel +// that configured `--write-thumbnail` or `--write-auto-subs` must not be able +// to turn this pass back into a partial download of things nobody asked for. +// Same rule fetchWindowManaged relies on, and for the same reason. +// +// --no-write-info-json, NOT the absence of --write-info-json: the metadata is +// ALREADY on disk from the prefetch, and it has to stay there — a video dir +// with no metadata.info.json is invisible to buildIndex, so the chat would be +// fetched and then never published. This pass must neither rewrite it nor +// delete it. +// +// No archive line is appended. An archive id means "downloaded" to +// verifyTranscripts and to the sync walk, and this video is not. +async function fetchLiveChatOnly( + opts: ManagedDownloadOpts, + channelDir: string, + cookies: string | undefined, +): Promise<AttemptOutcome> { + return runOneYtdlp(opts, channelDir, [ + "--ignore-config", + "--restrict-filenames", + ...channelConfigArgs(opts.channelConfig, cookies), + "--skip-download", + "--write-subs", + "--no-write-auto-subs", + "--sub-langs", + "live_chat", + "--no-write-info-json", + "--no-write-description", + "--no-write-thumbnail", + "--no-download-archive", + ...outputArgsForUrl(opts.videoUrl), + "--", + opts.videoUrl, + ]); +} + +// Did the chat pass actually leave a chat behind? A stream with chat replay +// disabled makes yt-dlp exit 0 and write nothing, and a directory holding only +// a metadata.info.json is the leftover this slice exists to stop creating. +async function hasLiveChatOnDisk(videoDir: string): Promise<boolean> { + const entries = await readdir(videoDir).catch(() => [] as string[]); + return entries.includes(LIVE_CHAT_FILENAME); +} + function attemptSucceeded(exitCode: number | null): boolean { // yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop). return exitCode === 0 || exitCode === 101; @@ -593,6 +738,82 @@ async function runManagedDownload( opts.onLog( `Skipping ${canonicalId}: ${decision.reason} [filter=${decision.filter}]\n`, ); + // ── CHAT ONLY, AND IT RUNS BEFORE THE STORE IS WRITTEN ─────────────── + // + // The operator asked for this livestream's chat and not its media, so + // this is the one rejection that FETCHES something — and the one that + // keeps its prefetch directory. A chat-only dir holds the + // metadata.info.json (without it buildIndex never sees the video and the + // chat is published nowhere) plus transcript.live_chat.json and its + // normalized live_chat.cues.json, beside the outcome sidecar and the log + // every managed download leaves. Every "is it downloaded?" predicate + // tests whisper, an English VTT or audio, so it still answers no — which + // is right: this is a chat track, not a download. + // + // IT RUNS FIRST because its result belongs in the scan entry below. A + // stream whose chat replay is OFF has to be remembered somewhere, or the + // id sits in chatOnlyPending forever and every runner restart re-prefetches + // it and re-runs a pass that will never return anything. + let chatOnlyFetched = false; + let chatOnlyUnavailable = false; + if (decision.chatOnly && !opts.signal.aborted) { + opts.onLog( + `Fetching the live chat for ${canonicalId} (media skipped by the download filter).\n`, + ); + const chatRes = await fetchLiveChatOnly( + opts, + channelDir, + alwaysCookies(cookiePolicy) ?? + (prefetchNeededCookies ? authRetryCookies(cookiePolicy) : undefined), + ); + attempts.push({ + n: 1, + kind: "live-chat-only", + handling: opts.channelConfig.handling, + usedCookies: Boolean(alwaysCookies(cookiePolicy)), + ytdlpExitCode: chatRes.exitCode, + error: attemptSucceeded(chatRes.exitCode) + ? undefined + : trimError(chatRes.stderrTail), + }); + if (attemptSucceeded(chatRes.exitCode)) { + chatOnlyFetched = await hasLiveChatOnDisk(videoDir); + if (chatOnlyFetched) { + // The CUES sidecar, not just the raw file: buildIndex prefers + // live_chat.cues.json when it is fresh, so normalizing here makes + // the video index-ready the moment it lands instead of waiting for + // the corpus-wide normalize pass. + try { + await normalizeLiveChat({ + videoDir, + channelSlug: opts.channelSlug, + ...(opts.channelConfig.name + ? { configName: opts.channelConfig.name } + : {}), + log: (m: string) => opts.onLog(`${m}\n`), + }); + } catch (err) { + opts.onLog( + `Could not normalize the live chat: ${(err as Error).message}\n`, + ); + } + } else { + // A CLEAN PASS THAT FOUND NOTHING IS AN ANSWER, not a retry. Chat + // replay was off for this stream, or the source has since dropped + // it; asking again tomorrow gets the same nothing. Recorded on the + // scan entry below so the video leaves chatOnlyPending for good and + // settles as the ordinary filtered-out livestream it is. + // + // A FAILED pass (non-zero exit, a cooldown, an abort) is NOT this: + // it leaves the flag alone and the id stays pending, which is the + // whole reason this is gated on the exit code. + chatOnlyUnavailable = true; + opts.onLog( + `No live chat was available for ${canonicalId}; recording that so it is not asked for again.\n`, + ); + } + } + } // A title-filter rejection feeds the metadata-scan store, so this video is // settled from here on WITHOUT a second metadata fetch. The store is the // one place a settled verdict is derived from; the outcome below records @@ -608,6 +829,7 @@ async function runManagedDownload( ...(typeof metadata.duration === "number" ? { duration: metadata.duration } : {}), + ...(chatOnlyUnavailable ? { noLiveChat: true } : {}), scannedAt: new Date().toISOString(), }; // A BATCH COLLECTS; A SINGLE VIDEO WRITES. The store is channel-level, @@ -637,19 +859,38 @@ async function runManagedDownload( const record: DownloadOutcomeRecord = { videoId: canonicalId, webpageUrl: opts.videoUrl, - status: "skipped-filtered", + status: chatOnlyFetched ? "chat-only" : "skipped-filtered", startedAt, finishedAt, attempts, filter: { name: decision.filter, reason: decision.reason }, }; - try { - await mkdir(videoDir, { recursive: true }); - await writeDownloadOutcome(videoDir, record); - } catch (err) { + // The operator's own rejection takes its prefetch directory with it — + // see discardPrefetchDir for why a metadata-only dir is not a neutral + // leftover, and why the rule is an allow-list. Every other filter's skip + // is RETRYABLE and its outcome sidecar is what `skippedByFilter` derives + // from, so only this one discards. + // + // A fetched chat is not "mine to delete" either way — the allow-list + // refuses the directory the moment transcript.live_chat.json is in it — + // so this condition is a shortcut past a readdir, not the safety rail. + const discarded = + decision.filter === "titleFilter" && metadata && !chatOnlyFetched + ? await discardPrefetchDir(videoDir, opts.onLog) + : false; + if (discarded) { opts.onLog( - `Failed to write download-outcome.json: ${(err as Error).message}\n`, + `Removed the metadata-only directory for ${canonicalId}: a filtered video is not a member of this corpus.\n`, ); + } else { + try { + await mkdir(videoDir, { recursive: true }); + await writeDownloadOutcome(videoDir, record); + } catch (err) { + opts.onLog( + `Failed to write download-outcome.json: ${(err as Error).message}\n`, + ); + } } return record; } diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts @@ -948,6 +948,12 @@ async function runManagedDownloads( } if (outcome.status === "skipped-filtered") { skippedCount++; + } else if (outcome.status === "chat-only") { + // The media was skipped and the live chat was fetched. Counted as a + // SKIP, because the batch's `okCount` means "videos downloaded" and + // this one deliberately was not — the chat is a different artifact + // and the report's own chatOnly bucket is where it is counted. + skippedCount++; } else if (outcome.status === "corrupt-full-source") { // Completed but stayed malformed after one re-download: terminal and // kept on disk, but not a usable download. Count as skipped (not ok, @@ -1587,7 +1593,23 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> { .filter((e) => e.isDirectory()) .map((e) => e.name), ); - const sets = deriveChannelSets({ roster, listedIds, onDiskIds }); + // THE SETTLED SET IS SUBTRACTED, and it is not a nicety. A title-filter + // rejection leaves no directory now, so without this a filtered-out video + // that has since left the listing reads as `missingNeverFetched` — "we were + // told about this, never got it, and it is gone" — and the sweep tells the + // operator to attempt a direct-link recovery of the very video they + // configured us not to fetch. + const sweepSettledIds = await settledByTitleFilterIds( + opts.paths, + opts.channelSlug, + opts.channelConfig, + ); + const sets = deriveChannelSets({ + roster, + listedIds, + onDiskIds, + settledIds: sweepSettledIds, + }); maybeMissing = sets.missingDownloaded; await writeMaybeMissing(opts.paths, opts.channelSlug, { checkedAt: now, diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx @@ -55,6 +55,8 @@ type Props = { nonStandardVttIds: string[]; skippedByFilterIds: string[]; skippedByTitleFilterIds: string[]; + chatOnlyIds: string[]; + chatOnlyPendingIds: string[]; totals: { videos: number; transcribed: number; downloaded: number }; availability: AvailabilitySnapshot; maybeMissing: { ids: string[]; checkedAt: string }; @@ -73,6 +75,8 @@ export function DiagnosticsStage({ nonStandardVttIds, skippedByFilterIds, skippedByTitleFilterIds, + chatOnlyIds, + chatOnlyPendingIds, totals, availability, maybeMissing, @@ -131,6 +135,23 @@ export function DiagnosticsStage({ "The metadata scan read these titles and this channel's download filter (Configure > Advanced) rejects them, so nothing downloads or re-attempts them. Change the include/exclude patterns and they are re-decided on the next report — no rescan needed.", ariaLabel: "filtered out", }, + { + ids: chatOnlyIds, + label: "Chat only (filtered livestreams)", + // No Retry either, and for the same reason as the bucket above: these are + // settled, and the thing to change is the mode in Configure. A retry + // would offer to download the media the operator said not to. + description: + "Livestreams this channel's download filter rejected while \u201cFiltered-out livestreams\u201d is set to keep the chat (Configure > Advanced). The media was never fetched; the live chat was, and it is published as a chat track with no captions. Set the mode back to Skip and they stop being fetched \u2014 what is already on disk stays.", + ariaLabel: "chat only", + }, + { + ids: chatOnlyPendingIds, + label: "Chat only: chat not fetched yet", + description: + "Filtered-out livestreams whose live chat has not been fetched yet. The download lane picks these up last \u2014 after every real download \u2014 because a chat pass costs one metadata-sized request and no media.", + ariaLabel: "chat only pending", + }, ]; const populated = buckets.filter((b) => b.ids.length > 0); const total = populated.reduce((acc, b) => acc + b.ids.length, 0); diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -559,6 +559,8 @@ export default async function ChannelDetailPage({ nonStandardVttIds={buckets.nonStandardVtt} skippedByFilterIds={buckets.skippedByFilter} skippedByTitleFilterIds={buckets.skippedByTitleFilter} + chatOnlyIds={buckets.chatOnly ?? []} + chatOnlyPendingIds={buckets.chatOnlyPending ?? []} totals={snapshot.totals} availability={normalizeAvailability(snapshot.availability)} maybeMissing={normalizeMaybeMissing(snapshot.maybeMissing)} diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts @@ -25,7 +25,7 @@ import { import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp"; import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore"; import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged"; -import { runMetadataScan } from "yt-dlp-transcript-common/ytdlp/metadataScan"; +import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob"; import { getSettings } from "yt-dlp-transcript-common/lib/settings"; import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates"; import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy"; @@ -351,22 +351,17 @@ export async function runMetadataScanAction( `The metadata scan will run once the cooldown lapses.`, }; } - return runManagedFunction({ - kind: "metadata-scan", - queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey), + // THE JOB ITSELF LIVES IN common/controller/metadataScanJob.ts, because the + // download lane's runner dispatches the same one and cannot reach a server + // action. What stays here is what a CLICK is owed and a runner is not: the + // cooldown answered as a sentence rather than a queued job, and the three + // revalidations. + return runMetadataScanJob({ paths, - channelSlug: slug, - spec: { kind: "metadata-scan", slug, params: { queueKey } }, - fn: async (onLog, signal, setProgress) => { - await runMetadataScan({ - paths, - channelSlug: slug, - channelConfig, - onLog, - signal, - setProgress, - onPlatformBackoff: () => recordDownloadBackoff(platform, paths), - }); + slug, + channelConfig, + queueKey, + afterRun: () => { revalidatePath(`/channels/${slug}`); revalidatePath("/channels"); revalidatePath("/operations/[id]", "page"); diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx @@ -1603,6 +1603,11 @@ function DownloadOutcomeBadge({ ? `Filtered out by the download filter${why}` : `Skipped by filter${why}`; } + // NOT a download, and the badge has to say so plainly: this video's media + // was never fetched and never will be while the filter stands. What is + // here is its live chat, which is why the video is in the corpus at all. + case "chat-only": + return "Chat only (filtered livestream) — live chat kept, no media and no transcript"; default: return outcome.status; } diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx @@ -731,6 +731,27 @@ export function ChannelForm({ </span> </span> </label> + <label className="flex flex-col gap-1 text-sm"> + <span className="font-medium">Filtered-out livestreams</span> + <select + name="downloadFilterRejectedLivestreams" + defaultValue={c?.downloadFilter?.rejectedLivestreams ?? "skip"} + aria-label="filtered-out livestreams" + className="rounded border border-border bg-card px-2 py-1 text-sm" + > + <option value="skip">Skip them — download nothing</option> + <option value="chat-only"> + Keep the live chat — no audio, no transcript + </option> + </select> + <span className="text-xs text-muted-foreground"> + What to do with a livestream the filter above rejected. A multi-hour + stream whose title says nothing is rarely worth its audio, but its + live chat is text, it is small, and it is the only record of what the + room said. Chat-only videos join the corpus as a chat track with no + captions; they are never downloaded and never transcribed. + </span> + </label> <Field label="Cookies from browser" name="cookiesFromBrowser" diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts @@ -48,6 +48,7 @@ export const CHANNEL_FORM_VALUES = [ "cookieMode", "downloadFilterInclude", "downloadFilterExclude", + "downloadFilterRejectedLivestreams", "audioCheckIntervalSeconds", "audioCheckMaxRollbacks", "audioCheckCopyTimeoutSeconds", @@ -103,6 +104,14 @@ export function channelConfigToFormData(config: ChannelConfig): FormData { if (config.downloadFilter?.includeLivestreams) { fd.set("downloadFilterIncludeLivestreams", "on"); } + // Omitted when it is "skip" — the parser reads an absent field as "skip", so + // emitting it would be a value with no effect, and a patch clearing it ("") + // has to land on the same default the form's own select does. + put( + fd, + "downloadFilterRejectedLivestreams", + config.downloadFilter?.rejectedLivestreams, + ); if (config.audioCheck?.enabled) { fd.set("audioCheckEnabled", "on"); putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds); diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts @@ -22,6 +22,7 @@ import { handleFromAccountUrl } from "yt-dlp-transcript-common/social/fetchers"; import { isCookieMode } from "yt-dlp-transcript-common/lib/cookiePolicy"; import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat"; import { downloadFilterPatternProblem } from "yt-dlp-transcript-common/lib/downloadFilters"; +import { isRejectedLivestreamMode } from "yt-dlp-transcript-common/lib/channelConfig"; export type ParsedChannelForm = { name: string; @@ -241,12 +242,33 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm { } const includeLivestreams = formData.get("downloadFilterIncludeLivestreams") != null; + // WHAT TO DO WITH A REJECTED LIVESTREAM. Only ever stored alongside a real + // filter and only when it is not the default — with nothing rejecting, the + // mode names a decision that can never be taken, and a stored "skip" would be + // a key on disk that changes nothing. + const rejectedLivestreamsRaw = String( + formData.get("downloadFilterRejectedLivestreams") ?? "", + ).trim(); + if ( + rejectedLivestreamsRaw && + !isRejectedLivestreamMode(rejectedLivestreamsRaw) + ) { + throw new Error( + `Filtered-out livestreams must be "skip" or "chat-only" (got ${JSON.stringify( + rejectedLivestreamsRaw, + )})`, + ); + } + const rejectedLivestreams = isRejectedLivestreamMode(rejectedLivestreamsRaw) + ? rejectedLivestreamsRaw + : "skip"; const downloadFilter: ChannelConfig["downloadFilter"] | undefined = downloadFilterInclude || downloadFilterExclude || includeLivestreams ? { ...(downloadFilterInclude ? { include: downloadFilterInclude } : {}), ...(downloadFilterExclude ? { exclude: downloadFilterExclude } : {}), ...(includeLivestreams ? { includeLivestreams: true } : {}), + ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}), } : undefined; diff --git a/editor/app/operations/components/InFlightList.tsx b/editor/app/operations/components/InFlightList.tsx @@ -41,16 +41,38 @@ export function InFlightList({ aria-hidden="true" className="size-1.5 shrink-0 self-center rounded-full bg-info animate-pulse motion-reduce:animate-none" /> - <Link - href={`/channels/${item.channelSlug}/videos/${item.videoId}`} - className="font-mono text-foreground underline underline-offset-2 hover:text-brand" - > - {item.channelSlug}/{item.videoId} - </Link> - <span className="text-xs text-muted-foreground"> - {at >= 0 ? `rule ${at + 1}` : item.leafId} - {leaf ? ` · ${leafSentence(leaf, channels)}` : ""} - </span> + {/* A UNIT THAT IS NOT A VIDEO SAYS SO INSTEAD OF LINKING. + The download lane's metadata-scan unit is channel-scoped: + its `videoId` is a synthetic key naming no directory, so the + usual link would be a 404 and the usual "rule N" would name + a leaf that exists in no tree. It carries a `note`, and a + unit with one renders the note — see AutoRunnerInFlight. */} + {item.note ? ( + <> + <Link + href={`/channels/${item.channelSlug}`} + className="font-mono text-foreground underline underline-offset-2 hover:text-brand" + > + {item.channelSlug} + </Link> + <span className="text-xs text-muted-foreground"> + {item.note} + </span> + </> + ) : ( + <> + <Link + href={`/channels/${item.channelSlug}/videos/${item.videoId}`} + className="font-mono text-foreground underline underline-offset-2 hover:text-brand" + > + {item.channelSlug}/{item.videoId} + </Link> + <span className="text-xs text-muted-foreground"> + {at >= 0 ? `rule ${at + 1}` : item.leafId} + {leaf ? ` · ${leafSentence(leaf, channels)}` : ""} + </span> + </> + )} <span className="ml-auto tabular-nums text-xs text-muted-foreground"> {now !== null ? formatElapsed(now - item.startedAt) : "—"} </span> diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts @@ -1,4 +1,4 @@ -import { mkdir, writeFile } from "node:fs/promises"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; import { test, expect } from "@playwright/test"; import { channelStage, @@ -795,6 +795,125 @@ test("download: prioritizes channels across the per-platform queue", async ({ ]); }); +// A FILTERED download channel whose snapshot advertises a metadata-scan +// backlog. The fake yt-dlp titles every video "Synthetic <id>", so an include +// of /guest/ keeps guestvid* and rejects plainvid* with no fixture metadata. +// The channel's yt-dlp invocation log, or "" before the first one. +async function readText(relPath: string): Promise<string> { + return readFile(resolvePath(relPath), "utf8").catch(() => ""); +} + +async function makeFilteredDownloadChannel( + slug: string, + ids: string[], + unscanned: number, +) { + const root = resolvePath(`test-transcripts/channels/${slug}`); + await mkdir(`${root}/data`, { recursive: true }); + await writeFile( + `${root}/config.json`, + JSON.stringify({ + handling: "youtube", + name: slug, + url: `https://www.youtube.com/@${slug}/videos`, + downloadFilter: { include: "guest" }, + }), + ); + await writeFile( + `${root}/playlist`, + ids.map((id) => `https://www.youtube.com/watch?v=${id}`).join("\n") + "\n", + ); + await writeFile( + `${root}/snapshot.json`, + JSON.stringify({ + generatedAt: "2026-06-01T00:00:00.000Z", + totals: { videos: 0, transcribed: 0, downloaded: 0 }, + buckets: {}, + undownloadedIds: ids, + metadataScan: { scanned: 0, errors: 0, unscanned }, + }), + ); +} + +test("the download lane scans a filtered channel before it downloads from it", async ({ + request, +}) => { + // THE SCAN DECIDES WHAT THE DOWNLOADS ARE, so it is not another kind of work + // competing with them — it goes first. Without it every non-matching video on + // a filtered channel is prefetched, rejected and discarded one yt-dlp + // invocation at a time, against a source that is counting them; one batch + // answers the whole channel. + await resetData(null); + await makeFilteredDownloadChannel( + "filtered", + ["guestvid0001", "plainvid0001", "plainvid0002"], + 3, + ); + + const root: Group = { + id: "root", + mode: "strict", + children: [{ id: "leaf-all", match: { type: "all" } }], + }; + await writeSettings({ + adminTitle: "Test Admin", + maxTranscriptPageBytes: 8388608, + sleepBetweenDownloadsSeconds: 0, + minFreeDiskGB: 0, + workers: ONE_WORKER, + autoQueue: downloadAutoQueue(root), + }); + + await startRunner(request, "download"); + + const invocationsPath = + "test-transcripts/channels/filtered/fake-ytdlp.invocations"; + await expect + .poll( + async () => + (await readText(invocationsPath)).includes("prefetch:") || + (await pathExists( + "test-transcripts/channels/filtered/data/guestvid0001/transcript.en.vtt", + )), + { timeout: 60_000 }, + ) + .toBe(true); + + const invocations = await readText(invocationsPath); + const scanAt = invocations.indexOf("metadata-scan:"); + const firstPrefetchAt = invocations.indexOf("prefetch:"); + // The scan ran, and it ran BEFORE a single video was prefetched. + expect(scanAt).toBeGreaterThanOrEqual(0); + expect(firstPrefetchAt).toBeGreaterThan(scanAt); + + // And it read the whole listing in ONE batch — which is the saving. What is + // deliberately NOT asserted here is that the non-matches are never prefetched + // afterwards: the scan asks for a snapshot regen and the runner picks up the + // new work list on a later tick, so whether a plain video slips through in + // that window is a race with the regen, not a fact about the feature. The + // settlement itself is pinned in title-filter.spec.ts. + const scan = await readJson<{ entries: Record<string, unknown> }>( + "test-transcripts/channels/filtered/metadata-scan.json", + ); + expect(Object.keys(scan.entries).sort()).toEqual([ + "guestvid0001", + "plainvid0001", + "plainvid0002", + ]); + + // ONE SCAN PER CHANNEL PER RUNNER. The loop keeps ticking every three + // seconds; a second scan would mean the source is re-asked for the whole + // listing forever. + await expect + .poll(async () => (await getStatus(request)).download.picks.length, { + timeout: 30_000, + }) + .toBeGreaterThan(0); + expect( + (await readText(invocationsPath)).split("metadata-scan:").length - 1, + ).toBe(1); +}); + // A React state update that lands before hydration is discarded silently, so a // select or a click made too early leaves the page looking changed while the // component still holds the old value — and the next save writes the old value. diff --git a/editor/e2e/chat-only.spec.ts b/editor/e2e/chat-only.spec.ts @@ -0,0 +1,379 @@ +import { readFile, rm, writeFile } from "node:fs/promises"; +import { test, expect, type Page } from "@playwright/test"; +import { + buildIndex, + channelStage, + generateReport, + pathExists, + readJson, + resetData, + resolvePath, + writeSite, +} from "./helpers"; +import { baseUrl } from "./baseUrl"; + +// THE THIRD ANSWER FOR A FILTERED-OUT LIVESTREAM. +// +// A multi-hour stream whose title says nothing about the subject is rarely +// worth its audio, but its live chat is text, it is small, and it is the only +// record of what the room said. `downloadFilter.rejectedLivestreams: +// "chat-only"` is that answer: the media is never fetched, the chat is, and the +// video joins the corpus as a chat track with no captions. +// +// What every test here is really guarding is one risk. A chat-only directory +// makes "metadata, no transcript" a LEGITIMATE shape — and that is exactly what +// buildIndex.ts:309-315 and deriveChannelSets read as "this video was fetched". +// Get the bucket rules wrong and filtered livestreams silently leave +// missingNeverFetched and enter the published site as blank pages. + +const CHANNEL = "test-filter"; +const ROOT = `test-transcripts/channels/${CHANNEL}`; +const GUEST = ["guestvid0001", "guestvid0002"]; +const PLAIN = ["plainvid0001", "plainvid0002"]; +// The fixture's cookie-gated video. Its METADATA SCAN is refused without +// cookies, but a download-time prefetch reads it fine — so on this path (no +// scan first) the filter decides it like any other non-match. +const NEEDS_AUTH = "needsauthvid01"; +const TOKEN = "test-worker-token"; +const AUTH = { authorization: `Bearer ${TOKEN}` }; +// The fixture's FINISHED livestream VOD: `livevid` makes the fake report +// live_status "was_live", and its title contains neither "guest" nor "plain", +// so the include pattern rejects it and only the livestream rules decide it. +const LIVE = "livevid000001"; + +type Snapshot = { + generatedAt: string; + totals: { videos: number; transcribed: number; downloaded: number }; + buckets: { + skippedByTitleFilter?: string[]; + chatOnly?: string[]; + chatOnlyPending?: string[]; + noTranscript?: string[]; + downloadedNoTranscript?: string[]; + }; + undownloadedIds: string[]; +}; + +async function writeFilterConfig(filter: Record<string, unknown> | null) { + await writeFile( + resolvePath(`${ROOT}/config.json`), + JSON.stringify( + { + handling: "youtube", + name: "Test Title Filter", + url: "https://www.youtube.com/@example/videos", + ...(filter ? { downloadFilter: filter } : {}), + }, + null, + 2, + ), + ); + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); +} + +async function refreshReport( + page: Page, + after: string, + until?: (snapshot: Snapshot) => boolean, +): Promise<Snapshot> { + await page.goto("/channels"); + const refresh = page.getByRole("button", { + name: `refresh report ${CHANNEL}`, + }); + await refresh.waitFor({ state: "visible" }); + await expect + .poll( + async () => { + await refresh.click({ timeout: 5_000 }).catch(() => {}); + for (let i = 0; i < 20; i++) { + const cur = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch( + () => null, + ); + if (cur && cur.generatedAt > after && (!until || until(cur))) { + return true; + } + await new Promise((r) => setTimeout(r, 250)); + } + return false; + }, + { timeout: 60_000, intervals: [1000] }, + ) + .toBe(true); + return readJson<Snapshot>(`${ROOT}/snapshot.json`); +} + +async function download(page: Page): Promise<void> { + await page.goto(channelStage(CHANNEL, "download")); + await page.getByRole("button", { name: "Download videos" }).click(); + await expect(page.getByLabel("Download videos output")).toContainText( + "Managed download complete", + { timeout: 60_000 }, + ); +} + +test("a chat-only livestream gets its chat and stays undownloaded", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" }); + await generateReport(page, CHANNEL); + + await download(page); + + // The chat, and the metadata that makes it publishable. Nothing else. + expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.live_chat.json`)).toBe( + true, + ); + // metadata.info.json HAS to survive: buildIndex stats it per video dir and + // skips the directory when it throws, so without it the chat would be fetched + // and then published nowhere. + expect(await pathExists(`${ROOT}/data/${LIVE}/metadata.info.json`)).toBe(true); + // Normalized on the spot, so the video is index-ready when it lands rather + // than waiting for the corpus-wide normalize pass. + expect(await pathExists(`${ROOT}/data/${LIVE}/live_chat.cues.json`)).toBe(true); + // NOT a download, by every definition the app has. + expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.en.vtt`)).toBe(false); + expect(await pathExists(`${ROOT}/data/${LIVE}/audio.mp3`)).toBe(false); + const outcome = await readJson<{ status: string; attempts: unknown[] }>( + `${ROOT}/data/${LIVE}/download-outcome.json`, + ); + expect(outcome.status).toBe("chat-only"); + // No archive line: an archive id means "downloaded" to the sync walk and to + // verifyTranscripts, and this video is not. + const archive = await readFile(resolvePath(`${ROOT}/archive`), "utf8").catch( + () => "", + ); + expect(archive).not.toContain(LIVE); + + // The ordinary rejections still leave nothing behind, and the matches still + // download — the mode changes what happens to LIVESTREAMS and nothing else. + for (const id of PLAIN) { + expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); + } + for (const id of GUEST) { + expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); + } +}); + +test("a chat-only video is in chatOnly, and in no bucket that means work", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" }); + await generateReport(page, CHANNEL); + await download(page); + + const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const snap = await refreshReport( + page, + before.generatedAt, + (s) => (s.buckets.chatOnly ?? []).length > 0, + ); + + expect(snap.buckets.chatOnly ?? []).toEqual([LIVE]); + // NOT "we decided not to have it" — we decided to have its chat. + expect(snap.buckets.skippedByTitleFilter ?? []).toEqual( + [...PLAIN, NEEDS_AUTH].sort(), + ); + // Nothing is going to transcribe a video whose audio we chose not to fetch, + // so it must be out of both transcription-facing buckets AND the download + // queue, or the lanes would fight the filter forever. + expect(snap.buckets.noTranscript ?? []).not.toContain(LIVE); + expect(snap.buckets.downloadedNoTranscript ?? []).not.toContain(LIVE); + expect(snap.undownloadedIds).not.toContain(LIVE); + // Its chat is on disk, so there is no outstanding chat work either. + expect(snap.buckets.chatOnlyPending ?? []).toEqual([]); + // A MEMBER OF THIS CORPUS, not a stub: it has a directory and it is counted. + // Exactly three — the two matches plus the chat-only livestream. Every other + // rejection took its prefetch directory with it. + expect(snap.totals.videos).toBe(GUEST.length + 1); + // And it is not a DOWNLOAD: only the two matches are. + expect(snap.totals.downloaded).toBe(GUEST.length); +}); + +test("before the chat lands it is chatOnlyPending, which is download-lane work", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("title-filter-channel"); + await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" }); + await generateReport(page, CHANNEL); + + // Seed the scan store directly: the scan is what turns a listed id into a + // chat-only verdict without fetching anything, and this is the state the + // download lane is meant to find. + await writeFile( + resolvePath(`${ROOT}/metadata-scan.json`), + JSON.stringify({ + version: 1, + entries: { + [LIVE]: { + title: `Synthetic ${LIVE}`, + description: "", + uploadDate: "20240101", + liveStatus: "was_live", + scannedAt: "2026-01-01T00:00:00.000Z", + }, + }, + errors: {}, + lastRun: null, + }), + ); + await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {}); + + const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const snap = await refreshReport( + page, + before.generatedAt, + (s) => (s.buckets.chatOnlyPending ?? []).length > 0, + ); + expect(snap.buckets.chatOnlyPending ?? []).toEqual([LIVE]); + // It is NOT in undownloadedIds — fetching the media is the one thing this + // video must not have done to it — and not in skippedByTitleFilter either. + expect(snap.undownloadedIds).not.toContain(LIVE); + expect(snap.buckets.skippedByTitleFilter ?? []).not.toContain(LIVE); +}); + +// The index half, on the cheapest fixture that has one. The claim: a directory +// holding metadata.info.json and transcript.live_chat.json and NO transcript is +// admitted, published with its chat track, and carries no cues. +const INDEX_CHANNEL = "test-youtube"; +const INDEX_DIR = "20240101_test1234567"; +const INDEX_DATA = `test-transcripts/channels/${INDEX_CHANNEL}/data/${INDEX_DIR}`; + +test("the export publishes a chat-only video as a chat track with no captions", async ({ + page, +}) => { + test.setTimeout(180_000); + await resetData("one-youtube-channel-with-data"); + // Make it chat-only-shaped: the chat, and no transcript at all. + await rm(resolvePath(`${INDEX_DATA}/transcript.en.vtt`), { force: true }); + // yt-dlp's live_chat shape: one `replayChatItemAction` envelope per line, + // carrying the offset and the renderer parseLiveChat reads. + await writeFile( + resolvePath(`${INDEX_DATA}/transcript.live_chat.json`), + JSON.stringify({ + replayChatItemAction: { + videoOffsetTimeMsec: "11000", + actions: [ + { + addChatItemAction: { + item: { + liveChatTextMessageRenderer: { + authorName: { simpleText: "viewer" }, + message: { runs: [{ text: "platypus in chat" }] }, + }, + }, + }, + }, + ], + }, + }) + "\n", + ); + await writeSite("testsite", { + channels: [{ slug: INDEX_CHANNEL, groupId: "default" }], + }); + await buildIndex(page); + + const subsPage = await readJson< + Array<{ id: string; tracks?: Record<string, unknown[]> }> + >(`test-transcripts/.export-index/shared/subs/${INDEX_CHANNEL}/page-0000.json`); + const detail = subsPage.find((d) => d.id === INDEX_DIR); + expect(detail).toBeDefined(); + expect(Object.keys(detail?.tracks ?? {})).toEqual(["live_chat"]); + expect((detail?.tracks?.live_chat ?? []).length).toBeGreaterThan(0); + + // And no captions: the transcripts shard carries the video (it has metadata) + // with no cues at all. + const transcriptsPage = await readJson<Array<{ id: string; cues?: unknown[] }>>( + `test-transcripts/.export-index/shared/transcripts/${INDEX_CHANNEL}/page-0000.json`, + ); + const t = transcriptsPage.find((d) => d.id === INDEX_DIR); + expect(t).toBeDefined(); + expect(t?.cues?.length ?? 0).toBe(0); +}); + +test("the form round-trips the chat-only tier and the ops API sets it", async ({ + page, + request, +}) => { + test.setTimeout(120_000); + await resetData("title-filter-channel"); + await generateReport(page, CHANNEL); + await page.goto(channelStage(CHANNEL, "configure")); + + await page.locator("summary").filter({ hasText: "Advanced" }).click(); + const select = page.getByLabel("filtered-out livestreams"); + // Absent on disk reads as "skip", which is what every channel written before + // the field did. + await expect(select).toHaveValue("skip"); + + await select.selectOption("chat-only"); + await page.getByRole("button", { name: "Save changes" }).click(); + await expect + .poll( + async () => + ( + await readJson<{ + downloadFilter?: { rejectedLivestreams?: string }; + }>(`${ROOT}/config.json`) + ).downloadFilter?.rejectedLivestreams ?? null, + { timeout: 30_000 }, + ) + .toBe("chat-only"); + + // Back to the default, and the key is REMOVED rather than stored as "skip": + // a stored default is a key on disk that changes nothing. + await page.goto(channelStage(CHANNEL, "configure")); + await page.locator("summary").filter({ hasText: "Advanced" }).click(); + await expect(page.getByLabel("filtered-out livestreams")).toHaveValue( + "chat-only", + ); + await page.getByLabel("filtered-out livestreams").selectOption("skip"); + await page.getByRole("button", { name: "Save changes" }).click(); + await expect + .poll( + async () => + "rejectedLivestreams" in + (( + await readJson<{ downloadFilter?: Record<string, unknown> }>( + `${ROOT}/config.json`, + ) + ).downloadFilter ?? {}), + { timeout: 30_000 }, + ) + .toBe(false); + + // The ops API goes through the same parser, so it gets the same validation. + const ok = await request.post(`${baseUrl}/api/ops/channel-config`, { + headers: AUTH, + data: { + slug: CHANNEL, + patch: { downloadFilterRejectedLivestreams: "chat-only" }, + }, + }); + expect(ok.ok()).toBeTruthy(); + expect( + ( + await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>( + `${ROOT}/config.json`, + ) + ).downloadFilter?.rejectedLivestreams, + ).toBe("chat-only"); + + const bad = await request.post(`${baseUrl}/api/ops/channel-config`, { + headers: AUTH, + data: { slug: CHANNEL, patch: { downloadFilterRejectedLivestreams: "maybe" } }, + }); + expect(bad.status()).toBeGreaterThanOrEqual(400); + // Refused, and the stored value is untouched. + expect( + ( + await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>( + `${ROOT}/config.json`, + ) + ).downloadFilter?.rejectedLivestreams, + ).toBe("chat-only"); +}); diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs @@ -772,6 +772,50 @@ async function main() { return; } + // CHAT ONLY: --skip-download --write-subs --sub-langs live_chat, with + // auto-subs and the info json explicitly refused. Must come BEFORE the + // youtube single-URL branch below, which matches --write-auto-subs (this + // invocation passes --no-write-auto-subs, so it would fall through to the + // final error instead). + // + // It writes ONLY transcript.live_chat.json — no metadata.info.json (the + // prefetch already wrote it and this pass must not touch it), no transcript, + // no audio, and no DLOM_ARCHIVE line. A fake that wrote any of those would + // make the spec pass for the wrong reason: the whole claim is that a + // chat-only video is NOT downloaded. + if ( + has("--skip-download") && + has("--write-subs") && + has("--no-write-auto-subs") && + arg("--sub-langs") === "live_chat" && + !has("--flat-playlist") + ) { + const url = lastNonFlag(); + const id = urlIdYouTube(url ?? ""); + if (!id) { + process.stderr.write(`[fake-ytdlp] chat-only mode missing URL\n`); + process.exit(2); + } + const videoDir = path.join("data", id); + await ensureDir(videoDir); + await appendFile( + "fake-ytdlp.invocations", + `live-chat-only:${url} cookies=${cookieArg()}\n`, + ); + if (cookieGateBlocked(url)) failCookieGate(url); + // `nochat` is the stream whose chat replay is off: yt-dlp exits 0 and + // writes nothing, which is the case the downloader has to notice (it keeps + // no directory for it). + if (!(url ?? "").toLowerCase().includes("nochat")) { + await writeFile( + path.join(videoDir, "transcript.live_chat.json"), + '{"clientId":"fake","action":{"addChatItemAction":{}}}\n', + ); + } + process.stdout.write(`[fake-ytdlp] live chat only ${id}\n`); + return; + } + // Managed YouTube-handling per-URL invocation: --skip-download with // --write-auto-subs, no -a. The real download reuses the prefetched // metadata via --load-info-json (no positional URL) — derive the id from diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts @@ -1,4 +1,4 @@ -import { readFile, writeFile } from "node:fs/promises"; +import { mkdir, readFile, writeFile } from "node:fs/promises"; import { test, expect, type Page } from "@playwright/test"; import { channelStage, @@ -201,6 +201,112 @@ test("a settled video is never downloaded", async ({ page }) => { ); }); +test("a title-filter rejection leaves no video directory behind", async ({ + page, +}) => { + test.setTimeout(180_000); + // NO SCAN FIRST, deliberately: that is the only way a rejection reaches the + // downloader at all. Once the scan has read a video the filter settles it and + // yt-dlp is never invoked for it (the test above). The leftover this pins + // belongs to the other case — a newly listed video the download lane reaches + // before any scan does — where the metadata PREFETCH has already written + // data/<id>/metadata.info.json by the time the filter gets to say no. + // + // A directory holding a metadata.info.json is admitted to the LMDB index and + // the published site by buildIndex, and its name is what deriveChannelSets + // reads as "ever fetched". The scan creates none of them; a rejection must + // not either, or the same channel gets two different answers depending on + // which path reached the video first. + await resetData("title-filter-channel"); + await generateReport(page, CHANNEL); + + await download(page); + + const invocations = await readInvocations(); + for (const id of GUEST) { + expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true); + } + for (const id of SETTLED_BY_GUEST) { + // It WAS prefetched — that is the whole difference from the settled case — + // and the directory that prefetch made is gone again. + expect(invocations).toContain( + `prefetch:https://www.youtube.com/watch?v=${id}`, + ); + expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false); + expect(await readArchive()).not.toContain(id); + } + + // AND THE COUNT DOES NOT MOVE. The bucket is derived from the metadata-scan + // store — the rejection recorded its entry there before discarding the + // directory — so removing the directory costs it nothing. + const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`); + const snapshot = await refreshReport( + page, + before.generatedAt, + (s) => (s.buckets.skippedByTitleFilter ?? []).length > 0, + ); + // FOUR, not three, and the extra one is the point of this path rather than a + // fixture quirk: `needsauthvid01` is the video the METADATA SCAN cannot read + // (the scan's fake refuses it without cookies, so the scan-first test above + // records an ERROR for it and settles nothing). A download-time prefetch + // reads it fine, so the filter decides it here — which is exactly the + // difference this test exists to exercise. + const settledByDownload = [...SETTLED_BY_GUEST, NEEDS_AUTH].sort(); + expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(settledByDownload); + const scan = await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`); + expect(Object.keys(scan.entries).sort()).toEqual(settledByDownload); +}); + +test("a rejection never deletes a directory that holds somebody else's data", async ({ + page, +}) => { + test.setTimeout(180_000); + // The discard refuses anything but its own prefetch. A CLIP WINDOW is the + // case worth pinning: `data/<id>/clips/` is media another tool asked this + // editor for (umtool's POST /api/media/fetch-window), it is not a + // "destination" so the run still attempts the video, and it is invisible to + // every video-dir enumerator — so a discard that walked past it would delete + // bytes nothing else would ever mention again. + // + // The second video carries DOWNLOADED MEDIA. On a youtube-handling channel + // the destination is transcript.en.vtt, so an audio.mp3 does not prefilter + // the video away — the run attempts it, the filter rejects it, and the bytes + // have to survive. A video downloaded before the filter was written is + // downloaded: that is a fact, not a preference. + await resetData("title-filter-channel"); + const [withClips, withAudio] = PLAIN; + await mkdir(resolvePath(`${ROOT}/data/${withClips}/clips`), { + recursive: true, + }); + await writeFile( + resolvePath(`${ROOT}/data/${withClips}/clips/0.00-30.00.mp3`), + "fake clip audio\n", + ); + await mkdir(resolvePath(`${ROOT}/data/${withAudio}`), { recursive: true }); + await writeFile( + resolvePath(`${ROOT}/data/${withAudio}/audio.mp3`), + "fake audio bytes\n", + ); + await generateReport(page, CHANNEL); + + await download(page); + + for (const [id, file] of [ + [withClips, "clips/0.00-30.00.mp3"], + [withAudio, "audio.mp3"], + ] as const) { + expect(await pathExists(`${ROOT}/data/${id}/${file}`)).toBe(true); + // The directory stayed, so the retryable outcome sidecar is written for it + // exactly as it was before this rule existed. + const outcome = await readJson<{ status: string }>( + `${ROOT}/data/${id}/download-outcome.json`, + ); + expect(outcome.status).toBe("skipped-filtered"); + } + // And the rejection that had nothing of its own is still gone. + expect(await pathExists(`${ROOT}/data/${LIVE}`)).toBe(false); +}); + test("changing the filter re-decides the channel with no rescan", async ({ page, }) => { diff --git a/plans/FACTS.md b/plans/FACTS.md @@ -4128,11 +4128,16 @@ never pulls `reader-fs.ts` into a client chunk. - **The scan is a catalogued operation**, `metadata-scan`: second channel-scoped entry in `operationCatalog()`, `trigger: "backlog"` (sync stays the ONE `"cadence"` entry — that is how `/operations/sync` is chosen), platform download - queue, `needsMedia: false`, and `pauseLaneFor` answers null for it. It is - deliberately NOT held by the downloads pause: it fetches no media, and the - operator runs it precisely to decide what a paused lane should fetch when it - resumes. Its backlog is `snapshot.metadataScan.unscanned`, which must keep + queue. Its backlog is `snapshot.metadataScan.unscanned`, which must keep meaning exactly what `metadataScanTargets()` will fetch. + **TWO CORRECTIONS, 2026-09-21.** Its job kind declares **`needsMedia: TRUE`**, + not false — `jobKinds.ts` says why at length: the scan's target set is + "listed, minus what is already on disk", so it OPENS `data/`, and against an + unmounted drive it would re-request the whole channel. (It writes no media; + that is a different question.) And `pauseLaneFor` answers **`"download"`** + now, because the download runner dispatches it — see the follow-ups section + below. What the old exemption was FOR is intact and now belongs to the Run + button, which checks the platform cooldown and no gate at all. - **A scan error suppresses a re-scan of that id for 24 h** (`METADATA_SCAN_ERROR_COOLDOWN_MS`). Without it a members-only video keeps the backlog above zero forever and the operation page never stops offering work. @@ -4148,9 +4153,170 @@ never pulls `reader-fs.ts` into a client chunk. has no `runner`, so the operator presses Run. The obvious next step is for the download lane to run it for a channel whose filter has unscanned listed videos — which is exactly the backlog the snapshot already carries. + **DONE 2026-09-21 — see the section below, which also amends the pause note + above.** --- +## Filtered-channel follow-ups (verified 2026-09-21, branch `feat/filtered-channel-followups`) + +Three things the filter slice left: a rejection's leftover directory, nothing +auto-dispatching the scan, and no answer but "skip" for a livestream the filter +rejects. + +- **A TITLE-FILTER REJECTION NOW DELETES ITS OWN PREFETCH DIRECTORY** + (`ytdlp/downloadOneManaged.ts`, `discardPrefetchDir`). The prefetch writes + `data/<id>/metadata.info.json` BEFORE the filters look at it, so by the time + the operator's "not this one" is known the directory exists — and a directory + holding a metadata.info.json is admitted to the LMDB index and the published + site by `buildIndex.ts:309-315`, transcript or not, while `deriveChannelSets` + reads its NAME as "ever fetched". The scan creates none of them for exactly + those two reasons; the downloader now creates none either. It fires when a + rejection reaches the DOWNLOADER at all — a video the filter needed a live + prefetch for, i.e. one the scan has not read. The two batch paths never invoke + yt-dlp for a settled video (`selectDownloadableUrls` counts it an archived + hit), but `importVideoAction` reaches `downloadOneManaged` by URL and bypasses + both, so "a settled video is never invoked" is true of a sync and a + download-missing and NOT of a hand-pasted link. +- **The count does not move, and the reason is that it never came from there.** + `buckets.skippedByTitleFilter` is built at `channelSnapshot.ts:1378-1383` from + `settledIds` ← `settledIdsFrom(metadataScanStore, config)` at `:927`. Nothing + in it reads `download-outcome.json`. The only on-disk reader of + `"skipped-filtered"` is the RETRYABLE `skippedByFilter` bucket at `:1101-1105`, + which is why every OTHER filter (skip-live) still writes its outcome sidecar: + that skip says "we will try again", and a video with neither directory nor + outcome would silently leave the bucket. `channelSnapshot.test.ts` runs a whole + filtered channel with NO `data/` dirs at all and still gets the bucket. +- **`videoHasAnyArtifact` was a private copy of `isVideoDownloaded`, byte for + byte**, in a module that already imported the original (it is what increments + `downloaded` in the same walk). It is an alias now: two names for one + predicate is how the settled short-circuit and `totals.videos` end up + disagreeing about what an artifact is, and the five call sites read better + asking "does this dir hold anything we fetched?". +- **IT IS AN ALLOW-LIST, AND THE DENY-LIST IT REPLACED WAS A BUG.** This is a + recursive delete, so the question it must answer is "is EVERYTHING in here + mine?". The first version named `isVideoDownloaded`, `clips/` and + `saved-video.json` — and would therefore have deleted a chat-only corpus + member (`transcript.live_chat.json` + `live_chat.cues.json`), a + `transcript.es.vtt`, a resumable `audio.mp3.part`, a `diarization.json` or a + `digest.json`. Reachable, not hypothetical: the chat-only branch calls this + whenever its chat pass came back with no file, so a chat re-fetch that failed + or was aborted would have deleted the chat that was already there. The rule is + now `PREFETCH_OWN_FILES` = `{metadata.info.json, download.log, + download-outcome.json}`, files only — **a directory is never ours** — and + anything else keeps the dir. `ytdlp/downloadOneManaged.test.ts` runs one case + per protected shape. The returned `DownloadOutcomeRecord` is unchanged either + way, so `runYtdlp.ts:949` and the runner's `unitStatus` check need no edit. +- **`deriveChannelSets` takes the settled set**, and both call sites pass it + (`channelSnapshot.ts`, and `syncFullSweep` in `runYtdlp.ts`). Both sets it + derives are defined by the ABSENCE of a directory, and a settled video has none + now — so without this a filtered-out video that later leaves the listing reads + as `missingNeverFetched`, the loudest alarm this system raises, pointed at a + video the operator asked us not to fetch. `undownloaded` is subtracted for the + same reason: settled work is not work. +- **THE DOWNLOAD LANE AUTO-DISPATCHES THE SCAN.** `METADATA_SCAN_OPERATION` now + declares `runner: "download"`, and the runner's `next()` runs a pre-pick before + any video: `pickMetadataScanChannel` (pure, exported, unit-tested) takes the + first channel in `metaCache` order that has a COMPILING filter, a non-zero + `snapshot.metadataScan.unscanned`, no scan already done by this runner, and a + platform that is neither busy nor cooling. It goes first because it decides + what the downloads ARE. +- **The scan unit is channel-scoped and holds a platform slot.** `Picked` gains + `scan`; the synthetic key is `metadata-scan <slug>` (a space, which no video id + contains) with leaf id `metadata-scan` and an EMPTY `path`, so no node's + `maxWorkers` moves for it. It takes `platformInFlight` exactly as a download + unit does, so a scan and a download never hit one source at once. + `AutoRunnerInFlight` gains `note?`, and `InFlightList` renders the note (and + links the CHANNEL) instead of a dead video link. +- **A FOCUS HOLDS THE SCAN TOO.** The compiled priority tree holds every + download pick by strict descent, and the pre-pick does not go through that + tree — so without this, "only jeralyzer is moving" would be false the moment + another channel had titles to read, on the same network the focus is trying to + have to itself. `pickMetadataScanChannel` takes the `focusSummary` the lane + already computes for its log line: while `holding` (i.e. `focusPending > 0`) + only a focused channel is offered a scan; once the focus is exhausted the lane + is free and so is the scan. +- **ONE SCAN PER CHANNEL PER RUNNER, deliberately.** A scan that left the backlog + above zero was stopped by something — a soft block, a cooldown, a batch that + ended early. `scannedThisRun` means this runner does not re-offer it three + seconds later; the operator's Run and the next restart both still work. +- **`pauseLaneFor("metadata-scan")` is `"download"` now, not null, and the old + note's reasoning moved to the Run button.** `pauseLaneFor` asks `runner` first, + so naming the runner gives the operation the download lane's console AND makes + the download pause hold the AUTO-dispatch. What the exemption was FOR is + intact: `runMetadataScanAction` checks the platform cooldown and no gate at + all, so the operator can still scan while downloads are paused — which is + precisely when they want to decide what the lane should fetch on resume. +- **`downloadFilter.rejectedLivestreams: "skip" | "chat-only"`, absent = "skip"** + (`lib/channelConfig.ts`), and the sanitizer stores it only alongside a real + filter and only when it is not the default. A channel without the field + behaves byte-for-byte as it did. +- **"chat-only" IS A REJECTION with an instruction attached, not a fourth way to + pass.** `DownloadFilterVerdict` gains it; `classifyAgainstFilter` returns it + from ONE place (`rejection()`), so an `exclude`-matched livestream and an + unmatched one cannot get different answers. **`titleFilterRejects` still + answers TRUE for it** — that is what keeps `settledIdsFrom`, `undownloadedIds` + and the sync walk correct with no other change. `titleFilterWantsChat` is the + next question, asked only by the downloader and the snapshot. + `chatOnlyIdsFrom` (metadataScanStore) is the derived STRICT SUBSET of the + settled set. +- **The download path runs ONE extra yt-dlp pass** (`fetchLiveChatOnly`): + `--skip-download --write-subs --no-write-auto-subs --sub-langs live_chat`, with + the refusals AFTER `channelConfigArgs` and `-o` after those — yt-dlp takes the + LAST occurrence, so a channel's own `--write-thumbnail` cannot turn this into a + partial download. `--no-write-info-json`, NOT the absence of + `--write-info-json`: the prefetch's `metadata.info.json` must SURVIVE, or + `buildIndex.ts:309-315` never sees the video and the chat is published nowhere. + Then `normalizeLiveChat` writes `live_chat.cues.json` on the spot. No archive + line — an archive id means "downloaded" to the sync walk and to + `verifyTranscripts`. +- **A CLEAN PASS THAT FOUND NOTHING IS AN ANSWER, not a retry.** A stream with + chat replay off makes yt-dlp exit 0 and write nothing. The directory goes like + any other rejection's, and — because nothing on disk can then remember it — + the scan entry carries **`noLiveChat: true`**, which `chatOnlyIdsFrom` drops. + Without it the id sits in `chatOnlyPending` forever: `markCompleted` retires it + for the SESSION only, so every runner restart re-prefetched it to run a pass + that will never return anything. A pass that FAILED (non-zero exit, cooldown, + abort) deliberately does not set it — that one stays retryable — which is why + the chat pass now runs BEFORE the store is written rather than after. +- **The chat-only verdict is decided off the FILE, not the mode**, in the + snapshot's settled branch: a dir holding `transcript.live_chat.json` is a + `chatOnly` member however `rejectedLivestreams` currently reads. Asking + `chatOnlyIds` there made the same directory a settled STUB the moment the mode + flipped back — and `settledOnDisk` is subtracted from `totals.videos`, so the + site kept serving a video the report had stopped counting. +- **`DownloadOutcomeStatus` gains `"chat-only"`** and `DownloadAttemptKind` gains + `"live-chat-only"`. The batch counts it as a SKIP (`okCount` means videos + downloaded); the runner counts it as a SUCCESS (the unit did the work it was + picked for, and the platform's cooldown clears). +- **Two buckets, and neither is `skippedByTitleFilter`.** `chatOnly` = the + verdict plus `transcript.live_chat.json` on disk — a corpus MEMBER, counted in + `totals.videos`, never in `settledOnDisk` (which is subtracted from it), and + short-circuited out of every bucket that means something is missing. + `chatOnlyPending` = the verdict without the file, appended LAST to + `DOWNLOAD_BUCKETS` so a chat backlog can never delay a real download. Both are + `[]` for every channel without the field, so the fold, the lane work list and + the pick order are byte-identical for them. +- **It cannot ride in `undownloadedIds`**, and that is the reason `chatOnlyPending` + is a bucket rather than a derived count: that list means "fetch the media", and + fetching the media is the one thing this video must not have done to it. +- **THE RISK THE TESTS EXIST FOR: a chat-only dir makes "metadata, no + transcript" a legitimate shape** — exactly what `buildIndex.ts:309-315` and + `deriveChannelSets` (`channelSets.ts:58-67`) read as "ever fetched". Get the + bucket rules wrong and filtered livestreams silently leave + `missingNeverFetched` and enter the published site as blank pages. + `channelSnapshot.test.ts` runs all three states (chat landed / chat pending / + mode turned off) against a real `generateChannelSnapshot`, and + `editor/e2e/chat-only.spec.ts` builds the index and asserts the published + record carries ONLY a `live_chat` track and zero cues. +- **`common/controller/metadataScanJob.ts` is the one definition of the scan as a + JOB** — kind, per-platform queue key, replay spec and the shared + `onPlatformBackoff` wiring. The editor action keeps only what a CLICK is owed + (the cooldown answered as a sentence, and three `revalidatePath` calls, passed + as `afterRun`); the runner passes `background: true` and neither. `needsMedia` + comes from the kind (TRUE for this one), so `runManagedFunction` refuses it + against an unmounted drive for both callers with nothing said in the module. + ## Clip windows: media sourced for another tool (2026-09-20) umtool asks the editor for a clip window instead of running yt-dlp