commit 4d1226bb1b5cf9ddac31dcdb12c9260ff02386fa
parent 8a955cde3e0169e4de307391227b6ec7c87e4059
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 02:43:44 -0400
Merge branch 'feat/filtered-channel-followups'
Diffstat:
35 files changed, 2646 insertions(+), 80 deletions(-)
diff --git a/common/controller/autoRunner.test.ts b/common/controller/autoRunner.test.ts
@@ -7,6 +7,7 @@ import {
focusHoldLine,
laneDispatchRoot,
makeFocusHoldReporter,
+ pickMetadataScanChannel,
priorityContextFor,
resetPriorityContextForTest,
} from "./autoRunner";
@@ -21,6 +22,7 @@ import { LANES } from "../lib/autoQueueTypes";
import type {
AutoQueueGroup,
AutoQueuePolicy,
+ ChannelWork,
} from "../jobs/autoQueuePolicy";
import type { Paths } from "../lib/paths";
import type { SiteSettings } from "../lib/settings";
@@ -331,3 +333,149 @@ test("the context is cached on the document, and a changed document re-resolves"
"only-here",
]);
});
+
+// ---------------------------------------------------------------------------
+// The download lane's metadata-scan pre-pick.
+//
+// It runs before every video pick on the lane, so its rule is the one thing
+// standing between a filtered channel and 1,800 prefetch-reject-discard cycles
+// against a source that is counting them.
+
+function work(over: Partial<ChannelWork> & { slug: string }): ChannelWork {
+ return { platform: "youtube", buckets: {}, ...over };
+}
+
+const noSkip = {
+ scanned: new Set<string>(),
+ platformSkip: new Set<string>(),
+ platformOf: () => "youtube",
+};
+
+test("a filtered channel with an unscanned backlog is the pick", () => {
+ const pick = pickMetadataScanChannel(
+ [
+ work({ slug: "plain", scanUnscanned: 500, filtered: false }),
+ work({ slug: "filtered", scanUnscanned: 12, filtered: true }),
+ ],
+ noSkip,
+ );
+ assert.deepEqual(pick, { slug: "filtered", targets: 12 });
+});
+
+test("no filter, no scan — however big the backlog", () => {
+ // The scan's whole output is a verdict the filter makes. Without a filter it
+ // would fetch a title per listed video, decide nothing, and spend the
+ // source's patience doing it.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "plain", scanUnscanned: 9999, filtered: false })],
+ noSkip,
+ ),
+ null,
+ );
+});
+
+test("an empty backlog is not work", () => {
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 0, filtered: true })],
+ noSkip,
+ ),
+ null,
+ );
+ // And a projection with no scan fields at all — every lane but this one —
+ // offers nothing rather than throwing or defaulting to "yes".
+ assert.equal(
+ pickMetadataScanChannel([work({ slug: "filtered" })], noSkip),
+ null,
+ );
+});
+
+test("once per runner: a channel already scanned this run is skipped", () => {
+ // A scan that left the backlog above zero was stopped by something — a soft
+ // block, a cooldown, a batch that ended early. Re-offering it three seconds
+ // later would hammer the source that just refused us.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 40, filtered: true })],
+ { ...noSkip, scanned: new Set(["filtered"]) },
+ ),
+ null,
+ );
+});
+
+test("a busy or cooling platform offers no scan either", () => {
+ // The same gate the downloads honour: the scan is one more thing asking that
+ // source for titles, and the per-platform cap is one at a time.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "filtered", scanUnscanned: 40, filtered: true })],
+ { ...noSkip, platformSkip: new Set(["youtube"]) },
+ ),
+ null,
+ );
+ // A DIFFERENT platform is unaffected — the gate is per source, not global.
+ assert.deepEqual(
+ pickMetadataScanChannel(
+ [
+ work({ slug: "yt", scanUnscanned: 40, filtered: true }),
+ work({ slug: "od", scanUnscanned: 7, filtered: true, platform: "odysee" }),
+ ],
+ {
+ scanned: new Set<string>(),
+ platformSkip: new Set(["youtube"]),
+ platformOf: (slug) => (slug === "od" ? "odysee" : "youtube"),
+ },
+ ),
+ { slug: "od", targets: 7 },
+ );
+});
+
+test("the first candidate in list order wins, because that order is the priority", () => {
+ assert.deepEqual(
+ pickMetadataScanChannel(
+ [
+ work({ slug: "first", scanUnscanned: 1, filtered: true }),
+ work({ slug: "second", scanUnscanned: 900, filtered: true }),
+ ],
+ noSkip,
+ ),
+ { slug: "first", targets: 1 },
+ );
+});
+
+test("a focus holds the scan the same way it holds every download pick", () => {
+ // The compiled tree makes a focus hold downloads by strict descent, and the
+ // pre-pick does not go through that tree. Without this, "only jeralyzer is
+ // moving" is false the moment another channel has titles to read — on the
+ // same network the focus is trying to have to itself.
+ const channels = [
+ work({ slug: "other", scanUnscanned: 50, filtered: true }),
+ work({ slug: "focused", scanUnscanned: 4, filtered: true }),
+ ];
+ assert.deepEqual(
+ pickMetadataScanChannel(channels, {
+ ...noSkip,
+ focus: { slugs: new Set(["focused"]), holding: true },
+ }),
+ { slug: "focused", targets: 4 },
+ );
+ // Focus exhausted — strict descent moves on, and so does the scan.
+ assert.deepEqual(
+ pickMetadataScanChannel(channels, {
+ ...noSkip,
+ focus: { slugs: new Set(["focused"]), holding: false },
+ }),
+ { slug: "other", targets: 50 },
+ );
+ // A holding focus with no scan work of its own offers nothing, rather than
+ // falling through to the channel it is holding back.
+ assert.equal(
+ pickMetadataScanChannel(
+ [work({ slug: "other", scanUnscanned: 50, filtered: true })],
+ { ...noSkip, focus: { slugs: new Set(["focused"]), holding: true } },
+ ),
+ null,
+ );
+});
+
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -7,6 +7,8 @@ import { getSettings, type SiteSettings } from "../lib/settings";
import { diskGate } from "../lib/diskSpace";
import { formatBytes } from "../lib/format";
import { detectPlatform } from "../lib/platform";
+import { compileDownloadFilter } from "../lib/downloadFilters";
+import { runMetadataScanJob } from "./metadataScanJob";
import { getWorkerPool } from "../jobs/workerPool";
import { getRegistry } from "../jobs/registry";
import { runManagedFunction } from "../jobs/streamCommand";
@@ -160,6 +162,14 @@ const CHANNEL_LIST_TTL_MS = 30_000;
// grow it unbounded. FIFO eviction; the snapshot catches up within ~1s anyway.
const COMPLETED_CAP = 5000;
+// THE DOWNLOAD LANE'S CHANNEL-SCOPED UNIT. A metadata scan has no video id, so
+// it gets a synthetic one under a prefix nothing else can produce (a video id
+// never contains a space) and a leaf id that exists in no tree — the pick is
+// not made by `selectNextWork` and never reaches one. Both are exported so the
+// console and the tests name the same strings the runner writes.
+export const METADATA_SCAN_UNIT_PREFIX = "metadata-scan ";
+export const METADATA_SCAN_LEAF_ID = "metadata-scan";
+
// --- Live status (for /api/auto-queue/status), per kind --------------------
export type AutoRunnerInFlight = {
@@ -167,6 +177,13 @@ export type AutoRunnerInFlight = {
leafId: string;
channelSlug: string;
startedAt: number;
+ // A SENTENCE FOR A UNIT THAT IS NOT A VIDEO. `videoId` is the map key and
+ // therefore always set, but the download lane's metadata-scan unit is
+ // channel-scoped: its key is a synthetic `metadata-scan <slug>` that names no
+ // video directory. A console must not link it as one, so a unit carrying a
+ // note renders the note instead of a video link. Absent for every ordinary
+ // unit, which is how the surfaces stay unchanged.
+ note?: string;
};
// Why the runner is up but dispatching nothing. Every one of these was already
@@ -675,7 +692,21 @@ async function buildChannelWork(
for (const id of ids) if (!owner.has(id)) owner.set(id, slug);
}
}
- channels.push({ slug, platform, buckets, operations: ops });
+ // THE CHANNEL-SCOPED HALF, off the snapshot already read. The download
+ // lane's pre-pick uses these two to decide whether to run a metadata scan
+ // for this channel before it downloads anything from it; every other lane
+ // ignores them. Both come out of `snap`/`config` with no extra I/O — the
+ // scan backlog is a number the snapshot already publishes, and the filter
+ // is the config listChannelConfigs already parsed.
+ const config = meta[i].config;
+ channels.push({
+ slug,
+ platform,
+ buckets,
+ operations: ops,
+ scanUnscanned: snap.metadataScan?.unscanned ?? 0,
+ filtered: compileDownloadFilter(config.downloadFilter) !== null,
+ });
}
return { channels, owner };
}
@@ -1013,7 +1044,74 @@ function eligibleSlots(): number {
// One unit of work selected by the policy, carried from the runPool `next`
// (selection + reservation) into `run` (execution + release).
-type Picked = { pick: WorkPick; channelSlug: string; unitPlatform: string | null };
+// WHICH CHANNEL GETS A METADATA SCAN THIS TICK, or none — the download lane's
+// pre-pick, as a pure function over the projection the runner already built.
+//
+// Pure because the rule is the whole feature and everything else about it is
+// plumbing. Three conditions, and each one is a bug if it goes:
+//
+// - A FILTER. Without one a scan settles nothing: it would fetch a title per
+// listed video, change no verdict, and spend the source's patience to do
+// it. `filtered` is "the channel's downloadFilter compiles", which is the
+// same predicate `settledByTitleFilterIds` opens the store for.
+// - A BACKLOG. `scanUnscanned` is the snapshot's own `metadataScan.unscanned`,
+// defined to mean exactly what `metadataScanTargets()` will fetch, so a
+// channel offered here is a channel the scan has work for.
+// - ONCE PER RUNNER. A channel whose scan left the backlog above zero — a
+// soft block, a cooldown, a batch that stopped early — is not retried by
+// this runner. Re-offering it on the next three-second tick would hammer
+// the source that just refused us; the operator's Run and the next restart
+// both still work.
+//
+// The platform gate is the same one the downloads honour: a busy or cooling
+// platform offers no scan either, because a scan is one more thing asking that
+// source for titles.
+//
+// A FOCUS HOLDS THE SCAN TOO. The compiled priority tree makes a focus hold
+// every download pick by strict descent (see laneDispatchRoot), and the pre-pick
+// does not go through that tree — so without this, "only jeralyzer is moving"
+// would be false the moment another channel had titles to read, on the same
+// network the focus is trying to have to itself. The rule mirrors what strict
+// descent does rather than re-implementing it: while the focus still has
+// pending work, only a focused channel is offered a scan; once it is exhausted
+// the lane is free and so is the scan.
+//
+// FIRST IN `channels` ORDER, which is `metaCache` order, which is the priority
+// order every other pick on this lane already uses.
+export function pickMetadataScanChannel(
+ channels: ReadonlyArray<ChannelWork>,
+ opts: {
+ scanned: ReadonlySet<string>;
+ platformSkip: ReadonlySet<string>;
+ platformOf: (slug: string) => string;
+ // The resolved focus, when one is holding the lane. Absent (or with
+ // `holding: false`) means every channel is eligible, which is the default
+ // and the byte-identical path for a corpus with no focus configured.
+ focus?: { slugs: ReadonlySet<string>; holding: boolean };
+ },
+): { slug: string; targets: number } | null {
+ const held = opts.focus?.holding === true;
+ for (const c of channels) {
+ const targets = c.scanUnscanned ?? 0;
+ if (!c.filtered || targets <= 0) continue;
+ if (held && !opts.focus!.slugs.has(c.slug)) continue;
+ if (opts.scanned.has(c.slug)) continue;
+ if (opts.platformSkip.has(opts.platformOf(c.slug))) continue;
+ return { slug: c.slug, targets };
+ }
+ return null;
+}
+
+type Picked = {
+ pick: WorkPick;
+ channelSlug: string;
+ unitPlatform: string | null;
+ // THE CHANNEL-SCOPED UNIT. Set only by the download lane's metadata-scan
+ // pre-pick: `pick.videoId` is then a synthetic key naming no video and
+ // `pick.path` is empty, so no node's active count moves for it. `run()`
+ // branches on this and nothing else.
+ scan?: { slug: string; targets: number };
+};
async function runLoop(
kind: AutoQueueKind,
@@ -1111,6 +1209,13 @@ async function runLoop(
// "unknown" for unrecognized hosts). Unused for transcription.
const platformInFlight = new Map<string, number>();
const PER_PLATFORM_CAP = 1;
+ // Channels this runner has already scanned (or tried to). See the pre-pick in
+ // next() for why a scan is never retried inside one runner's lifetime.
+ const scannedThisRun = new Set<string>();
+ // Platforms this tick may not touch — busy or cooling down. Hoisted out of the
+ // download branch because the metadata-scan pre-pick honours it too: a
+ // platform in a 429 cooldown must not be asked for 1,800 titles either.
+ const platformSkip = new Set<string>();
const platformKey = (slug: string, slugToPlatform: Map<string, string>) =>
slugToPlatform.get(slug) ?? "unknown";
@@ -1472,11 +1577,14 @@ async function runLoop(
// Say — ONCE per transition — whether a focus is holding this lane. Read off
// the `prio-*` leaf ids in the map just built, so it costs one pass over
// keys and no new read, and it is skipped entirely while no focus resolves.
- reportFocusHold(
+ // Computed ONCE and reused by the scan pre-pick below, which has to honour
+ // the same hold: `holding` is `focusPending > 0`, i.e. exactly "strict
+ // descent is still inside the focus group".
+ const focus =
ctx.focusSlugs.length > 0
? focusSummary(ctx.model, ctx.focusSlugs, pending)
- : null,
- );
+ : null;
+ reportFocusHold(focus);
// THE RUNNER JOB'S OWN BAR, on the metrics the per-channel jobs already
// use — so a lane's runner row reads like the manual verb's row rather than
// like an opaque long-lived loop. `target` moves as the corpus does (this
@@ -1489,6 +1597,7 @@ async function runLoop(
// download (busy) OR is in a rate-limit/network backoff window (cooling
// down), so a busy/throttled platform yields to the next-priority free one.
let anyCooling = false;
+ platformSkip.clear();
if (kind === "download") {
// Merge in any cooldown a manual sync/import wrote to the shared state
// (read-modify-write from outside the runner) since our last persist.
@@ -1504,29 +1613,104 @@ async function runLoop(
} catch {
// Best-effort: a transient read failure just skips this iteration's merge.
}
- const skip = new Set<string>();
for (const [pf, n] of platformInFlight) {
- if (n >= PER_PLATFORM_CAP) skip.add(pf);
+ if (n >= PER_PLATFORM_CAP) platformSkip.add(pf);
}
for (const pf of Object.keys(kindState.platformBackoff)) {
if (isCoolingDown(kindState.platformBackoff, pf, now)) {
- skip.add(pf);
+ platformSkip.add(pf);
// A platform skipped for a cooldown is a different answer to "why is
// it idle?" than one skipped for being busy, so the two are tracked
// apart rather than both reading as "capped".
anyCooling = true;
}
}
- if (skip.size > 0) {
+ if (platformSkip.size > 0) {
for (const leafId of Object.keys(pending)) {
pending[leafId] = pending[leafId].filter((id) => {
const slug = owner.get(id);
- return !(slug && skip.has(platformKey(slug, slugToPlatform)));
+ return !(slug && platformSkip.has(platformKey(slug, slugToPlatform)));
});
}
}
}
+ // ── THE SCAN COMES BEFORE THE DOWNLOADS ────────────────────────────────
+ //
+ // A filtered channel with unscanned listed videos is a channel whose
+ // download queue is WRONG, not merely incomplete: every non-matching video
+ // in it will be prefetched, rejected and discarded one at a time, at one
+ // yt-dlp invocation each, against a source whose patience is the scarce
+ // resource. The scan answers the same question for the whole channel in one
+ // batch. So it is not another kind of work competing with downloads — it is
+ // the thing that decides what the downloads ARE, and it goes first.
+ //
+ // ONE PER CHANNEL PER RUNNER, and never a second while one is in flight.
+ // `scannedThisRun` is the session's memory: a channel whose scan left the
+ // backlog above zero (a soft block, a cooldown, a batch that stopped early)
+ // is NOT retried by this runner — the operator's Run button and the next
+ // restart both still work, and re-offering it on the next three-second tick
+ // would hammer the very source that just refused us.
+ //
+ // It takes a platform slot exactly as a download unit does, so a scan and a
+ // download never hit one source at once, and it inherits the platform
+ // cooldown for free: a platform in `platformSkip` offers no scan either.
+ if (kind === "download") {
+ const scanCandidate = pickMetadataScanChannel(channels, {
+ scanned: scannedThisRun,
+ platformSkip,
+ platformOf: (slug) => platformKey(slug, slugToPlatform),
+ ...(focus
+ ? {
+ focus: {
+ slugs: new Set(ctx.focusSlugs),
+ holding: focus.holding,
+ },
+ }
+ : {}),
+ });
+ if (scanCandidate) {
+ const { slug, targets } = scanCandidate;
+ const unitPlatform = platformKey(slug, slugToPlatform);
+ scannedThisRun.add(slug);
+ const videoId = `${METADATA_SCAN_UNIT_PREFIX}${slug}`;
+ const note = `scanning ${targets} listed video${
+ targets === 1 ? "" : "s"
+ } for ${slug}`;
+ live.idleReason = null;
+ // No `path`, so no node's active count moves: the scan is not a leaf's
+ // work and must not consume a leaf's or a group's maxWorkers.
+ live.inFlight.set(videoId, {
+ videoId,
+ leafId: METADATA_SCAN_LEAF_ID,
+ channelSlug: slug,
+ startedAt: Date.now(),
+ note,
+ });
+ platformInFlight.set(
+ unitPlatform,
+ (platformInFlight.get(unitPlatform) ?? 0) + 1,
+ );
+ // IN THE PICK LOG LIKE ANY OTHER UNIT. The log is "what did this lane
+ // do", and a tick that ran a scan and no download would otherwise read
+ // as a tick that did nothing at all.
+ recordPick(kindState, {
+ at: Date.now(),
+ leafId: METADATA_SCAN_LEAF_ID,
+ videoId,
+ channelSlug: slug,
+ });
+ persist();
+ onLog(`Auto-download: ${note}.`);
+ return {
+ pick: { leafId: METADATA_SCAN_LEAF_ID, videoId, path: [] },
+ channelSlug: slug,
+ unitPlatform,
+ scan: { slug, targets },
+ };
+ }
+ }
+
const pick = selectNextWork(root, pending, runtime, live.active);
if (!pick) {
// Attribute the idleness. Nothing pending at all is a different situation
@@ -1682,6 +1866,17 @@ async function runLoop(
result = { outcome: "skipped" };
return;
}
+ if (picked.scan) {
+ result = await launchMetadataScan({
+ paths,
+ slug: picked.scan.slug,
+ targets: picked.scan.targets,
+ tracker,
+ onLog,
+ onChildJob: (jid) => childJobIds.set(pick.videoId, jid),
+ });
+ return;
+ }
result = isOperationLane(kind)
? await runOperationPick(picked, runSignal)
: await launchUnit({
@@ -1776,6 +1971,55 @@ async function runLoop(
);
}
+// ONE METADATA SCAN, as a child job on the channel's platform download queue.
+//
+// The SAME job the operator's Run button dispatches — kind, queue key, replay
+// spec and the shared per-platform cooldown all come from
+// controller/metadataScanJob.ts — so the registry serializes it against a
+// manual sync on that platform exactly as it serializes a download unit, and
+// there is one definition of what a scan IS.
+//
+// `background: true` for the same reason a download unit is: a clicked action
+// preempts queued units without interrupting a running one.
+//
+// A SCAN IS NEVER A FAILURE FOR BACKOFF PURPOSES. runMetadataScan does not
+// throw on a rate limit — it stops, records the shared cooldown through its own
+// onPlatformBackoff, and keeps what it flushed. So this returns "skipped" or
+// "transcribed" and never a `failureClass`: the cooldown is already recorded by
+// the time we get here, and a second backoff entry from the runner would
+// double-count one refusal.
+async function launchMetadataScan(args: {
+ paths: Paths;
+ slug: string;
+ targets: number;
+ tracker: ReturnType<typeof makeTaskTracker>;
+ onLog: (line: string) => void;
+ onChildJob?: (jobId: string) => void;
+}): Promise<UnitResult> {
+ const config = await readChannelConfig(args.paths, args.slug);
+ if (!config) return { outcome: "skipped" };
+ const res = await runMetadataScanJob({
+ paths: args.paths,
+ slug: args.slug,
+ channelConfig: config,
+ background: true,
+ });
+ if (!res.ok) {
+ args.onLog(
+ `Auto-download: could not enqueue the metadata scan for ${args.slug}: ${res.error}`,
+ );
+ return { outcome: "skipped" };
+ }
+ args.onChildJob?.(res.jobId);
+ const term = await res.done;
+ if (term.status === "cancelled") return { outcome: "skipped" };
+ if (term.status === "failed") {
+ args.onLog(`Auto-download: the metadata scan for ${args.slug} failed.`);
+ return { outcome: "skipped" };
+ }
+ return { outcome: "transcribed" };
+}
+
type LaunchArgs = {
kind: AutoQueueKind;
paths: Paths;
@@ -1978,6 +2222,11 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
// Cancelled (runner stopped, or dropped while still queued) → not a failure.
if (term.status === "cancelled") return { outcome: "skipped" };
if (unitStatus === "skipped-filtered") return { outcome: "skipped" };
+ // The filter declined the media and the live chat was fetched instead. A
+ // SUCCESS for the lane — the unit did the work it was picked for, the video
+ // leaves chatOnlyPending, and a platform whose chat pass came back clean has
+ // its cooldown cleared exactly as a download would clear it.
+ if (unitStatus === "chat-only") return { outcome: "transcribed" };
// Complete-but-malformed source: terminal and kept on disk. Treat as skipped
// (not failed) so it doesn't drive backoff and isn't re-picked for download.
if (unitStatus === "corrupt-full-source") return { outcome: "skipped" };
diff --git a/common/controller/channelSets.test.ts b/common/controller/channelSets.test.ts
@@ -21,11 +21,13 @@ function derive(input: {
roster: ReadonlyArray<string>;
listed: ReadonlyArray<string>;
onDisk: ReadonlyArray<string>;
+ settled?: ReadonlyArray<string>;
}) {
return deriveChannelSets({
roster: rosterOf(input.roster),
listedIds: new Set(input.listed),
onDiskIds: new Set(input.onDisk),
+ ...(input.settled ? { settledIds: new Set(input.settled) } : {}),
});
}
@@ -111,3 +113,40 @@ test("an empty channel derives empty sets rather than throwing", () => {
orphaned: [],
});
});
+
+test("a settled video is neither never-fetched nor undownloaded", () => {
+ // BOTH SETS ARE DEFINED BY THE ABSENCE OF A DIRECTORY, and a video the
+ // download filter settled has none — a title-filter rejection deletes its own
+ // prefetch dir. So without subtracting the settled set, a filtered-out video
+ // that later leaves the listing reads as "we were told about this, never got
+ // it, and it is gone", which is the loudest alarm this system raises and the
+ // sweep prints a recovery prompt for it.
+ const sets = derive({
+ roster: ["keep", "drop"],
+ listed: ["keep"],
+ onDisk: [],
+ settled: ["drop"],
+ });
+ assert.deepEqual(sets.missingNeverFetched, []);
+ assert.deepEqual(sets.undownloaded, ["keep"]);
+ // And the control: the same corpus with nothing settled still raises it, so
+ // the subtraction has not made the category unreachable.
+ const unsettled = derive({
+ roster: ["keep", "drop"],
+ listed: ["keep"],
+ onDisk: [],
+ });
+ assert.deepEqual(unsettled.missingNeverFetched, ["drop"]);
+});
+
+test("an omitted settled set changes nothing, for every channel without a filter", () => {
+ const withEmpty = derive({
+ roster: ["a", "b"],
+ listed: ["a"],
+ onDisk: [],
+ settled: [],
+ });
+ const without = derive({ roster: ["a", "b"], listed: ["a"], onDisk: [] });
+ assert.deepEqual(withEmpty, without);
+});
+
diff --git a/common/controller/channelSets.ts b/common/controller/channelSets.ts
@@ -37,15 +37,28 @@ export function deriveChannelSets(input: {
roster: Roster;
listedIds: ReadonlySet<string>;
onDiskIds: ReadonlySet<string>;
+ // Ids this channel's download filter has SETTLED — the operator's own "not
+ // this one". Optional, and empty for every channel without a filter.
+ //
+ // WHY IT BELONGS HERE. Both sets below are defined by the ABSENCE of a
+ // directory, and since a title-filter rejection stopped leaving its prefetch
+ // dir behind, a settled video has none. Without this, a filtered-out video
+ // that later left the listing read as `missingNeverFetched` — "we were told
+ // about this, never got it, and now it is gone" — which is the highest-value
+ // alarm this system raises, pointed at a video the operator asked us not to
+ // fetch. `undownloaded` has the same shape of wrongness: it is a work list,
+ // and settled work is not work.
+ settledIds?: ReadonlySet<string>;
}): ChannelSets {
const { roster, listedIds, onDiskIds } = input;
+ const settledIds = input.settledIds ?? new Set<string>();
const rosterIds = Object.keys(roster.entries);
const listed: string[] = [];
const undownloaded: string[] = [];
for (const id of listedIds) {
listed.push(id);
- if (!onDiskIds.has(id)) undownloaded.push(id);
+ if (!onDiskIds.has(id) && !settledIds.has(id)) undownloaded.push(id);
}
// missingDownloaded is derived from the DISK set, not from `roster ∩ disk`,
@@ -64,6 +77,7 @@ export function deriveChannelSets(input: {
const missingNeverFetched: string[] = [];
for (const id of rosterIds) {
+ if (settledIds.has(id)) continue;
if (!listedIds.has(id) && !onDiskIds.has(id)) missingNeverFetched.push(id);
}
diff --git a/common/controller/channelSnapshot.test.ts b/common/controller/channelSnapshot.test.ts
@@ -358,3 +358,326 @@ test("generateChannelSnapshot refuses an unreachable channel rather than writing
await rm(dir, { recursive: true, force: true });
}
});
+
+// ---------------------------------------------------------------------------
+// The filtered channel, end to end through generateChannelSnapshot.
+//
+// It needs nothing but a directory: a config, a playlist and the metadata-scan
+// store. That matters here because the whole point of the two facts below is
+// that they are derived from the STORE and not from anything on a video's
+// disk — a claim only a run with no video dirs at all can actually pin.
+
+type FilterFixture = {
+ filter?: Record<string, unknown>;
+ listed: string[];
+ scanned?: Record<
+ string,
+ { title?: string; liveStatus?: string; noLiveChat?: boolean }
+ >;
+ // Ids the channel has EVER been seen to contain. Seeded so the
+ // missingNeverFetched rule — "we were told about this, never got it, and it
+ // is gone" — can be exercised: it is derived from the roster, not the disk.
+ roster?: string[];
+};
+
+async function filteredChannel(
+ fixture: FilterFixture,
+): Promise<{ dir: string; paths: Paths; channelDir: string }> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-filter-"));
+ const paths = {
+ channelsDir: path.join(dir, "channels"),
+ transcriptsDir: dir,
+ savedVideosDir: path.join(dir, "saved-videos"),
+ } as Paths;
+ const channelDir = path.join(paths.channelsDir, "alpha");
+ await mkdir(path.join(channelDir, "data"), { recursive: true });
+ await writeFile(
+ path.join(channelDir, "config.json"),
+ JSON.stringify({
+ handling: "youtube",
+ url: "https://www.youtube.com/@alpha/videos",
+ ...(fixture.filter ? { downloadFilter: fixture.filter } : {}),
+ }),
+ );
+ await writeFile(
+ path.join(channelDir, "playlist"),
+ fixture.listed
+ .map((id) => `https://www.youtube.com/watch?v=${id}`)
+ .join("\n"),
+ );
+ const entries: Record<string, unknown> = {};
+ for (const [id, e] of Object.entries(fixture.scanned ?? {})) {
+ entries[id] = {
+ title: e.title ?? `Synthetic ${id}`,
+ description: "",
+ uploadDate: "20240101",
+ ...(e.liveStatus ? { liveStatus: e.liveStatus } : {}),
+ ...(e.noLiveChat ? { noLiveChat: true } : {}),
+ scannedAt: "2026-01-01T00:00:00.000Z",
+ };
+ }
+ if (fixture.roster) {
+ await writeFile(
+ path.join(channelDir, "roster.json"),
+ JSON.stringify({
+ version: 1,
+ entries: Object.fromEntries(
+ fixture.roster.map((id) => [
+ id,
+ {
+ url: `https://www.youtube.com/watch?v=${id}`,
+ firstSeenAt: "2026-01-01T00:00:00.000Z",
+ },
+ ]),
+ ),
+ }),
+ );
+ }
+ await writeFile(
+ path.join(channelDir, "metadata-scan.json"),
+ JSON.stringify({ version: 1, entries, errors: {}, lastRun: null }),
+ );
+ return { dir, paths, channelDir };
+}
+
+test("a settled video is in skippedByTitleFilter with NO directory of its own", async () => {
+ // THE BUCKET IS SOURCED FROM THE SCAN STORE, not from a download-outcome
+ // sidecar — which is what lets a title-filter rejection delete the prefetch
+ // directory it made (ytdlp/downloadOneManaged.ts, discardPrefetchDir) without
+ // the count moving. This fixture has no data/ dirs at all; if the bucket
+ // depended on one, it would be empty here.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: ["aaaa0000001", "bbbb0000002"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ bbbb0000002: { title: "drop this one" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]);
+ // And it is out of the download lane's queue, which is the point of it.
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]);
+ // A settled id was never a video: it has no directory, so it was never in
+ // totals.videos either.
+ assert.equal(snap.totals.videos, 0);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a chat-only livestream is a corpus member, not a settled stub", async () => {
+ // THE RISK THIS PINS. A chat-only dir makes "metadata, no transcript" a
+ // LEGITIMATE shape — which is exactly what buildIndex and deriveChannelSets
+ // read as "this video was fetched". If the bucket rules are wrong, filtered
+ // livestreams either vanish from the report or reappear as download work
+ // forever.
+ const { dir, paths, channelDir } = await filteredChannel({
+ filter: { include: "keep", rejectedLivestreams: "chat-only" },
+ listed: ["aaaa0000001", "bbbb0000002", "cccc0000003"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ bbbb0000002: { title: "drop this one" },
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ // The chat has landed for the livestream: metadata + the raw chat, and
+ // nothing else on disk.
+ const videoDir = path.join(channelDir, "data", "cccc0000003");
+ await mkdir(videoDir, { recursive: true });
+ await writeFile(
+ path.join(videoDir, "metadata.info.json"),
+ JSON.stringify({ id: "cccc0000003", title: "a long stream" }),
+ );
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"action":{}}\n',
+ );
+
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]);
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ // NOT "we decided not to have it".
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]);
+ // Nothing will transcribe a video whose audio we chose not to fetch, and
+ // nothing will download it.
+ assert.ok(!snap.buckets.noTranscript.includes("cccc0000003"));
+ assert.ok(!snap.buckets.downloadedNoTranscript.includes("cccc0000003"));
+ assert.ok(!snap.undownloadedIds.includes("cccc0000003"));
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]);
+ // A video we deliberately have. `settledOnDisk` is subtracted from this and
+ // a chat-only dir must never be in it.
+ assert.equal(snap.totals.videos, 1);
+ assert.equal(snap.totals.downloaded, 0);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("before the chat lands it is chatOnlyPending — the download lane's bucket", async () => {
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep", rejectedLivestreams: "chat-only" },
+ listed: ["aaaa0000001", "cccc0000003"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnlyPending, ["cccc0000003"]);
+ assert.deepEqual(snap.buckets.chatOnly, []);
+ // It cannot ride in undownloadedIds: that list means "fetch the media", and
+ // fetching the media is the one thing this video must not have done to it.
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]);
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, []);
+ // It IS the download lane's work, through the lane work list every lane
+ // reads its dispatch from.
+ assert.ok(snap.backfill?.download?.ids.includes("cccc0000003"));
+ // And LAST in it, behind the real download — the fold walks
+ // DOWNLOAD_BUCKETS in order.
+ const ids = snap.backfill?.download?.ids ?? [];
+ assert.equal(ids[ids.length - 1], "cccc0000003");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("turning the mode off re-decides the channel with nothing to migrate", async () => {
+ // Same store, same disk, no filter field: the livestream is an ordinary
+ // settled rejection again and the chat-only buckets are empty. Nothing about
+ // the verdict was ever stored per video.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: ["cccc0000003"],
+ scanned: {
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnly, []);
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a settled video that left the listing is NOT reported as never fetched", async () => {
+ // THE LOUDEST ALARM THIS REPORT RAISES, pointed at the wrong video. Both
+ // missingNeverFetched and undownloaded are defined by the ABSENCE of a
+ // directory, and a settled video has none since a rejection stopped leaving
+ // its prefetch dir behind — so without subtracting the settled set, the sweep
+ // tells the operator to attempt a direct-link recovery of a video they
+ // configured us not to fetch.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep" },
+ // `bbbb0000002` is in the roster and NOT in the listing any more.
+ listed: ["aaaa0000001"],
+ roster: ["aaaa0000001", "bbbb0000002"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ bbbb0000002: { title: "drop this one" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(
+ (snap.missingNeverFetched ?? []).map((m) => m.id),
+ [],
+ );
+ // It is still settled, and still reported as such — it left the alarm, not
+ // the report.
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("an unsettled video that left the listing still raises the alarm", async () => {
+ // The control for the test above: subtracting the settled set must not have
+ // made the category unreachable.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: [],
+ roster: ["aaaa0000001"],
+ scanned: { aaaa0000001: { title: "keep this one" } },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(
+ (snap.missingNeverFetched ?? []).map((m) => m.id),
+ ["aaaa0000001"],
+ );
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a stream with no chat replay leaves chatOnlyPending for good", async () => {
+ // A clean pass that found nothing is an ANSWER: chat replay was off, and
+ // asking again tomorrow gets the same nothing. Without the flag the id sits
+ // in chatOnlyPending forever and every runner restart re-prefetches it to run
+ // a pass that will never return anything.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep", rejectedLivestreams: "chat-only" },
+ listed: ["cccc0000003"],
+ scanned: {
+ cccc0000003: {
+ title: "a long stream",
+ liveStatus: "was_live",
+ noLiveChat: true,
+ },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ assert.deepEqual(snap.buckets.chatOnly, []);
+ // It settles as the ordinary filtered-out livestream it is.
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]);
+ assert.ok(!snap.undownloadedIds.includes("cccc0000003"));
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("a chat already on disk stays a member after the mode is turned off", async () => {
+ // buildIndex has already published it as a chat track, and flipping the mode
+ // back to "skip" does not un-publish it. Deciding this off `chatOnlyIds`
+ // would make the same directory a settled STUB — and settledOnDisk is
+ // subtracted from totals.videos, so the site would serve a video the report
+ // had stopped counting.
+ const { dir, paths, channelDir } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: ["cccc0000003"],
+ scanned: {
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ const videoDir = path.join(channelDir, "data", "cccc0000003");
+ await mkdir(videoDir, { recursive: true });
+ await writeFile(
+ path.join(videoDir, "metadata.info.json"),
+ JSON.stringify({ id: "cccc0000003", title: "a long stream" }),
+ );
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"action":{}}\n',
+ );
+
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]);
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, []);
+ assert.equal(snap.totals.videos, 1);
+ // And nothing outstanding: the mode is off, so there is no chat to fetch.
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -8,6 +8,7 @@ import {
isVideoFetched,
isVideoTranscribed,
readVideoFiles,
+ LIVE_CHAT_FILENAME,
VTT_FILENAME,
type VideoFiles,
} from "../lib/videoStatus";
@@ -48,6 +49,7 @@ import {
import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
import {
+ chatOnlyIdsFrom,
loadMetadataScan,
metadataScanWanted,
settledIdsFrom,
@@ -213,6 +215,32 @@ export type ChannelSnapshot = {
// re-appear as ordinary undownloaded videos on the next report.
// Optional: older snapshots lack it; readers must default to [].
skippedByTitleFilter: string[];
+ // CHAT-ONLY videos whose chat is on disk. A rejected livestream on a channel
+ // whose `downloadFilter.rejectedLivestreams` is "chat-only": the media was
+ // never fetched, the live chat was, and the directory holds
+ // metadata.info.json, transcript.live_chat.json and its normalized
+ // live_chat.cues.json, beside the download-outcome.json and download.log
+ // every managed download leaves. No media, and no transcript.
+ //
+ // THE FILE ON DISK DECIDES, not the mode: an id whose chat has landed stays
+ // in this bucket after `rejectedLivestreams` is set back to "skip",
+ // because buildIndex has already published it and flipping a setting does
+ // not un-publish it.
+ //
+ // A MEMBER OF THIS CORPUS, not a stub. It is in totals.videos, it is in the
+ // LMDB index, and the site publishes it as a chat track with no captions.
+ // It is deliberately NOT in skippedByTitleFilter (that bucket means "we
+ // decided not to have it"), NOT in noTranscript or downloadedNoTranscript
+ // (nothing is going to transcribe a video whose audio we chose not to
+ // fetch), and NOT in undownloadedIds. Optional: older snapshots lack it;
+ // readers must default to [] — which normalizeBuckets does, like every
+ // other bucket declared required here and absent from older files.
+ chatOnly: string[];
+ // The same population, chat NOT yet on disk — one of the download lane's
+ // buckets (DOWNLOAD_BUCKETS). It cannot ride in undownloadedIds: that list
+ // means "fetch the media", and fetching the media is the one thing this
+ // video must not have done to it. Absent from older files in the same way.
+ chatOnlyPending: string[];
// Videos whose transcript covers only a small fraction of the video's
// duration — the audio download silently truncated (yt-dlp exited "ok") so
// whisper transcribed just the first few minutes. The detection threshold
@@ -638,9 +666,14 @@ async function readPlaylistUrls(file: string): Promise<string[]> {
}
}
-function videoHasAnyArtifact(files: VideoFiles): boolean {
- return files.hasYtVtt || files.hasWhisper || files.audioFiles.length > 0;
-}
+// WAS A PRIVATE COPY OF isVideoDownloaded, byte for byte, and this module
+// already imported that one (it is what increments `downloaded` in the same
+// walk). Two names for one predicate is how the settled short-circuit and the
+// totals end up disagreeing about what an artifact is, so the copy is gone and
+// the name stays as an alias: "does this dir hold anything we fetched?" reads
+// better at the five call sites than "is it downloaded?", and it is now
+// guaranteed to be the same question.
+const videoHasAnyArtifact = isVideoDownloaded;
export async function generateChannelSnapshot(
paths: Paths,
@@ -925,6 +958,11 @@ export async function generateChannelSnapshot(
// whole channel on the next report, with no rescan and nothing to migrate.
const metadataScanStore = await loadMetadataScan(paths, slug);
const settledIds = settledIdsFrom(metadataScanStore, config);
+ // A STRICT SUBSET of the settled set: the rejections the operator asked for
+ // the live chat of (downloadFilter.rejectedLivestreams === "chat-only").
+ // Settled means the MEDIA is not wanted, which is true of these too — what
+ // this adds is that the video is still a corpus member, as a chat track.
+ const chatOnlyIds = chatOnlyIdsFrom(metadataScanStore, config);
const noTranscript: string[] = [];
const downloadedNoTranscript: string[] = [];
@@ -942,6 +980,12 @@ export async function generateChannelSnapshot(
// their metadata before deciding). Tracked separately from the full settled
// set only so totals.videos can subtract exactly the dirs it counted.
const settledOnDisk: string[] = [];
+ // Chat-only videos whose chat is on disk — corpus members, counted in
+ // totals.videos, and out of every bucket that means "something is missing".
+ const chatOnly: string[] = [];
+ // Chat-only videos whose chat is NOT on disk yet: the download lane's work,
+ // and the reason this is a bucket rather than a derived count.
+ const chatOnlyPending: string[] = [];
const incompleteTranscript: string[] = [];
const shortAudio: string[] = [];
const autoSubsOnly: string[] = [];
@@ -1079,8 +1123,27 @@ export async function generateChannelSnapshot(
// it from transcribedWithAudio and the cleanup estimates while it still
// counted in totals.transcribed, i.e. a channel reporting more transcripts
// than videos. Settlement only ever decides what NOT to fetch.
+ // SETTLED, and the two answers it can have.
+ //
+ // THE CHAT ON DISK DECIDES, NOT THE MODE. A dir holding
+ // transcript.live_chat.json is already published by buildIndex as a chat
+ // track with no captions — that is a fact about the corpus, and flipping
+ // `rejectedLivestreams` back to "skip" does not un-publish it. Asking
+ // `chatOnlyIds` here instead would make the same directory count as a
+ // settled STUB the moment the mode changed, and settledOnDisk is subtracted
+ // from totals.videos: the site would still serve the video while the report
+ // stopped counting it. So the file is the test, and the Diagnostics copy —
+ // "what is already on disk stays" — is true.
+ //
+ // Short-circuited for the same reason the settled branch is: everything
+ // under this line classifies a video by what is MISSING, and nothing is
+ // missing from a chat-only video.
if (settledIds.has(id) && !videoHasAnyArtifact(files)) {
- settledOnDisk.push(id);
+ if (files.entries.includes(LIVE_CHAT_FILENAME)) {
+ chatOnly.push(id);
+ } else {
+ settledOnDisk.push(id);
+ }
continue;
}
if (!files.hasMeta && !excludedById.has(id)) noMetadata.push(id);
@@ -1275,6 +1338,12 @@ export async function generateChannelSnapshot(
urls.map((u) => extractVideoId(u)).filter((id): id is string => Boolean(id)),
),
onDiskIds: new Set(videoDirNames),
+ // See deriveChannelSets: both sets it derives are defined by the ABSENCE of
+ // a directory, and a settled video has none since a rejection stopped
+ // leaving its prefetch dir behind. Without this, `missingNeverFetched` —
+ // the loudest alarm this report raises — would fire for videos the operator
+ // asked us not to fetch.
+ settledIds,
});
const listedIdSet = new Set(sets.listed);
@@ -1307,6 +1376,19 @@ export async function generateChannelSnapshot(
) {
metadataScanUnscanned++;
}
+ // CHAT ONLY: the media is settled (so this id never reaches
+ // undownloadedIds) but the CHAT may still be outstanding, and that is real
+ // download-lane work with no other home — the id has no artifact, so no
+ // artifact-derived bucket can carry it. Once the chat lands it drops out
+ // here and appears in `chatOnly` instead.
+ if (chatOnlyIds.has(dirId)) {
+ // Already a member (the chat is on disk) => nothing outstanding. The scan
+ // store's `noLiveChat` flag is the other way out of this list: a stream
+ // we asked and got nothing from is not in `chatOnlyIds` at all, so it
+ // settles as an ordinary rejection instead of sitting here forever.
+ if (!f?.entries.includes(LIVE_CHAT_FILENAME)) chatOnlyPending.push(dirId);
+ continue;
+ }
// A settled video has no artifact and never will while the filter stands.
// This is the line that makes the settlement STICK: undownloadedIds is the
// auto-download runner's work queue, and it is derived from artifacts, so an
@@ -1352,6 +1434,7 @@ export async function generateChannelSnapshot(
const corruptSourceSet = new Set(corruptSource);
const corruptFullSourceSet = new Set(corruptFullSource);
+ const chatOnlySet = new Set(chatOnly);
const snapshotBuckets: ChannelSnapshot["buckets"] = {
noTranscript: noTranscript.sort(),
downloadedNoTranscript: downloadedNoTranscript.sort(),
@@ -1374,13 +1457,26 @@ export async function generateChannelSnapshot(
nonStandardVtt: nonStandardVtt.sort(),
skippedByFilter: skippedByFilter.sort(),
// The settled set minus anything already on disk — same rule as the
- // short-circuit above, so the bucket and the classification cannot disagree.
+ // short-circuit above, so the bucket and the classification cannot disagree
+ // — and minus the chat-only subset, which has its own two buckets. A
+ // chat-only video IS settled (its media is not wanted) but reporting it as
+ // "skipped by the title filter" would say we decided not to have it, when
+ // we decided to have its chat.
skippedByTitleFilter: [...settledIds]
.filter((id) => {
+ // Minus the two chat buckets, whichever way an id got into them — a
+ // chat-only video IS settled (its media is not wanted) but reporting it
+ // as "skipped by the title filter" would say we decided not to have it,
+ // when we decided to have its chat. `chatOnlySet` is what the loop
+ // above actually classified (the chat is on disk); `chatOnlyIds` is the
+ // outstanding half.
+ if (chatOnlySet.has(id) || chatOnlyIds.has(id)) return false;
const f = filesById.get(id);
return !f || !videoHasAnyArtifact(f);
})
.sort(),
+ chatOnly: chatOnly.sort(),
+ chatOnlyPending: chatOnlyPending.sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
autoSubsOnly: autoSubsOnly.sort(),
diff --git a/common/controller/metadataScanJob.ts b/common/controller/metadataScanJob.ts
@@ -0,0 +1,80 @@
+// THE METADATA SCAN AS A JOB — one definition, two dispatchers.
+//
+// The scan itself is `ytdlp/metadataScan.ts`, which is pure work: it takes a
+// channel and a config and reads titles. What it is NOT is a job — the kind,
+// the queue key, the replay spec and the platform cooldown that surrounds it
+// all lived in the editor's server action, which is the one place the auto
+// runner cannot reach.
+//
+// So the JOB moves here and the editor action keeps what is genuinely its
+// own: the cooldown pre-check it answers to a click with (a returned
+// `{ ok: false, info: true }` sentence, not a queued job nobody watches) and
+// its three `revalidatePath` calls. The runner passes neither. What both get
+// from this module is identical: the same `kind`, the same per-platform queue
+// key, the same replay spec, and the same `onPlatformBackoff` wiring into the
+// SHARED cooldown the download lane already honours — which is what stops the
+// runner and a manual sync from arguing about whether YouTube is angry.
+//
+// `needsMedia` comes from the kind (jobs/jobKinds.ts declares it TRUE for this
+// one, because the scan's target set is "listed, minus what is already on
+// disk"), so `runManagedFunction` refuses the job against an unmounted drive
+// for either caller with nothing said here.
+
+import { detectPlatform } from "../lib/platform";
+import { downloadQueueKey, resolveQueueKey } from "../lib/queueKeys";
+import { recordDownloadBackoff } from "../jobs/downloadBackoff";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "../jobs/streamCommand";
+import type { ChannelConfig } from "../lib/channelConfig";
+import type { Paths } from "../lib/paths";
+import { runMetadataScan } from "../ytdlp/metadataScan";
+
+export const METADATA_SCAN_JOB_KIND = "metadata-scan";
+
+export type MetadataScanJobOpts = {
+ paths: Paths;
+ slug: string;
+ channelConfig: ChannelConfig;
+ // Per-run override of the platform queue key (the editor threads one through
+ // from the pipeline action's own queue plumbing).
+ queueKey?: string;
+ // Run in the background so a clicked action preempts it in the queue. The
+ // runner passes true for the same reason its download units do.
+ background?: boolean;
+ // Called after a successful scan, inside the job. The editor revalidates its
+ // three paths here; the runner asks for a snapshot regen instead.
+ afterRun?: () => void | Promise<void>;
+};
+
+export async function runMetadataScanJob(
+ opts: MetadataScanJobOpts,
+): Promise<StreamActionResult> {
+ const { paths, slug, channelConfig } = opts;
+ const platform = detectPlatform(channelConfig.url) ?? "unknown";
+ return runManagedFunction({
+ kind: METADATA_SCAN_JOB_KIND,
+ queueKey: resolveQueueKey(downloadQueueKey(channelConfig), opts.queueKey),
+ paths,
+ channelSlug: slug,
+ ...(opts.background ? { background: true } : {}),
+ spec: {
+ kind: METADATA_SCAN_JOB_KIND,
+ slug,
+ params: { queueKey: opts.queueKey },
+ },
+ fn: async (onLog, signal, setProgress) => {
+ await runMetadataScan({
+ paths,
+ channelSlug: slug,
+ channelConfig,
+ onLog,
+ signal,
+ setProgress,
+ onPlatformBackoff: () => recordDownloadBackoff(platform, paths),
+ });
+ await opts.afterRun?.();
+ },
+ });
+}
diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts
@@ -30,7 +30,11 @@ import path from "node:path";
import { readFile, rename, writeFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import type { ChannelConfig } from "../lib/channelConfig";
-import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters";
+import {
+ compileDownloadFilter,
+ titleFilterRejects,
+ titleFilterWantsChat,
+} from "../lib/downloadFilters";
export const METADATA_SCAN_FILENAME = "metadata-scan.json";
export const METADATA_SCAN_VERSION = 1;
@@ -45,6 +49,14 @@ export type MetadataScanEntry = {
uploadDate: string;
liveStatus?: string;
duration?: number;
+ // THE CHAT-ONLY DEAD END. Set when a chat-only pass ran cleanly and the
+ // source returned no live chat — replay was off for that stream, or it has
+ // since been dropped. Asking again tomorrow gets the same nothing, so this is
+ // an ANSWER and not a retry: `chatOnlyIdsFrom` drops the id, it leaves
+ // `chatOnlyPending` for good, and it settles as the ordinary filtered-out
+ // livestream it is. Deliberately NOT set by a pass that FAILED (a non-zero
+ // exit, a cooldown, an abort) — that one has to stay retryable.
+ noLiveChat?: boolean;
scannedAt: string;
};
@@ -103,6 +115,7 @@ function normalizeEntry(raw: unknown): MetadataScanEntry | null {
if (typeof r.duration === "number" && Number.isFinite(r.duration)) {
entry.duration = r.duration;
}
+ if (r.noLiveChat === true) entry.noLiveChat = true;
return entry;
}
@@ -224,7 +237,8 @@ export async function upsertMetadataScan(
prev.description !== entry.description ||
prev.uploadDate !== entry.uploadDate ||
prev.liveStatus !== entry.liveStatus ||
- prev.duration !== entry.duration
+ prev.duration !== entry.duration ||
+ prev.noLiveChat !== entry.noLiveChat
) {
scan.entries[id] = entry;
changed = true;
@@ -282,6 +296,34 @@ export function settledIdsFrom(
return settled;
}
+// THE CHAT-ONLY SUBSET OF THE SETTLED SET — a strict subset, never a second
+// population. `titleFilterRejects` answers true for a chat-only video (see
+// DownloadFilterVerdict), so every id here is also in `settledIdsFrom`'s
+// answer; what this adds is which of them the operator wants the live chat of.
+//
+// Derived from the config every time, exactly like settledIdsFrom, so turning
+// `rejectedLivestreams` off again re-decides the channel with no rescan and
+// nothing stored per video to undo.
+export function chatOnlyIdsFrom(
+ scan: MetadataScan,
+ config: Pick<ChannelConfig, "downloadFilter"> | null | undefined,
+): Set<string> {
+ const ids = new Set<string>();
+ const compiled = compileDownloadFilter(config?.downloadFilter);
+ // The mode only ever means something alongside a real filter, and a channel
+ // that is not chat-only pays one field read for this question.
+ if (!compiled || compiled.rejectedLivestreams !== "chat-only") return ids;
+ for (const [id, entry] of Object.entries(scan.entries)) {
+ // A stream we already asked and got nothing from is not chat-only work any
+ // more — it is an ordinary settled rejection. Without this it sits in
+ // chatOnlyPending forever and every runner restart re-prefetches it to run
+ // a pass that will never return anything.
+ if (entry.noLiveChat) continue;
+ if (titleFilterWantsChat(compiled, entry)) ids.add(id);
+ }
+ return ids;
+}
+
// The loading form, for callers that have no scan in hand.
export async function settledByTitleFilterIds(
paths: Paths,
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -329,9 +329,13 @@ test("bucketsForKind: per-kind ordered bucket lists", () => {
[...bucketsForKind("transcription")],
["downloadedNoTranscript", "failedListed"],
);
+ // `chatOnlyPending` is LAST: a chat fetch is the cheapest work on the lane and
+ // must never delay a real download. It is [] for every channel that has not
+ // set downloadFilter.rejectedLivestreams, so the order below is unchanged for
+ // all of them.
assert.deepEqual(
[...bucketsForKind("download")],
- ["partialDownloads", "undownloadedIds"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending"],
);
});
@@ -427,7 +431,7 @@ test("the default union is unchanged by the opt-in buckets", () => {
);
assert.deepEqual(
[...bucketsForKind("download")],
- ["partialDownloads", "undownloadedIds"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending"],
);
assert.deepEqual(
[...defaultBucketsForPolicy("transcription", { replaceAutoSubs: false })],
@@ -446,7 +450,7 @@ test("selectableBucketsForKind offers defaults plus the opt-in buckets", () => {
);
assert.deepEqual(
[...selectableBucketsForKind("download")],
- ["partialDownloads", "undownloadedIds", "autoSubsOnly"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending", "autoSubsOnly"],
);
});
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -79,7 +79,17 @@ export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder {
// its internal priority. Single source of truth for the runner, the pending-
// count helper, and the editor's bucket picker.
export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const;
-export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const;
+// `chatOnlyPending` is LAST on purpose: it is the smallest and cheapest work on
+// the lane (one --skip-download pass per video, no media), and putting it ahead
+// of real downloads would let a chat backlog delay the corpus. Empty for every
+// channel that has not set `downloadFilter.rejectedLivestreams` — which is
+// every channel that predates the field — so the fold, the lane's work list and
+// the pick order are byte-identical for them.
+export const DOWNLOAD_BUCKETS = [
+ "partialDownloads",
+ "undownloadedIds",
+ "chatOnlyPending",
+] as const;
// Buckets a runner will NOT draw from unless asked. Replacing YouTube's
// auto-captions with our own transcript costs an audio download plus a
@@ -303,6 +313,19 @@ export type ChannelWork = {
// Optional: every projection written before operations existed omits it, and
// a leaf naming an operation simply finds nothing.
operations?: Record<string, string[]>;
+ // THE CHANNEL-SCOPED WORK A LEAF CANNOT NAME. The metadata scan is a fact
+ // about the CHANNEL, not about a video — there is no id to put in a bucket,
+ // because the whole point of the scan is that these videos have no directory
+ // — so the download lane's pre-pick reads it off the projection here rather
+ // than through `pick()`. Both optional: a projection that omits them offers
+ // no scan, which is what every caller but the download runner wants.
+ //
+ // `scanUnscanned` is the snapshot's own `metadataScan.unscanned`, which is
+ // defined to mean exactly what metadataScanTargets() will fetch.
+ // `filtered` is "this channel has a download filter that compiles" — without
+ // one a scan settles nothing and is pure cost against the source.
+ scanUnscanned?: number;
+ filtered?: boolean;
};
// `retainLeaves(pending, root, "buckets" | "operations")` USED TO LIVE HERE, and
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -67,8 +67,31 @@ export type DownloadFilterConfig = {
// `{ includeLivestreams: true }` alone rejects plain uploads and passes
// livestreams. See titleFilterRejects.
includeLivestreams?: boolean;
+ // WHAT TO DO WITH A LIVESTREAM THE FILTER REJECTED. Absent = "skip", which
+ // is every channel that predates this field and is byte-for-byte what they
+ // did before it existed.
+ //
+ // "chat-only" is the middle answer that did not exist: a multi-hour stream
+ // whose TITLE says nothing about the subject is usually not worth its audio,
+ // but its live chat is text, it is small, and it is the only record of what
+ // the room said. So the video is not downloaded, its chat is, and it joins
+ // the corpus as a chat track with no captions — NOT as a downloaded video.
+ // See classifyAgainstFilter for why this is a third VERDICT rather than a
+ // flag read at download time.
+ rejectedLivestreams?: RejectedLivestreamMode;
};
+// Absent behaves as "skip". Spelled as a union rather than a boolean because a
+// third answer (keeping the audio at a lower quality, say) is a plausible next
+// one and a boolean would have to be migrated to make room for it.
+export type RejectedLivestreamMode = "skip" | "chat-only";
+
+export function isRejectedLivestreamMode(
+ v: unknown,
+): v is RejectedLivestreamMode {
+ return v === "skip" || v === "chat-only";
+}
+
export type ChannelConfig = {
handling: ChannelHandling;
// Omitted = "video" (every channel that predates the posts corpus).
@@ -365,11 +388,19 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null {
const include = typeof df.include === "string" ? df.include.trim() : "";
const exclude = typeof df.exclude === "string" ? df.exclude.trim() : "";
const includeLivestreams = df.includeLivestreams === true;
+ // ONLY ALONGSIDE A REAL FILTER, and only when it is not the default. The
+ // mode says what to do with a REJECTED livestream, so with nothing to
+ // reject it names a decision that can never be taken — storing it would put
+ // a setting on the Configure form that does nothing and explains nothing.
+ const rejectedLivestreams = isRejectedLivestreamMode(df.rejectedLivestreams)
+ ? df.rejectedLivestreams
+ : "skip";
if (include || exclude || includeLivestreams) {
config.downloadFilter = {
...(include ? { include } : {}),
...(exclude ? { exclude } : {}),
...(includeLivestreams ? { includeLivestreams: true } : {}),
+ ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
};
}
}
diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts
@@ -9,6 +9,7 @@ import {
downloadFilterText,
evaluateDownloadFilters,
titleFilterRejects,
+ titleFilterWantsChat,
type DownloadFilterContext,
} from "./downloadFilters";
import type { RawMetadata } from "./transcripts-server";
@@ -409,3 +410,91 @@ test("the matched description is capped", () => {
true,
);
});
+
+// --- rejectedLivestreams: the chat-only tier --------------------------------
+//
+// The whole design risk of this feature is one sentence: "chat-only" is a
+// REJECTION with an instruction attached, not a fourth way to pass. Everything
+// that asks "is this video's media wanted?" must keep answering no for it, or
+// the downloader fetches the very media the operator said not to.
+
+test("chat-only is a rejection, and titleFilterRejects still says so", () => {
+ const f = compileDownloadFilter({
+ include: "guest",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ const stream = liveMeta("was_live");
+ assert.equal(classifyAgainstFilter(f, stream), "chat-only");
+ // THE LOAD-BEARING ONE. settledIdsFrom and therefore undownloadedIds are
+ // built on this: a chat-only video is settled exactly like any other
+ // rejection, so the download queue never offers its media.
+ assert.equal(titleFilterRejects(f, stream), true);
+ assert.equal(titleFilterWantsChat(f, stream), true);
+});
+
+test("it applies to livestreams ONLY, and only when configured", () => {
+ const f = compileDownloadFilter({
+ include: "guest",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ // A rejected plain upload is a plain rejection: there is no chat to keep.
+ assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected");
+ assert.equal(titleFilterWantsChat(f, liveMeta("not_live")), false);
+ // A video the filter WANTS is unaffected either way.
+ assert.equal(
+ classifyAgainstFilter(f, liveMeta("was_live", "a guest appears")),
+ "text",
+ );
+
+ // The default, and every channel written before the field: absent = skip.
+ const plain = compileDownloadFilter({ include: "guest" });
+ assert.ok(plain);
+ assert.equal(plain.rejectedLivestreams, "skip");
+ assert.equal(classifyAgainstFilter(plain, liveMeta("was_live")), "rejected");
+ assert.equal(titleFilterWantsChat(plain, liveMeta("was_live")), false);
+});
+
+test("an EXCLUDE-matched livestream is chat-only too, because the mode is about the rejection", () => {
+ // The setting says what to do with a rejected livestream, not which selector
+ // did the rejecting. One rule, in one place, so the two cannot diverge.
+ const f = compileDownloadFilter({
+ exclude: "rerun",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ assert.equal(
+ classifyAgainstFilter(f, liveMeta("was_live", "rerun of last night")),
+ "chat-only",
+ );
+ // Exclude-only is still "everything else passes" — the mode adds no filtering.
+ assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text");
+});
+
+test("the mode alone is not a filter, and is never the reason a channel is filtered", () => {
+ // Nothing rejecting means nothing for the mode to say, so it does not make a
+ // channel filtered — which is also what keeps the download lane from offering
+ // a metadata scan to a channel that has no filter at all.
+ assert.equal(compileDownloadFilter({ rejectedLivestreams: "chat-only" }), null);
+});
+
+test("evaluateDownloadFilters marks the decision, and it is still a skip", () => {
+ const decision = evaluateDownloadFilters(
+ ctx({
+ metadata: meta({ title: "Synthetic plainvid0001", live_status: "was_live" }),
+ channelConfig: {
+ ...BASE_CONFIG,
+ downloadFilter: { include: "guest", rejectedLivestreams: "chat-only" },
+ },
+ }),
+ );
+ assert.ok(decision);
+ // `skip` is what every media-fetching path reads, and it is unchanged: the
+ // downloader is the one caller that asks the next question.
+ assert.equal(decision.skip, true);
+ assert.equal(decision.filter, "titleFilter");
+ assert.equal(decision.chatOnly, true);
+ assert.match(decision.reason, /live chat only/);
+});
+
diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts
@@ -10,7 +10,11 @@
// Adding a filter = append one entry to FILTERS. Each filter sees the same
// context (metadata + channel config + resolved settings).
-import type { ChannelConfig, DownloadFilterConfig } from "./channelConfig";
+import type {
+ ChannelConfig,
+ DownloadFilterConfig,
+ RejectedLivestreamMode,
+} from "./channelConfig";
import type { RawMetadata } from "./transcripts-server";
export type DownloadFilterSettings = {
@@ -32,6 +36,11 @@ export type DownloadFilterDecision = {
skip: boolean;
filter: string;
reason: string;
+ // The video's MEDIA is skipped and its live chat is wanted — see
+ // DownloadFilterConfig.rejectedLivestreams. `skip` is still true: everything
+ // that decides whether to fetch the media reads that and is unchanged. The
+ // downloader is the one caller that asks the next question.
+ chatOnly?: boolean;
};
type DownloadFilter = {
@@ -80,6 +89,9 @@ export type CompiledDownloadFilter = {
include: RegExp | null;
exclude: RegExp | null;
includeLivestreams: boolean;
+ // See DownloadFilterConfig.rejectedLivestreams. "skip" is the default and is
+ // what every channel written before the field did.
+ rejectedLivestreams: RejectedLivestreamMode;
};
// Compile a channel's filter. Returns null when there is no filter AND when a
@@ -96,11 +108,17 @@ export function compileDownloadFilter(
const exclude = filter?.exclude?.trim() ?? "";
const includeLivestreams = filter?.includeLivestreams === true;
if (!include && !exclude && !includeLivestreams) return null;
+ // The mode is NOT a positive selector and deliberately does not make a
+ // channel "filtered" on its own: it says what to do with a rejection, and
+ // with nothing rejecting there is nothing for it to say.
+ const rejectedLivestreams: RejectedLivestreamMode =
+ filter?.rejectedLivestreams === "chat-only" ? "chat-only" : "skip";
try {
return {
include: include ? new RegExp(include, "i") : null,
exclude: exclude ? new RegExp(exclude, "i") : null,
includeLivestreams,
+ rejectedLivestreams,
};
} catch {
return null;
@@ -192,7 +210,18 @@ export function downloadFilterPatternProblem(pattern: string): string | null {
// Why a video passed, or that it didn't. "livestream" exists so the UI can say
// how many videos a channel is keeping for a reason other than their name.
-export type DownloadFilterVerdict = "text" | "livestream" | "rejected";
+//
+// "chat-only" IS A REJECTION with an instruction attached, not a fourth way to
+// pass. The video is not wanted; its live chat is. Everything that asks "is
+// this video wanted?" — `titleFilterRejects`, and therefore the settled set and
+// the download queue — must keep answering yes-it-is-rejected for it, or the
+// downloader would fetch the media the operator asked it not to. Only the
+// callers that ask the NEXT question ("and then what?") look for this value.
+export type DownloadFilterVerdict =
+ | "text"
+ | "livestream"
+ | "rejected"
+ | "chat-only";
// PURE, and the SINGLE matcher. Both the download-time registry entry below and
// the derived settled set (controller/metadataScanStore.ts) call exactly this,
@@ -217,19 +246,50 @@ export function classifyAgainstFilter(
meta: FilterableVideo,
): DownloadFilterVerdict {
const text = downloadFilterText(meta);
- if (compiled.exclude && compiled.exclude.test(text)) return "rejected";
+ if (compiled.exclude && compiled.exclude.test(text)) {
+ return rejection(compiled, meta);
+ }
const hasPositive = Boolean(compiled.include || compiled.includeLivestreams);
if (!hasPositive) return "text";
if (compiled.include && compiled.include.test(text)) return "text";
if (compiled.includeLivestreams && isLivestream(meta)) return "livestream";
- return "rejected";
+ return rejection(compiled, meta);
}
+// HOW a rejection is spelled — the ONE place, so `exclude`-matched and
+// unmatched livestreams cannot get different answers. The operator's setting is
+// about what to do with a rejected livestream, not about which selector did the
+// rejecting.
+function rejection(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): DownloadFilterVerdict {
+ return compiled.rejectedLivestreams === "chat-only" && isLivestream(meta)
+ ? "chat-only"
+ : "rejected";
+}
+
+// IS THIS VIDEO'S MEDIA UNWANTED? True for "chat-only" too — see
+// DownloadFilterVerdict. This is what `settledByTitleFilterIds` and therefore
+// `undownloadedIds` are built on, so a chat-only video is settled exactly like
+// any other rejection and the download queue never offers its media.
export function titleFilterRejects(
compiled: CompiledDownloadFilter,
meta: FilterableVideo,
): boolean {
- return classifyAgainstFilter(compiled, meta) === "rejected";
+ const verdict = classifyAgainstFilter(compiled, meta);
+ return verdict === "rejected" || verdict === "chat-only";
+}
+
+// Is this one the operator wants the CHAT of? A separate question from the one
+// above, asked by exactly the two places that act on the answer: the downloader,
+// which fetches the chat instead of skipping, and the snapshot, which puts the
+// video in its own bucket rather than in `skippedByTitleFilter`.
+export function titleFilterWantsChat(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): boolean {
+ return classifyAgainstFilter(compiled, meta) === "chat-only";
}
// Why a rejection happened, for the log and the outcome record.
@@ -288,10 +348,14 @@ const titleFilter: DownloadFilter = {
const m = ctx.metadata;
if (!m) return null; // fail open — see above
if (!titleFilterRejects(compiled, m)) return null;
+ const chatOnly = titleFilterWantsChat(compiled, m);
return {
skip: true,
filter: "titleFilter",
- reason: titleFilterReason(compiled, m),
+ reason: chatOnly
+ ? `${titleFilterReason(compiled, m)} — fetching its live chat only`
+ : titleFilterReason(compiled, m),
+ ...(chatOnly ? { chatOnly: true } : {}),
};
},
};
diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts
@@ -27,7 +27,16 @@ export type DownloadOutcomeStatus =
// The app-level filter pass (e.g. skip-live) declined to download this video.
// Not a failure and not archived — the next sync/download-missing retries it
// once the filter no longer matches (e.g. a live stream becomes a VOD).
- | "skipped-filtered";
+ | "skipped-filtered"
+ // The download filter rejected this livestream and the channel's
+ // `rejectedLivestreams` mode is "chat-only": the MEDIA was not fetched, the
+ // live chat was. The directory holds metadata.info.json,
+ // transcript.live_chat.json and its live_chat.cues.json (plus this sidecar
+ // and download.log) — no media and no transcript — so the video is indexable
+ // as a chat track with no captions while every "is it downloaded?" predicate
+ // — all of which test whisper/VTT/audio — still answers no. Terminal on the operator's
+ // terms, like skipped-filtered, and equally re-decided by editing the filter.
+ | "chat-only";
export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus> = [
"ok",
@@ -39,6 +48,7 @@ export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus
"corrupt-full-source",
"failed-short-audio",
"skipped-filtered",
+ "chat-only",
];
export type DownloadAttemptKind =
@@ -52,7 +62,11 @@ export type DownloadAttemptKind =
// A cookie re-run of a metadata prefetch that failed with an auth/age error
// (cookie mode "always"/"when-required" with a cookie value configured).
// Recorded with n: 0 alongside the failed prefetch it retries.
- | "metadata-prefetch-auth-retry";
+ | "metadata-prefetch-auth-retry"
+ // The live-chat pass a "chat-only" filter verdict runs INSTEAD of a download:
+ // --skip-download --write-subs --sub-langs live_chat, no media, no archive
+ // line. Recorded with n: 1, since it is the only real attempt there is.
+ | "live-chat-only";
export type AudioCheckProbeVerdict = "clean" | "partial" | "malformed";
diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts
@@ -1245,10 +1245,16 @@ test("`runner` names the auto-queue runner, and only for the two that have one",
const runners = new Map(operationCatalog().map((o) => [o.id, o.runner]));
assert.equal(runners.get("download"), "download");
assert.equal(runners.get("transcription"), "transcription");
+ // THE THIRD ONE IS CHANNEL-SCOPED, and that is the point of it being here
+ // rather than inferred: the download runner dispatches the metadata scan as a
+ // channel-scoped unit before it picks any video, so the scan's console — and
+ // its pause — are the download lane's.
+ assert.equal(runners.get("metadata-scan"), "download");
// Everything the sweep dispatches must leave it unset — a backfill kind with
// a runner would render a runner console over a lane no runner feeds.
+ const dispatched = new Set(["download", "transcription", "metadata-scan"]);
for (const op of operationCatalog()) {
- if (op.id === "download" || op.id === "transcription") continue;
+ if (dispatched.has(op.id)) continue;
assert.equal(op.runner, undefined, `${op.id} declares a runner`);
}
// Sync included, and it is the interesting one: the sync scheduler's
diff --git a/common/lib/operations.ts b/common/lib/operations.ts
@@ -1360,10 +1360,23 @@ export const SYNC_OPERATION: OperationDescriptor = {
// download: it is one metadata request per listed video, and the source counts
// them the same way it counts a download's.
//
-// NO `runner`, deliberately: nothing auto-dispatches this yet. The operator
-// presses Run. The obvious follow-up is for the download lane to run it for a
-// channel whose filter has unscanned listed videos, which is exactly the
-// backlog this declares — see plans/FACTS.md.
+// THE DOWNLOAD LANE DISPATCHES IT, and that is what `runner` says. The runner
+// checks, before every pick, whether a filtered channel has unscanned listed
+// videos — exactly the backlog this entry declares — and runs the scan for it
+// as one channel-scoped unit on the platform download queue, holding the same
+// per-platform slot a download unit would.
+//
+// It goes FIRST because it decides what the downloads are. Every unscanned
+// non-match on a filtered channel is otherwise prefetched, rejected and
+// discarded one yt-dlp invocation at a time, against a source whose patience is
+// the scarce resource; one batch answers the whole channel.
+//
+// WHAT NAMING THE RUNNER ALSO CHANGES: `pauseLaneFor` asks `runner` first, so
+// this operation's console is the download lane's and the download pause now
+// holds the AUTO-dispatch. The operator's own Run button is unaffected — it
+// checks the platform cooldown and nothing else — which keeps the original
+// point of the exemption intact: you can still scan, while downloads are
+// paused, precisely to decide what the lane should fetch when it resumes.
export const METADATA_SCAN_OPERATION: OperationDescriptor = {
id: "metadata-scan",
label: "Metadata scan",
@@ -1375,6 +1388,7 @@ export const METADATA_SCAN_OPERATION: OperationDescriptor = {
dispatch: "external",
scope: "channel",
trigger: "backlog",
+ runner: "download",
};
export const EXTERNAL_OPERATIONS: readonly ExternalOperation[] = [
diff --git a/common/lib/pauseGates.test.ts b/common/lib/pauseGates.test.ts
@@ -167,11 +167,14 @@ test("pauseLaneFor answers for every catalog id", () => {
// Catalogued, and deliberately gateless: the sync scheduler's own `enabled`
// is its switch, and its queue key is neither sweep's.
sync: null,
- // Catalogued and gateless for a different reason: the metadata scan rides
- // the platform download queue but is NOT held by the download gate. It
- // fetches no media, and the operator runs it precisely to decide what a
- // paused download lane should fetch when it resumes.
- "metadata-scan": null,
+ // THE DOWNLOAD LANE'S, because the download runner is what dispatches it.
+ // It used to be null, on the reasoning that a scan fetches no media and the
+ // operator runs it precisely to decide what a paused lane should fetch when
+ // it resumes. That reasoning is intact and now belongs to the RUN BUTTON,
+ // which checks the platform cooldown and no gate at all: pausing downloads
+ // stops the lane from auto-dispatching a scan, and stops nothing the
+ // operator does by hand.
+ "metadata-scan": "download",
};
const ids = operationCatalog().map((o) => o.id);
assert.equal(ids.length, 8);
diff --git a/common/views/pipeline/channelFlow.ts b/common/views/pipeline/channelFlow.ts
@@ -413,6 +413,17 @@ export function computeChannelFlow(
"diagnostics",
"Declined as currently live or upcoming; retried on a later sync.",
),
+ // THE GAP IS REAL AND IT IS THIS STATION'S. A chat-only video is not
+ // settled out of the listing (it is not in skippedByTitleFilter), so it
+ // counts as expected here — and until its chat lands it has no directory,
+ // so it is missing from totals.videos. That is exactly the gap a siding
+ // exists to name.
+ ...siding(
+ "chat only, not fetched",
+ buckets.chatOnlyPending.length,
+ "diagnostics",
+ "Livestreams the filter rejected on a channel set to keep the chat. The lane fetches the live chat last — after every real download — because it costs no media.",
+ ),
...siding(
"need cookies",
diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts
@@ -39,6 +39,8 @@ export function normalizeBuckets(
nonStandardVtt: raw?.nonStandardVtt ?? [],
skippedByFilter: raw?.skippedByFilter ?? [],
skippedByTitleFilter: raw?.skippedByTitleFilter ?? [],
+ chatOnly: raw?.chatOnly ?? [],
+ chatOnlyPending: raw?.chatOnlyPending ?? [],
incompleteTranscript: raw?.incompleteTranscript ?? [],
shortAudio: raw?.shortAudio ?? [],
autoSubsOnly: raw?.autoSubsOnly ?? [],
diff --git a/common/ytdlp/downloadOneManaged.test.ts b/common/ytdlp/downloadOneManaged.test.ts
@@ -0,0 +1,140 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdir, mkdtemp, readdir, rm, writeFile } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import { __discardPrefetchDirForTest as discardPrefetchDir } from "./downloadOneManaged";
+
+// THE DISCARD IS A RECURSIVE DELETE, so the only test worth having is the one
+// that asks what it REFUSES.
+//
+// It used to be a deny-list naming isVideoDownloaded, `clips/` and
+// `saved-video.json`, which is a shape that cannot be right: every file nobody
+// thought of is deleted by default. The cases below are the ones that were
+// actually reachable — a chat-only corpus member is the worst of them, because
+// the chat-only branch calls this whenever its chat pass came back with no
+// file, so a failed re-fetch would have deleted the chat that was already
+// there.
+//
+// Each case is one directory, one call, one question: did it survive?
+
+async function dirWith(
+ files: Record<string, string>,
+ dirs: string[] = [],
+): Promise<{ root: string; videoDir: string }> {
+ const root = await mkdtemp(path.join(tmpdir(), "ttb-discard-"));
+ const videoDir = path.join(root, "data", "vid0000001");
+ await mkdir(videoDir, { recursive: true });
+ for (const [name, body] of Object.entries(files)) {
+ await writeFile(path.join(videoDir, name), body);
+ }
+ for (const name of dirs) await mkdir(path.join(videoDir, name));
+ return { root, videoDir };
+}
+
+const noop = () => {};
+
+async function discardOf(
+ files: Record<string, string>,
+ dirs: string[] = [],
+): Promise<{ removed: boolean; survivors: string[] }> {
+ const { root, videoDir } = await dirWith(files, dirs);
+ try {
+ const removed = await discardPrefetchDir(videoDir, noop);
+ const survivors = await readdir(videoDir).catch(() => null);
+ return { removed, survivors: survivors ?? [] };
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+}
+
+test("a directory holding nothing but this pass's own files is removed", async () => {
+ // Exactly what a rejected prefetch leaves: the info json it wrote, the
+ // managed download's log, and the outcome sidecar an EARLIER attempt on the
+ // same id left behind (including one a pre-rule rejection wrote).
+ const { removed, survivors } = await discardOf({
+ "metadata.info.json": "{}",
+ "download.log": "yt-dlp\n",
+ "download-outcome.json": '{"status":"skipped-filtered"}',
+ });
+ assert.equal(removed, true);
+ assert.deepEqual(survivors, []);
+});
+
+test("an empty directory is removed, and a missing one is not an error", async () => {
+ assert.equal((await discardOf({})).removed, true);
+ const root = await mkdtemp(path.join(tmpdir(), "ttb-discard-"));
+ try {
+ // Nothing to list means nothing to delete. An `rm -rf` on a path we could
+ // not read is the last thing to do about an unreadable path.
+ assert.equal(
+ await discardPrefetchDir(path.join(root, "data", "nope"), noop),
+ false,
+ );
+ } finally {
+ await rm(root, { recursive: true, force: true });
+ }
+});
+
+// Every one of these is a file the deny-list did not name and would have
+// deleted. The assertion is the same each time: the directory is still there,
+// with its contents intact.
+const PROTECTED: Array<[string, Record<string, string>, string[]?]> = [
+ [
+ "a chat-only corpus member",
+ {
+ "metadata.info.json": "{}",
+ "transcript.live_chat.json": '{"replayChatItemAction":{}}\n',
+ "live_chat.cues.json": '{"cues":[]}',
+ "download-outcome.json": '{"status":"chat-only"}',
+ },
+ ],
+ [
+ "a downloaded video",
+ { "metadata.info.json": "{}", "audio.mp3": "bytes" },
+ ],
+ [
+ "a video whose only transcript is ours",
+ { "metadata.info.json": "{}", "transcript.json": "{}" },
+ ],
+ [
+ "a video whose only transcript is a foreign VTT",
+ { "metadata.info.json": "{}", "transcript.es.vtt": "WEBVTT\n" },
+ ],
+ [
+ "a resumable partial download",
+ { "metadata.info.json": "{}", "audio.mp3.part": "half" },
+ ],
+ [
+ "a diarization sidecar",
+ { "metadata.info.json": "{}", "diarization.json": "{}" },
+ ],
+ [
+ "a digest sidecar",
+ { "metadata.info.json": "{}", "digest.json": "{}" },
+ ],
+ [
+ "a saved-video pointer",
+ { "metadata.info.json": "{}", "saved-video.json": "{}" },
+ ],
+ [
+ "a clip window another tool asked for",
+ { "metadata.info.json": "{}" },
+ ["clips"],
+ ],
+ [
+ "anything at all that nobody has thought of yet",
+ { "metadata.info.json": "{}", "something-new.json": "{}" },
+ ],
+];
+
+for (const [what, files, dirs] of PROTECTED) {
+ test(`${what} keeps its directory`, async () => {
+ const { removed, survivors } = await discardOf(files, dirs ?? []);
+ assert.equal(removed, false, `${what} was discarded`);
+ assert.deepEqual(
+ survivors.sort(),
+ [...Object.keys(files), ...(dirs ?? [])].sort(),
+ );
+ });
+}
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -1,6 +1,6 @@
import path from "node:path";
import { appendFile, mkdir, readdir, readFile, rm } from "node:fs/promises";
-import { createWriteStream, type WriteStream } from "node:fs";
+import { createWriteStream, type Dirent, type WriteStream } from "node:fs";
import { execa } from "execa";
import {
AUTH_RETRY_CLASSES,
@@ -46,6 +46,8 @@ import {
isLivestreamMetadata,
} from "../lib/transcripts-server";
import { evaluateDownloadFilters } from "../lib/downloadFilters";
+import { normalizeLiveChat } from "../controller/normalizeLiveChat";
+import { LIVE_CHAT_FILENAME } from "../lib/videoStatus";
import {
upsertMetadataScan,
type MetadataScanEntry,
@@ -321,6 +323,98 @@ function transcribeHandlingArgsForAudioCheck(
// the body lives in ytdlp/channelArgs.ts so the clip-window fetch shares it.
const channelConfigArgs = channelExtraArgs;
+// A TITLE-FILTER REJECTION MUST NOT LEAVE A VIDEO DIRECTORY BEHIND.
+//
+// The prefetch writes `data/<id>/metadata.info.json` BEFORE the filters get to
+// look at it — that is the whole point of the split — so by the time the
+// operator's own "not this one" is known, the directory exists. And a directory
+// holding a metadata.info.json is not a neutral leftover: `buildIndex.ts`
+// admits ANY such dir to the LMDB index and to the published site (transcript
+// or not), and `deriveChannelSets` reads the dir NAME as "ever fetched", which
+// is what takes a vanished video out of `missingNeverFetched`. The metadata
+// scan creates none of them for exactly these two reasons
+// (controller/metadataScanStore.ts); a rejection that ran through the
+// downloader must not create them either, or the same channel gets both
+// answers depending on which path reached the video first.
+//
+// THE BUCKET DOES NOT COME FROM THIS DIRECTORY. `skippedByTitleFilter` is
+// derived in the snapshot from the metadata-scan store against the channel's
+// CURRENT filter (channelSnapshot.ts, settledIdsFrom) — the entry recorded
+// beside this call — so removing the dir costs the count nothing. The retryable
+// `skippedByFilter` bucket IS read off `download-outcome.json`, which is why
+// every OTHER filter (skip-live) still writes one: that skip says "we will try
+// again", and a video with no directory and no outcome would silently leave it.
+//
+// ── IT IS AN ALLOW-LIST, AND IT HAS TO BE ────────────────────────────────────
+//
+// This function deletes a directory recursively, so the question it must answer
+// is "is EVERYTHING in here mine?", not "is anything in here one of the few
+// things I thought to check for". The deny-list it replaced named
+// isVideoDownloaded, `clips/` and `saved-video.json` — and would therefore have
+// deleted a chat-only corpus member (`transcript.live_chat.json` +
+// `live_chat.cues.json`), a foreign-language `transcript.es.vtt`, a resumable
+// `audio.mp3.part`, a `diarization.json` or a `digest.json`. That is not a
+// hypothetical: the chat-only branch below calls this whenever its chat pass
+// did NOT come back with a file, so a chat re-fetch that failed, was aborted
+// mid-write or hit a cooldown would have deleted the chat that was already
+// there, and an `importVideoAction` on an existing chat-only id would have done
+// the same.
+//
+// So: the dir goes only when every entry is something THIS pass wrote or could
+// have found from a previous run of itself. Anything else — any file, any
+// subdirectory, anything a future feature adds — keeps it, and the caller falls
+// back to writing the outcome sidecar exactly as it always did. Being wrong in
+// this direction costs one metadata stub; being wrong in the other costs bytes
+// nobody can enumerate.
+const PREFETCH_OWN_FILES: ReadonlySet<string> = new Set([
+ // What the prefetch pass itself writes: --write-info-json under the
+ // `infojson:` output template (outputArgsForUrl).
+ "metadata.info.json",
+ // The managed download's own per-video log, opened before the prefetch runs.
+ "download.log",
+ // A sidecar from an EARLIER managed attempt on this same id — including the
+ // one a previous rejection wrote before this rule existed. It records what
+ // happened, never what is on disk, so it is ours to drop with the rest.
+ "download-outcome.json",
+]);
+
+async function discardPrefetchDir(
+ videoDir: string,
+ onLog: (line: string) => void,
+): Promise<boolean> {
+ let entries: Dirent[];
+ try {
+ entries = await readdir(videoDir, { withFileTypes: true });
+ } catch {
+ // No directory at all — the legacy single-call path never made one — or it
+ // is unreadable. Either way there is nothing of ours to remove, and a
+ // `rm -rf` on a path we could not list is the last thing to do about it.
+ return false;
+ }
+ for (const entry of entries) {
+ // A DIRECTORY IS NEVER OURS. `clips/` is the one that exists today (media
+ // another tool asked this editor for, invisible to every video-dir
+ // enumerator); the rule is shaped so the next one needs no edit here.
+ if (!entry.isFile() || !PREFETCH_OWN_FILES.has(entry.name)) return false;
+ }
+ try {
+ await rm(videoDir, { recursive: true, force: true });
+ return true;
+ } catch (err) {
+ onLog(
+ `Could not remove the prefetch directory ${videoDir}: ${(err as Error).message}\n`,
+ );
+ return false;
+ }
+}
+
+// Exported for the unit test ONLY. The rule above is a delete, and the
+// difference between the allow-list and the deny-list it replaced is invisible
+// to every end-to-end path that does not happen to have the right file on disk
+// — which is exactly the shape of bug that reaches production. See
+// downloadOneManaged.test.ts.
+export const __discardPrefetchDirForTest = discardPrefetchDir;
+
// The one-invocation runner moved to ytdlp/runOneYtdlp.ts (the clip-window
// fetch needs the same log tee, stderr tail and archive scrape). This wrapper
// keeps the ManagedDownloadOpts-shaped call sites below unchanged.
@@ -336,6 +430,57 @@ function runOneYtdlp(
);
}
+// ONE EXTRA yt-dlp PASS THAT FETCHES A LIVESTREAM'S CHAT AND NOTHING ELSE.
+//
+// Reached only from the filter branch below, for a channel whose
+// `rejectedLivestreams` is "chat-only". The media is not wanted; the chat is.
+//
+// WHAT THE ARGUMENT ORDER IS FOR. yt-dlp takes the LAST occurrence of an
+// option, so the refusals go AFTER the channel's own extra args — a channel
+// that configured `--write-thumbnail` or `--write-auto-subs` must not be able
+// to turn this pass back into a partial download of things nobody asked for.
+// Same rule fetchWindowManaged relies on, and for the same reason.
+//
+// --no-write-info-json, NOT the absence of --write-info-json: the metadata is
+// ALREADY on disk from the prefetch, and it has to stay there — a video dir
+// with no metadata.info.json is invisible to buildIndex, so the chat would be
+// fetched and then never published. This pass must neither rewrite it nor
+// delete it.
+//
+// No archive line is appended. An archive id means "downloaded" to
+// verifyTranscripts and to the sync walk, and this video is not.
+async function fetchLiveChatOnly(
+ opts: ManagedDownloadOpts,
+ channelDir: string,
+ cookies: string | undefined,
+): Promise<AttemptOutcome> {
+ return runOneYtdlp(opts, channelDir, [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...channelConfigArgs(opts.channelConfig, cookies),
+ "--skip-download",
+ "--write-subs",
+ "--no-write-auto-subs",
+ "--sub-langs",
+ "live_chat",
+ "--no-write-info-json",
+ "--no-write-description",
+ "--no-write-thumbnail",
+ "--no-download-archive",
+ ...outputArgsForUrl(opts.videoUrl),
+ "--",
+ opts.videoUrl,
+ ]);
+}
+
+// Did the chat pass actually leave a chat behind? A stream with chat replay
+// disabled makes yt-dlp exit 0 and write nothing, and a directory holding only
+// a metadata.info.json is the leftover this slice exists to stop creating.
+async function hasLiveChatOnDisk(videoDir: string): Promise<boolean> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ return entries.includes(LIVE_CHAT_FILENAME);
+}
+
function attemptSucceeded(exitCode: number | null): boolean {
// yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop).
return exitCode === 0 || exitCode === 101;
@@ -593,6 +738,82 @@ async function runManagedDownload(
opts.onLog(
`Skipping ${canonicalId}: ${decision.reason} [filter=${decision.filter}]\n`,
);
+ // ── CHAT ONLY, AND IT RUNS BEFORE THE STORE IS WRITTEN ───────────────
+ //
+ // The operator asked for this livestream's chat and not its media, so
+ // this is the one rejection that FETCHES something — and the one that
+ // keeps its prefetch directory. A chat-only dir holds the
+ // metadata.info.json (without it buildIndex never sees the video and the
+ // chat is published nowhere) plus transcript.live_chat.json and its
+ // normalized live_chat.cues.json, beside the outcome sidecar and the log
+ // every managed download leaves. Every "is it downloaded?" predicate
+ // tests whisper, an English VTT or audio, so it still answers no — which
+ // is right: this is a chat track, not a download.
+ //
+ // IT RUNS FIRST because its result belongs in the scan entry below. A
+ // stream whose chat replay is OFF has to be remembered somewhere, or the
+ // id sits in chatOnlyPending forever and every runner restart re-prefetches
+ // it and re-runs a pass that will never return anything.
+ let chatOnlyFetched = false;
+ let chatOnlyUnavailable = false;
+ if (decision.chatOnly && !opts.signal.aborted) {
+ opts.onLog(
+ `Fetching the live chat for ${canonicalId} (media skipped by the download filter).\n`,
+ );
+ const chatRes = await fetchLiveChatOnly(
+ opts,
+ channelDir,
+ alwaysCookies(cookiePolicy) ??
+ (prefetchNeededCookies ? authRetryCookies(cookiePolicy) : undefined),
+ );
+ attempts.push({
+ n: 1,
+ kind: "live-chat-only",
+ handling: opts.channelConfig.handling,
+ usedCookies: Boolean(alwaysCookies(cookiePolicy)),
+ ytdlpExitCode: chatRes.exitCode,
+ error: attemptSucceeded(chatRes.exitCode)
+ ? undefined
+ : trimError(chatRes.stderrTail),
+ });
+ if (attemptSucceeded(chatRes.exitCode)) {
+ chatOnlyFetched = await hasLiveChatOnDisk(videoDir);
+ if (chatOnlyFetched) {
+ // The CUES sidecar, not just the raw file: buildIndex prefers
+ // live_chat.cues.json when it is fresh, so normalizing here makes
+ // the video index-ready the moment it lands instead of waiting for
+ // the corpus-wide normalize pass.
+ try {
+ await normalizeLiveChat({
+ videoDir,
+ channelSlug: opts.channelSlug,
+ ...(opts.channelConfig.name
+ ? { configName: opts.channelConfig.name }
+ : {}),
+ log: (m: string) => opts.onLog(`${m}\n`),
+ });
+ } catch (err) {
+ opts.onLog(
+ `Could not normalize the live chat: ${(err as Error).message}\n`,
+ );
+ }
+ } else {
+ // A CLEAN PASS THAT FOUND NOTHING IS AN ANSWER, not a retry. Chat
+ // replay was off for this stream, or the source has since dropped
+ // it; asking again tomorrow gets the same nothing. Recorded on the
+ // scan entry below so the video leaves chatOnlyPending for good and
+ // settles as the ordinary filtered-out livestream it is.
+ //
+ // A FAILED pass (non-zero exit, a cooldown, an abort) is NOT this:
+ // it leaves the flag alone and the id stays pending, which is the
+ // whole reason this is gated on the exit code.
+ chatOnlyUnavailable = true;
+ opts.onLog(
+ `No live chat was available for ${canonicalId}; recording that so it is not asked for again.\n`,
+ );
+ }
+ }
+ }
// A title-filter rejection feeds the metadata-scan store, so this video is
// settled from here on WITHOUT a second metadata fetch. The store is the
// one place a settled verdict is derived from; the outcome below records
@@ -608,6 +829,7 @@ async function runManagedDownload(
...(typeof metadata.duration === "number"
? { duration: metadata.duration }
: {}),
+ ...(chatOnlyUnavailable ? { noLiveChat: true } : {}),
scannedAt: new Date().toISOString(),
};
// A BATCH COLLECTS; A SINGLE VIDEO WRITES. The store is channel-level,
@@ -637,19 +859,38 @@ async function runManagedDownload(
const record: DownloadOutcomeRecord = {
videoId: canonicalId,
webpageUrl: opts.videoUrl,
- status: "skipped-filtered",
+ status: chatOnlyFetched ? "chat-only" : "skipped-filtered",
startedAt,
finishedAt,
attempts,
filter: { name: decision.filter, reason: decision.reason },
};
- try {
- await mkdir(videoDir, { recursive: true });
- await writeDownloadOutcome(videoDir, record);
- } catch (err) {
+ // The operator's own rejection takes its prefetch directory with it —
+ // see discardPrefetchDir for why a metadata-only dir is not a neutral
+ // leftover, and why the rule is an allow-list. Every other filter's skip
+ // is RETRYABLE and its outcome sidecar is what `skippedByFilter` derives
+ // from, so only this one discards.
+ //
+ // A fetched chat is not "mine to delete" either way — the allow-list
+ // refuses the directory the moment transcript.live_chat.json is in it —
+ // so this condition is a shortcut past a readdir, not the safety rail.
+ const discarded =
+ decision.filter === "titleFilter" && metadata && !chatOnlyFetched
+ ? await discardPrefetchDir(videoDir, opts.onLog)
+ : false;
+ if (discarded) {
opts.onLog(
- `Failed to write download-outcome.json: ${(err as Error).message}\n`,
+ `Removed the metadata-only directory for ${canonicalId}: a filtered video is not a member of this corpus.\n`,
);
+ } else {
+ try {
+ await mkdir(videoDir, { recursive: true });
+ await writeDownloadOutcome(videoDir, record);
+ } catch (err) {
+ opts.onLog(
+ `Failed to write download-outcome.json: ${(err as Error).message}\n`,
+ );
+ }
}
return record;
}
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -948,6 +948,12 @@ async function runManagedDownloads(
}
if (outcome.status === "skipped-filtered") {
skippedCount++;
+ } else if (outcome.status === "chat-only") {
+ // The media was skipped and the live chat was fetched. Counted as a
+ // SKIP, because the batch's `okCount` means "videos downloaded" and
+ // this one deliberately was not — the chat is a different artifact
+ // and the report's own chatOnly bucket is where it is counted.
+ skippedCount++;
} else if (outcome.status === "corrupt-full-source") {
// Completed but stayed malformed after one re-download: terminal and
// kept on disk, but not a usable download. Count as skipped (not ok,
@@ -1587,7 +1593,23 @@ async function syncFullSweep(opts: RunYtdlpOpts): Promise<void> {
.filter((e) => e.isDirectory())
.map((e) => e.name),
);
- const sets = deriveChannelSets({ roster, listedIds, onDiskIds });
+ // THE SETTLED SET IS SUBTRACTED, and it is not a nicety. A title-filter
+ // rejection leaves no directory now, so without this a filtered-out video
+ // that has since left the listing reads as `missingNeverFetched` — "we were
+ // told about this, never got it, and it is gone" — and the sweep tells the
+ // operator to attempt a direct-link recovery of the very video they
+ // configured us not to fetch.
+ const sweepSettledIds = await settledByTitleFilterIds(
+ opts.paths,
+ opts.channelSlug,
+ opts.channelConfig,
+ );
+ const sets = deriveChannelSets({
+ roster,
+ listedIds,
+ onDiskIds,
+ settledIds: sweepSettledIds,
+ });
maybeMissing = sets.missingDownloaded;
await writeMaybeMissing(opts.paths, opts.channelSlug, {
checkedAt: now,
diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx
@@ -55,6 +55,8 @@ type Props = {
nonStandardVttIds: string[];
skippedByFilterIds: string[];
skippedByTitleFilterIds: string[];
+ chatOnlyIds: string[];
+ chatOnlyPendingIds: string[];
totals: { videos: number; transcribed: number; downloaded: number };
availability: AvailabilitySnapshot;
maybeMissing: { ids: string[]; checkedAt: string };
@@ -73,6 +75,8 @@ export function DiagnosticsStage({
nonStandardVttIds,
skippedByFilterIds,
skippedByTitleFilterIds,
+ chatOnlyIds,
+ chatOnlyPendingIds,
totals,
availability,
maybeMissing,
@@ -131,6 +135,23 @@ export function DiagnosticsStage({
"The metadata scan read these titles and this channel's download filter (Configure > Advanced) rejects them, so nothing downloads or re-attempts them. Change the include/exclude patterns and they are re-decided on the next report — no rescan needed.",
ariaLabel: "filtered out",
},
+ {
+ ids: chatOnlyIds,
+ label: "Chat only (filtered livestreams)",
+ // No Retry either, and for the same reason as the bucket above: these are
+ // settled, and the thing to change is the mode in Configure. A retry
+ // would offer to download the media the operator said not to.
+ description:
+ "Livestreams this channel's download filter rejected while \u201cFiltered-out livestreams\u201d is set to keep the chat (Configure > Advanced). The media was never fetched; the live chat was, and it is published as a chat track with no captions. Set the mode back to Skip and they stop being fetched \u2014 what is already on disk stays.",
+ ariaLabel: "chat only",
+ },
+ {
+ ids: chatOnlyPendingIds,
+ label: "Chat only: chat not fetched yet",
+ description:
+ "Filtered-out livestreams whose live chat has not been fetched yet. The download lane picks these up last \u2014 after every real download \u2014 because a chat pass costs one metadata-sized request and no media.",
+ ariaLabel: "chat only pending",
+ },
];
const populated = buckets.filter((b) => b.ids.length > 0);
const total = populated.reduce((acc, b) => acc + b.ids.length, 0);
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -559,6 +559,8 @@ export default async function ChannelDetailPage({
nonStandardVttIds={buckets.nonStandardVtt}
skippedByFilterIds={buckets.skippedByFilter}
skippedByTitleFilterIds={buckets.skippedByTitleFilter}
+ chatOnlyIds={buckets.chatOnly ?? []}
+ chatOnlyPendingIds={buckets.chatOnlyPending ?? []}
totals={snapshot.totals}
availability={normalizeAvailability(snapshot.availability)}
maybeMissing={normalizeMaybeMissing(snapshot.maybeMissing)}
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -25,7 +25,7 @@ import {
import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import { mergeRosterFile } from "yt-dlp-transcript-common/controller/rosterStore";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
-import { runMetadataScan } from "yt-dlp-transcript-common/ytdlp/metadataScan";
+import { runMetadataScanJob } from "yt-dlp-transcript-common/controller/metadataScanJob";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
import { isGateHeld } from "yt-dlp-transcript-common/lib/pauseGates";
import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy";
@@ -351,22 +351,17 @@ export async function runMetadataScanAction(
`The metadata scan will run once the cooldown lapses.`,
};
}
- return runManagedFunction({
- kind: "metadata-scan",
- queueKey: resolveQueueKey(downloadQueueKey(channelConfig), queueKey),
+ // THE JOB ITSELF LIVES IN common/controller/metadataScanJob.ts, because the
+ // download lane's runner dispatches the same one and cannot reach a server
+ // action. What stays here is what a CLICK is owed and a runner is not: the
+ // cooldown answered as a sentence rather than a queued job, and the three
+ // revalidations.
+ return runMetadataScanJob({
paths,
- channelSlug: slug,
- spec: { kind: "metadata-scan", slug, params: { queueKey } },
- fn: async (onLog, signal, setProgress) => {
- await runMetadataScan({
- paths,
- channelSlug: slug,
- channelConfig,
- onLog,
- signal,
- setProgress,
- onPlatformBackoff: () => recordDownloadBackoff(platform, paths),
- });
+ slug,
+ channelConfig,
+ queueKey,
+ afterRun: () => {
revalidatePath(`/channels/${slug}`);
revalidatePath("/channels");
revalidatePath("/operations/[id]", "page");
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -1603,6 +1603,11 @@ function DownloadOutcomeBadge({
? `Filtered out by the download filter${why}`
: `Skipped by filter${why}`;
}
+ // NOT a download, and the badge has to say so plainly: this video's media
+ // was never fetched and never will be while the filter stands. What is
+ // here is its live chat, which is why the video is in the corpus at all.
+ case "chat-only":
+ return "Chat only (filtered livestream) — live chat kept, no media and no transcript";
default:
return outcome.status;
}
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -731,6 +731,27 @@ export function ChannelForm({
</span>
</span>
</label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Filtered-out livestreams</span>
+ <select
+ name="downloadFilterRejectedLivestreams"
+ defaultValue={c?.downloadFilter?.rejectedLivestreams ?? "skip"}
+ aria-label="filtered-out livestreams"
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="skip">Skip them — download nothing</option>
+ <option value="chat-only">
+ Keep the live chat — no audio, no transcript
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ What to do with a livestream the filter above rejected. A multi-hour
+ stream whose title says nothing is rarely worth its audio, but its
+ live chat is text, it is small, and it is the only record of what the
+ room said. Chat-only videos join the corpus as a chat track with no
+ captions; they are never downloaded and never transcribed.
+ </span>
+ </label>
<Field
label="Cookies from browser"
name="cookiesFromBrowser"
diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts
@@ -48,6 +48,7 @@ export const CHANNEL_FORM_VALUES = [
"cookieMode",
"downloadFilterInclude",
"downloadFilterExclude",
+ "downloadFilterRejectedLivestreams",
"audioCheckIntervalSeconds",
"audioCheckMaxRollbacks",
"audioCheckCopyTimeoutSeconds",
@@ -103,6 +104,14 @@ export function channelConfigToFormData(config: ChannelConfig): FormData {
if (config.downloadFilter?.includeLivestreams) {
fd.set("downloadFilterIncludeLivestreams", "on");
}
+ // Omitted when it is "skip" — the parser reads an absent field as "skip", so
+ // emitting it would be a value with no effect, and a patch clearing it ("")
+ // has to land on the same default the form's own select does.
+ put(
+ fd,
+ "downloadFilterRejectedLivestreams",
+ config.downloadFilter?.rejectedLivestreams,
+ );
if (config.audioCheck?.enabled) {
fd.set("audioCheckEnabled", "on");
putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds);
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -22,6 +22,7 @@ import { handleFromAccountUrl } from "yt-dlp-transcript-common/social/fetchers";
import { isCookieMode } from "yt-dlp-transcript-common/lib/cookiePolicy";
import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import { downloadFilterPatternProblem } from "yt-dlp-transcript-common/lib/downloadFilters";
+import { isRejectedLivestreamMode } from "yt-dlp-transcript-common/lib/channelConfig";
export type ParsedChannelForm = {
name: string;
@@ -241,12 +242,33 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
}
const includeLivestreams =
formData.get("downloadFilterIncludeLivestreams") != null;
+ // WHAT TO DO WITH A REJECTED LIVESTREAM. Only ever stored alongside a real
+ // filter and only when it is not the default — with nothing rejecting, the
+ // mode names a decision that can never be taken, and a stored "skip" would be
+ // a key on disk that changes nothing.
+ const rejectedLivestreamsRaw = String(
+ formData.get("downloadFilterRejectedLivestreams") ?? "",
+ ).trim();
+ if (
+ rejectedLivestreamsRaw &&
+ !isRejectedLivestreamMode(rejectedLivestreamsRaw)
+ ) {
+ throw new Error(
+ `Filtered-out livestreams must be "skip" or "chat-only" (got ${JSON.stringify(
+ rejectedLivestreamsRaw,
+ )})`,
+ );
+ }
+ const rejectedLivestreams = isRejectedLivestreamMode(rejectedLivestreamsRaw)
+ ? rejectedLivestreamsRaw
+ : "skip";
const downloadFilter: ChannelConfig["downloadFilter"] | undefined =
downloadFilterInclude || downloadFilterExclude || includeLivestreams
? {
...(downloadFilterInclude ? { include: downloadFilterInclude } : {}),
...(downloadFilterExclude ? { exclude: downloadFilterExclude } : {}),
...(includeLivestreams ? { includeLivestreams: true } : {}),
+ ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
}
: undefined;
diff --git a/editor/app/operations/components/InFlightList.tsx b/editor/app/operations/components/InFlightList.tsx
@@ -41,16 +41,38 @@ export function InFlightList({
aria-hidden="true"
className="size-1.5 shrink-0 self-center rounded-full bg-info animate-pulse motion-reduce:animate-none"
/>
- <Link
- href={`/channels/${item.channelSlug}/videos/${item.videoId}`}
- className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
- >
- {item.channelSlug}/{item.videoId}
- </Link>
- <span className="text-xs text-muted-foreground">
- {at >= 0 ? `rule ${at + 1}` : item.leafId}
- {leaf ? ` · ${leafSentence(leaf, channels)}` : ""}
- </span>
+ {/* A UNIT THAT IS NOT A VIDEO SAYS SO INSTEAD OF LINKING.
+ The download lane's metadata-scan unit is channel-scoped:
+ its `videoId` is a synthetic key naming no directory, so the
+ usual link would be a 404 and the usual "rule N" would name
+ a leaf that exists in no tree. It carries a `note`, and a
+ unit with one renders the note — see AutoRunnerInFlight. */}
+ {item.note ? (
+ <>
+ <Link
+ href={`/channels/${item.channelSlug}`}
+ className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
+ >
+ {item.channelSlug}
+ </Link>
+ <span className="text-xs text-muted-foreground">
+ {item.note}
+ </span>
+ </>
+ ) : (
+ <>
+ <Link
+ href={`/channels/${item.channelSlug}/videos/${item.videoId}`}
+ className="font-mono text-foreground underline underline-offset-2 hover:text-brand"
+ >
+ {item.channelSlug}/{item.videoId}
+ </Link>
+ <span className="text-xs text-muted-foreground">
+ {at >= 0 ? `rule ${at + 1}` : item.leafId}
+ {leaf ? ` · ${leafSentence(leaf, channels)}` : ""}
+ </span>
+ </>
+ )}
<span className="ml-auto tabular-nums text-xs text-muted-foreground">
{now !== null ? formatElapsed(now - item.startedAt) : "—"}
</span>
diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts
@@ -1,4 +1,4 @@
-import { mkdir, writeFile } from "node:fs/promises";
+import { mkdir, readFile, writeFile } from "node:fs/promises";
import { test, expect } from "@playwright/test";
import {
channelStage,
@@ -795,6 +795,125 @@ test("download: prioritizes channels across the per-platform queue", async ({
]);
});
+// A FILTERED download channel whose snapshot advertises a metadata-scan
+// backlog. The fake yt-dlp titles every video "Synthetic <id>", so an include
+// of /guest/ keeps guestvid* and rejects plainvid* with no fixture metadata.
+// The channel's yt-dlp invocation log, or "" before the first one.
+async function readText(relPath: string): Promise<string> {
+ return readFile(resolvePath(relPath), "utf8").catch(() => "");
+}
+
+async function makeFilteredDownloadChannel(
+ slug: string,
+ ids: string[],
+ unscanned: number,
+) {
+ const root = resolvePath(`test-transcripts/channels/${slug}`);
+ await mkdir(`${root}/data`, { recursive: true });
+ await writeFile(
+ `${root}/config.json`,
+ JSON.stringify({
+ handling: "youtube",
+ name: slug,
+ url: `https://www.youtube.com/@${slug}/videos`,
+ downloadFilter: { include: "guest" },
+ }),
+ );
+ await writeFile(
+ `${root}/playlist`,
+ ids.map((id) => `https://www.youtube.com/watch?v=${id}`).join("\n") + "\n",
+ );
+ await writeFile(
+ `${root}/snapshot.json`,
+ JSON.stringify({
+ generatedAt: "2026-06-01T00:00:00.000Z",
+ totals: { videos: 0, transcribed: 0, downloaded: 0 },
+ buckets: {},
+ undownloadedIds: ids,
+ metadataScan: { scanned: 0, errors: 0, unscanned },
+ }),
+ );
+}
+
+test("the download lane scans a filtered channel before it downloads from it", async ({
+ request,
+}) => {
+ // THE SCAN DECIDES WHAT THE DOWNLOADS ARE, so it is not another kind of work
+ // competing with them — it goes first. Without it every non-matching video on
+ // a filtered channel is prefetched, rejected and discarded one yt-dlp
+ // invocation at a time, against a source that is counting them; one batch
+ // answers the whole channel.
+ await resetData(null);
+ await makeFilteredDownloadChannel(
+ "filtered",
+ ["guestvid0001", "plainvid0001", "plainvid0002"],
+ 3,
+ );
+
+ const root: Group = {
+ id: "root",
+ mode: "strict",
+ children: [{ id: "leaf-all", match: { type: "all" } }],
+ };
+ await writeSettings({
+ adminTitle: "Test Admin",
+ maxTranscriptPageBytes: 8388608,
+ sleepBetweenDownloadsSeconds: 0,
+ minFreeDiskGB: 0,
+ workers: ONE_WORKER,
+ autoQueue: downloadAutoQueue(root),
+ });
+
+ await startRunner(request, "download");
+
+ const invocationsPath =
+ "test-transcripts/channels/filtered/fake-ytdlp.invocations";
+ await expect
+ .poll(
+ async () =>
+ (await readText(invocationsPath)).includes("prefetch:") ||
+ (await pathExists(
+ "test-transcripts/channels/filtered/data/guestvid0001/transcript.en.vtt",
+ )),
+ { timeout: 60_000 },
+ )
+ .toBe(true);
+
+ const invocations = await readText(invocationsPath);
+ const scanAt = invocations.indexOf("metadata-scan:");
+ const firstPrefetchAt = invocations.indexOf("prefetch:");
+ // The scan ran, and it ran BEFORE a single video was prefetched.
+ expect(scanAt).toBeGreaterThanOrEqual(0);
+ expect(firstPrefetchAt).toBeGreaterThan(scanAt);
+
+ // And it read the whole listing in ONE batch — which is the saving. What is
+ // deliberately NOT asserted here is that the non-matches are never prefetched
+ // afterwards: the scan asks for a snapshot regen and the runner picks up the
+ // new work list on a later tick, so whether a plain video slips through in
+ // that window is a race with the regen, not a fact about the feature. The
+ // settlement itself is pinned in title-filter.spec.ts.
+ const scan = await readJson<{ entries: Record<string, unknown> }>(
+ "test-transcripts/channels/filtered/metadata-scan.json",
+ );
+ expect(Object.keys(scan.entries).sort()).toEqual([
+ "guestvid0001",
+ "plainvid0001",
+ "plainvid0002",
+ ]);
+
+ // ONE SCAN PER CHANNEL PER RUNNER. The loop keeps ticking every three
+ // seconds; a second scan would mean the source is re-asked for the whole
+ // listing forever.
+ await expect
+ .poll(async () => (await getStatus(request)).download.picks.length, {
+ timeout: 30_000,
+ })
+ .toBeGreaterThan(0);
+ expect(
+ (await readText(invocationsPath)).split("metadata-scan:").length - 1,
+ ).toBe(1);
+});
+
// A React state update that lands before hydration is discarded silently, so a
// select or a click made too early leaves the page looking changed while the
// component still holds the old value — and the next save writes the old value.
diff --git a/editor/e2e/chat-only.spec.ts b/editor/e2e/chat-only.spec.ts
@@ -0,0 +1,379 @@
+import { readFile, rm, writeFile } from "node:fs/promises";
+import { test, expect, type Page } from "@playwright/test";
+import {
+ buildIndex,
+ channelStage,
+ generateReport,
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSite,
+} from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+// THE THIRD ANSWER FOR A FILTERED-OUT LIVESTREAM.
+//
+// A multi-hour stream whose title says nothing about the subject is rarely
+// worth its audio, but its live chat is text, it is small, and it is the only
+// record of what the room said. `downloadFilter.rejectedLivestreams:
+// "chat-only"` is that answer: the media is never fetched, the chat is, and the
+// video joins the corpus as a chat track with no captions.
+//
+// What every test here is really guarding is one risk. A chat-only directory
+// makes "metadata, no transcript" a LEGITIMATE shape — and that is exactly what
+// buildIndex.ts:309-315 and deriveChannelSets read as "this video was fetched".
+// Get the bucket rules wrong and filtered livestreams silently leave
+// missingNeverFetched and enter the published site as blank pages.
+
+const CHANNEL = "test-filter";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+const GUEST = ["guestvid0001", "guestvid0002"];
+const PLAIN = ["plainvid0001", "plainvid0002"];
+// The fixture's cookie-gated video. Its METADATA SCAN is refused without
+// cookies, but a download-time prefetch reads it fine — so on this path (no
+// scan first) the filter decides it like any other non-match.
+const NEEDS_AUTH = "needsauthvid01";
+const TOKEN = "test-worker-token";
+const AUTH = { authorization: `Bearer ${TOKEN}` };
+// The fixture's FINISHED livestream VOD: `livevid` makes the fake report
+// live_status "was_live", and its title contains neither "guest" nor "plain",
+// so the include pattern rejects it and only the livestream rules decide it.
+const LIVE = "livevid000001";
+
+type Snapshot = {
+ generatedAt: string;
+ totals: { videos: number; transcribed: number; downloaded: number };
+ buckets: {
+ skippedByTitleFilter?: string[];
+ chatOnly?: string[];
+ chatOnlyPending?: string[];
+ noTranscript?: string[];
+ downloadedNoTranscript?: string[];
+ };
+ undownloadedIds: string[];
+};
+
+async function writeFilterConfig(filter: Record<string, unknown> | null) {
+ await writeFile(
+ resolvePath(`${ROOT}/config.json`),
+ JSON.stringify(
+ {
+ handling: "youtube",
+ name: "Test Title Filter",
+ url: "https://www.youtube.com/@example/videos",
+ ...(filter ? { downloadFilter: filter } : {}),
+ },
+ null,
+ 2,
+ ),
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+}
+
+async function refreshReport(
+ page: Page,
+ after: string,
+ until?: (snapshot: Snapshot) => boolean,
+): Promise<Snapshot> {
+ await page.goto("/channels");
+ const refresh = page.getByRole("button", {
+ name: `refresh report ${CHANNEL}`,
+ });
+ await refresh.waitFor({ state: "visible" });
+ await expect
+ .poll(
+ async () => {
+ await refresh.click({ timeout: 5_000 }).catch(() => {});
+ for (let i = 0; i < 20; i++) {
+ const cur = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch(
+ () => null,
+ );
+ if (cur && cur.generatedAt > after && (!until || until(cur))) {
+ return true;
+ }
+ await new Promise((r) => setTimeout(r, 250));
+ }
+ return false;
+ },
+ { timeout: 60_000, intervals: [1000] },
+ )
+ .toBe(true);
+ return readJson<Snapshot>(`${ROOT}/snapshot.json`);
+}
+
+async function download(page: Page): Promise<void> {
+ await page.goto(channelStage(CHANNEL, "download"));
+ await page.getByRole("button", { name: "Download videos" }).click();
+ await expect(page.getByLabel("Download videos output")).toContainText(
+ "Managed download complete",
+ { timeout: 60_000 },
+ );
+}
+
+test("a chat-only livestream gets its chat and stays undownloaded", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+
+ await download(page);
+
+ // The chat, and the metadata that makes it publishable. Nothing else.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.live_chat.json`)).toBe(
+ true,
+ );
+ // metadata.info.json HAS to survive: buildIndex stats it per video dir and
+ // skips the directory when it throws, so without it the chat would be fetched
+ // and then published nowhere.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/metadata.info.json`)).toBe(true);
+ // Normalized on the spot, so the video is index-ready when it lands rather
+ // than waiting for the corpus-wide normalize pass.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/live_chat.cues.json`)).toBe(true);
+ // NOT a download, by every definition the app has.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.en.vtt`)).toBe(false);
+ expect(await pathExists(`${ROOT}/data/${LIVE}/audio.mp3`)).toBe(false);
+ const outcome = await readJson<{ status: string; attempts: unknown[] }>(
+ `${ROOT}/data/${LIVE}/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("chat-only");
+ // No archive line: an archive id means "downloaded" to the sync walk and to
+ // verifyTranscripts, and this video is not.
+ const archive = await readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(
+ () => "",
+ );
+ expect(archive).not.toContain(LIVE);
+
+ // The ordinary rejections still leave nothing behind, and the matches still
+ // download — the mode changes what happens to LIVESTREAMS and nothing else.
+ for (const id of PLAIN) {
+ expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false);
+ }
+ for (const id of GUEST) {
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true);
+ }
+});
+
+test("a chat-only video is in chatOnly, and in no bucket that means work", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+ await download(page);
+
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const snap = await refreshReport(
+ page,
+ before.generatedAt,
+ (s) => (s.buckets.chatOnly ?? []).length > 0,
+ );
+
+ expect(snap.buckets.chatOnly ?? []).toEqual([LIVE]);
+ // NOT "we decided not to have it" — we decided to have its chat.
+ expect(snap.buckets.skippedByTitleFilter ?? []).toEqual(
+ [...PLAIN, NEEDS_AUTH].sort(),
+ );
+ // Nothing is going to transcribe a video whose audio we chose not to fetch,
+ // so it must be out of both transcription-facing buckets AND the download
+ // queue, or the lanes would fight the filter forever.
+ expect(snap.buckets.noTranscript ?? []).not.toContain(LIVE);
+ expect(snap.buckets.downloadedNoTranscript ?? []).not.toContain(LIVE);
+ expect(snap.undownloadedIds).not.toContain(LIVE);
+ // Its chat is on disk, so there is no outstanding chat work either.
+ expect(snap.buckets.chatOnlyPending ?? []).toEqual([]);
+ // A MEMBER OF THIS CORPUS, not a stub: it has a directory and it is counted.
+ // Exactly three — the two matches plus the chat-only livestream. Every other
+ // rejection took its prefetch directory with it.
+ expect(snap.totals.videos).toBe(GUEST.length + 1);
+ // And it is not a DOWNLOAD: only the two matches are.
+ expect(snap.totals.downloaded).toBe(GUEST.length);
+});
+
+test("before the chat lands it is chatOnlyPending, which is download-lane work", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+
+ // Seed the scan store directly: the scan is what turns a listed id into a
+ // chat-only verdict without fetching anything, and this is the state the
+ // download lane is meant to find.
+ await writeFile(
+ resolvePath(`${ROOT}/metadata-scan.json`),
+ JSON.stringify({
+ version: 1,
+ entries: {
+ [LIVE]: {
+ title: `Synthetic ${LIVE}`,
+ description: "",
+ uploadDate: "20240101",
+ liveStatus: "was_live",
+ scannedAt: "2026-01-01T00:00:00.000Z",
+ },
+ },
+ errors: {},
+ lastRun: null,
+ }),
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const snap = await refreshReport(
+ page,
+ before.generatedAt,
+ (s) => (s.buckets.chatOnlyPending ?? []).length > 0,
+ );
+ expect(snap.buckets.chatOnlyPending ?? []).toEqual([LIVE]);
+ // It is NOT in undownloadedIds — fetching the media is the one thing this
+ // video must not have done to it — and not in skippedByTitleFilter either.
+ expect(snap.undownloadedIds).not.toContain(LIVE);
+ expect(snap.buckets.skippedByTitleFilter ?? []).not.toContain(LIVE);
+});
+
+// The index half, on the cheapest fixture that has one. The claim: a directory
+// holding metadata.info.json and transcript.live_chat.json and NO transcript is
+// admitted, published with its chat track, and carries no cues.
+const INDEX_CHANNEL = "test-youtube";
+const INDEX_DIR = "20240101_test1234567";
+const INDEX_DATA = `test-transcripts/channels/${INDEX_CHANNEL}/data/${INDEX_DIR}`;
+
+test("the export publishes a chat-only video as a chat track with no captions", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("one-youtube-channel-with-data");
+ // Make it chat-only-shaped: the chat, and no transcript at all.
+ await rm(resolvePath(`${INDEX_DATA}/transcript.en.vtt`), { force: true });
+ // yt-dlp's live_chat shape: one `replayChatItemAction` envelope per line,
+ // carrying the offset and the renderer parseLiveChat reads.
+ await writeFile(
+ resolvePath(`${INDEX_DATA}/transcript.live_chat.json`),
+ JSON.stringify({
+ replayChatItemAction: {
+ videoOffsetTimeMsec: "11000",
+ actions: [
+ {
+ addChatItemAction: {
+ item: {
+ liveChatTextMessageRenderer: {
+ authorName: { simpleText: "viewer" },
+ message: { runs: [{ text: "platypus in chat" }] },
+ },
+ },
+ },
+ },
+ ],
+ },
+ }) + "\n",
+ );
+ await writeSite("testsite", {
+ channels: [{ slug: INDEX_CHANNEL, groupId: "default" }],
+ });
+ await buildIndex(page);
+
+ const subsPage = await readJson<
+ Array<{ id: string; tracks?: Record<string, unknown[]> }>
+ >(`test-transcripts/.export-index/shared/subs/${INDEX_CHANNEL}/page-0000.json`);
+ const detail = subsPage.find((d) => d.id === INDEX_DIR);
+ expect(detail).toBeDefined();
+ expect(Object.keys(detail?.tracks ?? {})).toEqual(["live_chat"]);
+ expect((detail?.tracks?.live_chat ?? []).length).toBeGreaterThan(0);
+
+ // And no captions: the transcripts shard carries the video (it has metadata)
+ // with no cues at all.
+ const transcriptsPage = await readJson<Array<{ id: string; cues?: unknown[] }>>(
+ `test-transcripts/.export-index/shared/transcripts/${INDEX_CHANNEL}/page-0000.json`,
+ );
+ const t = transcriptsPage.find((d) => d.id === INDEX_DIR);
+ expect(t).toBeDefined();
+ expect(t?.cues?.length ?? 0).toBe(0);
+});
+
+test("the form round-trips the chat-only tier and the ops API sets it", async ({
+ page,
+ request,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await page.goto(channelStage(CHANNEL, "configure"));
+
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ const select = page.getByLabel("filtered-out livestreams");
+ // Absent on disk reads as "skip", which is what every channel written before
+ // the field did.
+ await expect(select).toHaveValue("skip");
+
+ await select.selectOption("chat-only");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () =>
+ (
+ await readJson<{
+ downloadFilter?: { rejectedLivestreams?: string };
+ }>(`${ROOT}/config.json`)
+ ).downloadFilter?.rejectedLivestreams ?? null,
+ { timeout: 30_000 },
+ )
+ .toBe("chat-only");
+
+ // Back to the default, and the key is REMOVED rather than stored as "skip":
+ // a stored default is a key on disk that changes nothing.
+ await page.goto(channelStage(CHANNEL, "configure"));
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await expect(page.getByLabel("filtered-out livestreams")).toHaveValue(
+ "chat-only",
+ );
+ await page.getByLabel("filtered-out livestreams").selectOption("skip");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () =>
+ "rejectedLivestreams" in
+ ((
+ await readJson<{ downloadFilter?: Record<string, unknown> }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter ?? {}),
+ { timeout: 30_000 },
+ )
+ .toBe(false);
+
+ // The ops API goes through the same parser, so it gets the same validation.
+ const ok = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ headers: AUTH,
+ data: {
+ slug: CHANNEL,
+ patch: { downloadFilterRejectedLivestreams: "chat-only" },
+ },
+ });
+ expect(ok.ok()).toBeTruthy();
+ expect(
+ (
+ await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter?.rejectedLivestreams,
+ ).toBe("chat-only");
+
+ const bad = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ headers: AUTH,
+ data: { slug: CHANNEL, patch: { downloadFilterRejectedLivestreams: "maybe" } },
+ });
+ expect(bad.status()).toBeGreaterThanOrEqual(400);
+ // Refused, and the stored value is untouched.
+ expect(
+ (
+ await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter?.rejectedLivestreams,
+ ).toBe("chat-only");
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -772,6 +772,50 @@ async function main() {
return;
}
+ // CHAT ONLY: --skip-download --write-subs --sub-langs live_chat, with
+ // auto-subs and the info json explicitly refused. Must come BEFORE the
+ // youtube single-URL branch below, which matches --write-auto-subs (this
+ // invocation passes --no-write-auto-subs, so it would fall through to the
+ // final error instead).
+ //
+ // It writes ONLY transcript.live_chat.json — no metadata.info.json (the
+ // prefetch already wrote it and this pass must not touch it), no transcript,
+ // no audio, and no DLOM_ARCHIVE line. A fake that wrote any of those would
+ // make the spec pass for the wrong reason: the whole claim is that a
+ // chat-only video is NOT downloaded.
+ if (
+ has("--skip-download") &&
+ has("--write-subs") &&
+ has("--no-write-auto-subs") &&
+ arg("--sub-langs") === "live_chat" &&
+ !has("--flat-playlist")
+ ) {
+ const url = lastNonFlag();
+ const id = urlIdYouTube(url ?? "");
+ if (!id) {
+ process.stderr.write(`[fake-ytdlp] chat-only mode missing URL\n`);
+ process.exit(2);
+ }
+ const videoDir = path.join("data", id);
+ await ensureDir(videoDir);
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `live-chat-only:${url} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
+ // `nochat` is the stream whose chat replay is off: yt-dlp exits 0 and
+ // writes nothing, which is the case the downloader has to notice (it keeps
+ // no directory for it).
+ if (!(url ?? "").toLowerCase().includes("nochat")) {
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"clientId":"fake","action":{"addChatItemAction":{}}}\n',
+ );
+ }
+ process.stdout.write(`[fake-ytdlp] live chat only ${id}\n`);
+ return;
+ }
+
// Managed YouTube-handling per-URL invocation: --skip-download with
// --write-auto-subs, no -a. The real download reuses the prefetched
// metadata via --load-info-json (no positional URL) — derive the id from
diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts
@@ -1,4 +1,4 @@
-import { readFile, writeFile } from "node:fs/promises";
+import { mkdir, readFile, writeFile } from "node:fs/promises";
import { test, expect, type Page } from "@playwright/test";
import {
channelStage,
@@ -201,6 +201,112 @@ test("a settled video is never downloaded", async ({ page }) => {
);
});
+test("a title-filter rejection leaves no video directory behind", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ // NO SCAN FIRST, deliberately: that is the only way a rejection reaches the
+ // downloader at all. Once the scan has read a video the filter settles it and
+ // yt-dlp is never invoked for it (the test above). The leftover this pins
+ // belongs to the other case — a newly listed video the download lane reaches
+ // before any scan does — where the metadata PREFETCH has already written
+ // data/<id>/metadata.info.json by the time the filter gets to say no.
+ //
+ // A directory holding a metadata.info.json is admitted to the LMDB index and
+ // the published site by buildIndex, and its name is what deriveChannelSets
+ // reads as "ever fetched". The scan creates none of them; a rejection must
+ // not either, or the same channel gets two different answers depending on
+ // which path reached the video first.
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+
+ await download(page);
+
+ const invocations = await readInvocations();
+ for (const id of GUEST) {
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true);
+ }
+ for (const id of SETTLED_BY_GUEST) {
+ // It WAS prefetched — that is the whole difference from the settled case —
+ // and the directory that prefetch made is gone again.
+ expect(invocations).toContain(
+ `prefetch:https://www.youtube.com/watch?v=${id}`,
+ );
+ expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false);
+ expect(await readArchive()).not.toContain(id);
+ }
+
+ // AND THE COUNT DOES NOT MOVE. The bucket is derived from the metadata-scan
+ // store — the rejection recorded its entry there before discarding the
+ // directory — so removing the directory costs it nothing.
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const snapshot = await refreshReport(
+ page,
+ before.generatedAt,
+ (s) => (s.buckets.skippedByTitleFilter ?? []).length > 0,
+ );
+ // FOUR, not three, and the extra one is the point of this path rather than a
+ // fixture quirk: `needsauthvid01` is the video the METADATA SCAN cannot read
+ // (the scan's fake refuses it without cookies, so the scan-first test above
+ // records an ERROR for it and settles nothing). A download-time prefetch
+ // reads it fine, so the filter decides it here — which is exactly the
+ // difference this test exists to exercise.
+ const settledByDownload = [...SETTLED_BY_GUEST, NEEDS_AUTH].sort();
+ expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(settledByDownload);
+ const scan = await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`);
+ expect(Object.keys(scan.entries).sort()).toEqual(settledByDownload);
+});
+
+test("a rejection never deletes a directory that holds somebody else's data", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ // The discard refuses anything but its own prefetch. A CLIP WINDOW is the
+ // case worth pinning: `data/<id>/clips/` is media another tool asked this
+ // editor for (umtool's POST /api/media/fetch-window), it is not a
+ // "destination" so the run still attempts the video, and it is invisible to
+ // every video-dir enumerator — so a discard that walked past it would delete
+ // bytes nothing else would ever mention again.
+ //
+ // The second video carries DOWNLOADED MEDIA. On a youtube-handling channel
+ // the destination is transcript.en.vtt, so an audio.mp3 does not prefilter
+ // the video away — the run attempts it, the filter rejects it, and the bytes
+ // have to survive. A video downloaded before the filter was written is
+ // downloaded: that is a fact, not a preference.
+ await resetData("title-filter-channel");
+ const [withClips, withAudio] = PLAIN;
+ await mkdir(resolvePath(`${ROOT}/data/${withClips}/clips`), {
+ recursive: true,
+ });
+ await writeFile(
+ resolvePath(`${ROOT}/data/${withClips}/clips/0.00-30.00.mp3`),
+ "fake clip audio\n",
+ );
+ await mkdir(resolvePath(`${ROOT}/data/${withAudio}`), { recursive: true });
+ await writeFile(
+ resolvePath(`${ROOT}/data/${withAudio}/audio.mp3`),
+ "fake audio bytes\n",
+ );
+ await generateReport(page, CHANNEL);
+
+ await download(page);
+
+ for (const [id, file] of [
+ [withClips, "clips/0.00-30.00.mp3"],
+ [withAudio, "audio.mp3"],
+ ] as const) {
+ expect(await pathExists(`${ROOT}/data/${id}/${file}`)).toBe(true);
+ // The directory stayed, so the retryable outcome sidecar is written for it
+ // exactly as it was before this rule existed.
+ const outcome = await readJson<{ status: string }>(
+ `${ROOT}/data/${id}/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("skipped-filtered");
+ }
+ // And the rejection that had nothing of its own is still gone.
+ expect(await pathExists(`${ROOT}/data/${LIVE}`)).toBe(false);
+});
+
test("changing the filter re-decides the channel with no rescan", async ({
page,
}) => {
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4128,11 +4128,16 @@ never pulls `reader-fs.ts` into a client chunk.
- **The scan is a catalogued operation**, `metadata-scan`: second channel-scoped
entry in `operationCatalog()`, `trigger: "backlog"` (sync stays the ONE
`"cadence"` entry — that is how `/operations/sync` is chosen), platform download
- queue, `needsMedia: false`, and `pauseLaneFor` answers null for it. It is
- deliberately NOT held by the downloads pause: it fetches no media, and the
- operator runs it precisely to decide what a paused lane should fetch when it
- resumes. Its backlog is `snapshot.metadataScan.unscanned`, which must keep
+ queue. Its backlog is `snapshot.metadataScan.unscanned`, which must keep
meaning exactly what `metadataScanTargets()` will fetch.
+ **TWO CORRECTIONS, 2026-09-21.** Its job kind declares **`needsMedia: TRUE`**,
+ not false — `jobKinds.ts` says why at length: the scan's target set is
+ "listed, minus what is already on disk", so it OPENS `data/`, and against an
+ unmounted drive it would re-request the whole channel. (It writes no media;
+ that is a different question.) And `pauseLaneFor` answers **`"download"`**
+ now, because the download runner dispatches it — see the follow-ups section
+ below. What the old exemption was FOR is intact and now belongs to the Run
+ button, which checks the platform cooldown and no gate at all.
- **A scan error suppresses a re-scan of that id for 24 h**
(`METADATA_SCAN_ERROR_COOLDOWN_MS`). Without it a members-only video keeps the
backlog above zero forever and the operation page never stops offering work.
@@ -4148,9 +4153,170 @@ never pulls `reader-fs.ts` into a client chunk.
has no `runner`, so the operator presses Run. The obvious next step is for the
download lane to run it for a channel whose filter has unscanned listed videos
— which is exactly the backlog the snapshot already carries.
+ **DONE 2026-09-21 — see the section below, which also amends the pause note
+ above.**
---
+## Filtered-channel follow-ups (verified 2026-09-21, branch `feat/filtered-channel-followups`)
+
+Three things the filter slice left: a rejection's leftover directory, nothing
+auto-dispatching the scan, and no answer but "skip" for a livestream the filter
+rejects.
+
+- **A TITLE-FILTER REJECTION NOW DELETES ITS OWN PREFETCH DIRECTORY**
+ (`ytdlp/downloadOneManaged.ts`, `discardPrefetchDir`). The prefetch writes
+ `data/<id>/metadata.info.json` BEFORE the filters look at it, so by the time
+ the operator's "not this one" is known the directory exists — and a directory
+ holding a metadata.info.json is admitted to the LMDB index and the published
+ site by `buildIndex.ts:309-315`, transcript or not, while `deriveChannelSets`
+ reads its NAME as "ever fetched". The scan creates none of them for exactly
+ those two reasons; the downloader now creates none either. It fires when a
+ rejection reaches the DOWNLOADER at all — a video the filter needed a live
+ prefetch for, i.e. one the scan has not read. The two batch paths never invoke
+ yt-dlp for a settled video (`selectDownloadableUrls` counts it an archived
+ hit), but `importVideoAction` reaches `downloadOneManaged` by URL and bypasses
+ both, so "a settled video is never invoked" is true of a sync and a
+ download-missing and NOT of a hand-pasted link.
+- **The count does not move, and the reason is that it never came from there.**
+ `buckets.skippedByTitleFilter` is built at `channelSnapshot.ts:1378-1383` from
+ `settledIds` ← `settledIdsFrom(metadataScanStore, config)` at `:927`. Nothing
+ in it reads `download-outcome.json`. The only on-disk reader of
+ `"skipped-filtered"` is the RETRYABLE `skippedByFilter` bucket at `:1101-1105`,
+ which is why every OTHER filter (skip-live) still writes its outcome sidecar:
+ that skip says "we will try again", and a video with neither directory nor
+ outcome would silently leave the bucket. `channelSnapshot.test.ts` runs a whole
+ filtered channel with NO `data/` dirs at all and still gets the bucket.
+- **`videoHasAnyArtifact` was a private copy of `isVideoDownloaded`, byte for
+ byte**, in a module that already imported the original (it is what increments
+ `downloaded` in the same walk). It is an alias now: two names for one
+ predicate is how the settled short-circuit and `totals.videos` end up
+ disagreeing about what an artifact is, and the five call sites read better
+ asking "does this dir hold anything we fetched?".
+- **IT IS AN ALLOW-LIST, AND THE DENY-LIST IT REPLACED WAS A BUG.** This is a
+ recursive delete, so the question it must answer is "is EVERYTHING in here
+ mine?". The first version named `isVideoDownloaded`, `clips/` and
+ `saved-video.json` — and would therefore have deleted a chat-only corpus
+ member (`transcript.live_chat.json` + `live_chat.cues.json`), a
+ `transcript.es.vtt`, a resumable `audio.mp3.part`, a `diarization.json` or a
+ `digest.json`. Reachable, not hypothetical: the chat-only branch calls this
+ whenever its chat pass came back with no file, so a chat re-fetch that failed
+ or was aborted would have deleted the chat that was already there. The rule is
+ now `PREFETCH_OWN_FILES` = `{metadata.info.json, download.log,
+ download-outcome.json}`, files only — **a directory is never ours** — and
+ anything else keeps the dir. `ytdlp/downloadOneManaged.test.ts` runs one case
+ per protected shape. The returned `DownloadOutcomeRecord` is unchanged either
+ way, so `runYtdlp.ts:949` and the runner's `unitStatus` check need no edit.
+- **`deriveChannelSets` takes the settled set**, and both call sites pass it
+ (`channelSnapshot.ts`, and `syncFullSweep` in `runYtdlp.ts`). Both sets it
+ derives are defined by the ABSENCE of a directory, and a settled video has none
+ now — so without this a filtered-out video that later leaves the listing reads
+ as `missingNeverFetched`, the loudest alarm this system raises, pointed at a
+ video the operator asked us not to fetch. `undownloaded` is subtracted for the
+ same reason: settled work is not work.
+- **THE DOWNLOAD LANE AUTO-DISPATCHES THE SCAN.** `METADATA_SCAN_OPERATION` now
+ declares `runner: "download"`, and the runner's `next()` runs a pre-pick before
+ any video: `pickMetadataScanChannel` (pure, exported, unit-tested) takes the
+ first channel in `metaCache` order that has a COMPILING filter, a non-zero
+ `snapshot.metadataScan.unscanned`, no scan already done by this runner, and a
+ platform that is neither busy nor cooling. It goes first because it decides
+ what the downloads ARE.
+- **The scan unit is channel-scoped and holds a platform slot.** `Picked` gains
+ `scan`; the synthetic key is `metadata-scan <slug>` (a space, which no video id
+ contains) with leaf id `metadata-scan` and an EMPTY `path`, so no node's
+ `maxWorkers` moves for it. It takes `platformInFlight` exactly as a download
+ unit does, so a scan and a download never hit one source at once.
+ `AutoRunnerInFlight` gains `note?`, and `InFlightList` renders the note (and
+ links the CHANNEL) instead of a dead video link.
+- **A FOCUS HOLDS THE SCAN TOO.** The compiled priority tree holds every
+ download pick by strict descent, and the pre-pick does not go through that
+ tree — so without this, "only jeralyzer is moving" would be false the moment
+ another channel had titles to read, on the same network the focus is trying to
+ have to itself. `pickMetadataScanChannel` takes the `focusSummary` the lane
+ already computes for its log line: while `holding` (i.e. `focusPending > 0`)
+ only a focused channel is offered a scan; once the focus is exhausted the lane
+ is free and so is the scan.
+- **ONE SCAN PER CHANNEL PER RUNNER, deliberately.** A scan that left the backlog
+ above zero was stopped by something — a soft block, a cooldown, a batch that
+ ended early. `scannedThisRun` means this runner does not re-offer it three
+ seconds later; the operator's Run and the next restart both still work.
+- **`pauseLaneFor("metadata-scan")` is `"download"` now, not null, and the old
+ note's reasoning moved to the Run button.** `pauseLaneFor` asks `runner` first,
+ so naming the runner gives the operation the download lane's console AND makes
+ the download pause hold the AUTO-dispatch. What the exemption was FOR is
+ intact: `runMetadataScanAction` checks the platform cooldown and no gate at
+ all, so the operator can still scan while downloads are paused — which is
+ precisely when they want to decide what the lane should fetch on resume.
+- **`downloadFilter.rejectedLivestreams: "skip" | "chat-only"`, absent = "skip"**
+ (`lib/channelConfig.ts`), and the sanitizer stores it only alongside a real
+ filter and only when it is not the default. A channel without the field
+ behaves byte-for-byte as it did.
+- **"chat-only" IS A REJECTION with an instruction attached, not a fourth way to
+ pass.** `DownloadFilterVerdict` gains it; `classifyAgainstFilter` returns it
+ from ONE place (`rejection()`), so an `exclude`-matched livestream and an
+ unmatched one cannot get different answers. **`titleFilterRejects` still
+ answers TRUE for it** — that is what keeps `settledIdsFrom`, `undownloadedIds`
+ and the sync walk correct with no other change. `titleFilterWantsChat` is the
+ next question, asked only by the downloader and the snapshot.
+ `chatOnlyIdsFrom` (metadataScanStore) is the derived STRICT SUBSET of the
+ settled set.
+- **The download path runs ONE extra yt-dlp pass** (`fetchLiveChatOnly`):
+ `--skip-download --write-subs --no-write-auto-subs --sub-langs live_chat`, with
+ the refusals AFTER `channelConfigArgs` and `-o` after those — yt-dlp takes the
+ LAST occurrence, so a channel's own `--write-thumbnail` cannot turn this into a
+ partial download. `--no-write-info-json`, NOT the absence of
+ `--write-info-json`: the prefetch's `metadata.info.json` must SURVIVE, or
+ `buildIndex.ts:309-315` never sees the video and the chat is published nowhere.
+ Then `normalizeLiveChat` writes `live_chat.cues.json` on the spot. No archive
+ line — an archive id means "downloaded" to the sync walk and to
+ `verifyTranscripts`.
+- **A CLEAN PASS THAT FOUND NOTHING IS AN ANSWER, not a retry.** A stream with
+ chat replay off makes yt-dlp exit 0 and write nothing. The directory goes like
+ any other rejection's, and — because nothing on disk can then remember it —
+ the scan entry carries **`noLiveChat: true`**, which `chatOnlyIdsFrom` drops.
+ Without it the id sits in `chatOnlyPending` forever: `markCompleted` retires it
+ for the SESSION only, so every runner restart re-prefetched it to run a pass
+ that will never return anything. A pass that FAILED (non-zero exit, cooldown,
+ abort) deliberately does not set it — that one stays retryable — which is why
+ the chat pass now runs BEFORE the store is written rather than after.
+- **The chat-only verdict is decided off the FILE, not the mode**, in the
+ snapshot's settled branch: a dir holding `transcript.live_chat.json` is a
+ `chatOnly` member however `rejectedLivestreams` currently reads. Asking
+ `chatOnlyIds` there made the same directory a settled STUB the moment the mode
+ flipped back — and `settledOnDisk` is subtracted from `totals.videos`, so the
+ site kept serving a video the report had stopped counting.
+- **`DownloadOutcomeStatus` gains `"chat-only"`** and `DownloadAttemptKind` gains
+ `"live-chat-only"`. The batch counts it as a SKIP (`okCount` means videos
+ downloaded); the runner counts it as a SUCCESS (the unit did the work it was
+ picked for, and the platform's cooldown clears).
+- **Two buckets, and neither is `skippedByTitleFilter`.** `chatOnly` = the
+ verdict plus `transcript.live_chat.json` on disk — a corpus MEMBER, counted in
+ `totals.videos`, never in `settledOnDisk` (which is subtracted from it), and
+ short-circuited out of every bucket that means something is missing.
+ `chatOnlyPending` = the verdict without the file, appended LAST to
+ `DOWNLOAD_BUCKETS` so a chat backlog can never delay a real download. Both are
+ `[]` for every channel without the field, so the fold, the lane work list and
+ the pick order are byte-identical for them.
+- **It cannot ride in `undownloadedIds`**, and that is the reason `chatOnlyPending`
+ is a bucket rather than a derived count: that list means "fetch the media", and
+ fetching the media is the one thing this video must not have done to it.
+- **THE RISK THE TESTS EXIST FOR: a chat-only dir makes "metadata, no
+ transcript" a legitimate shape** — exactly what `buildIndex.ts:309-315` and
+ `deriveChannelSets` (`channelSets.ts:58-67`) read as "ever fetched". Get the
+ bucket rules wrong and filtered livestreams silently leave
+ `missingNeverFetched` and enter the published site as blank pages.
+ `channelSnapshot.test.ts` runs all three states (chat landed / chat pending /
+ mode turned off) against a real `generateChannelSnapshot`, and
+ `editor/e2e/chat-only.spec.ts` builds the index and asserts the published
+ record carries ONLY a `live_chat` track and zero cues.
+- **`common/controller/metadataScanJob.ts` is the one definition of the scan as a
+ JOB** — kind, per-platform queue key, replay spec and the shared
+ `onPlatformBackoff` wiring. The editor action keeps only what a CLICK is owed
+ (the cooldown answered as a sentence, and three `revalidatePath` calls, passed
+ as `afterRun`); the runner passes `background: true` and neither. `needsMedia`
+ comes from the kind (TRUE for this one), so `runManagedFunction` refuses it
+ against an unmounted drive for both callers with nothing said in the module.
+
## Clip windows: media sourced for another tool (2026-09-20)
umtool asks the editor for a clip window instead of running yt-dlp