commit 4d0c68f87f3904f93ef6e0e0b0824f3e549b59f1
parent 6e75b2cb001dafa18a2b20e42e2a26cee12e300f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 02:03:04 -0400
download filter: a third answer for a filtered-out livestream
Until now a livestream the filter rejected had exactly one fate: nothing. But
a multi-hour stream whose title says nothing about the subject is rarely worth
its audio, while its live chat is text, it is small, and it is the only record
of what the room said. `downloadFilter.rejectedLivestreams: "chat-only"` is
that middle answer. Absent means "skip", which is byte-for-byte what every
channel written before this field did.
It is a third VERDICT, not a flag read at download time: classifyAgainstFilter
returns "chat-only" from one place, so an exclude-matched livestream and an
unmatched one cannot get different answers. titleFilterRejects STILL answers
true for it — that is what keeps the settled set, undownloadedIds and the sync
walk correct with no other change — and titleFilterWantsChat is the next
question, asked only by the two callers that act on it.
The download path runs one extra pass: --skip-download --write-subs
--no-write-auto-subs --sub-langs live_chat, refusals after the channel's own
args and -o after those, because yt-dlp takes the last occurrence and a
channel's --write-thumbnail must not be able to turn this into a partial
download. --no-write-info-json rather than the absence of --write-info-json:
the prefetch's metadata.info.json has to SURVIVE or buildIndex never sees the
video and the chat is published nowhere. normalizeLiveChat runs on the spot.
No archive line: an archive id means "downloaded", and this is not.
Two buckets, and neither is skippedByTitleFilter — that one means "we decided
not to have it", and we decided to have its chat. `chatOnly` is a corpus
member: counted in totals.videos, out of noTranscript, downloadedNoTranscript
and undownloadedIds. `chatOnlyPending` is the download lane's, appended LAST
to DOWNLOAD_BUCKETS so a chat backlog can never delay a real download. Both
are [] for every channel without the field.
The risk the tests exist for: a chat-only dir makes "metadata, no transcript"
a legitimate shape, which is exactly what buildIndex and deriveChannelSets
read as "ever fetched". channelSnapshot.test.ts runs all three states against
a real generateChannelSnapshot; chat-only.spec.ts builds the index and asserts
the published record carries only a live_chat track and zero cues.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
23 files changed, 1084 insertions(+), 16 deletions(-)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -2186,6 +2186,11 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
// Cancelled (runner stopped, or dropped while still queued) → not a failure.
if (term.status === "cancelled") return { outcome: "skipped" };
if (unitStatus === "skipped-filtered") return { outcome: "skipped" };
+ // The filter declined the media and the live chat was fetched instead. A
+ // SUCCESS for the lane — the unit did the work it was picked for, the video
+ // leaves chatOnlyPending, and a platform whose chat pass came back clean has
+ // its cooldown cleared exactly as a download would clear it.
+ if (unitStatus === "chat-only") return { outcome: "transcribed" };
// Complete-but-malformed source: terminal and kept on disk. Treat as skipped
// (not failed) so it doesn't drive backoff and isn't re-picked for download.
if (unitStatus === "corrupt-full-source") return { outcome: "skipped" };
diff --git a/common/controller/channelSnapshot.test.ts b/common/controller/channelSnapshot.test.ts
@@ -442,3 +442,102 @@ test("a settled video is in skippedByTitleFilter with NO directory of its own",
}
});
+test("a chat-only livestream is a corpus member, not a settled stub", async () => {
+ // THE RISK THIS PINS. A chat-only dir makes "metadata, no transcript" a
+ // LEGITIMATE shape — which is exactly what buildIndex and deriveChannelSets
+ // read as "this video was fetched". If the bucket rules are wrong, filtered
+ // livestreams either vanish from the report or reappear as download work
+ // forever.
+ const { dir, paths, channelDir } = await filteredChannel({
+ filter: { include: "keep", rejectedLivestreams: "chat-only" },
+ listed: ["aaaa0000001", "bbbb0000002", "cccc0000003"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ bbbb0000002: { title: "drop this one" },
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ // The chat has landed for the livestream: metadata + the raw chat, and
+ // nothing else on disk.
+ const videoDir = path.join(channelDir, "data", "cccc0000003");
+ await mkdir(videoDir, { recursive: true });
+ await writeFile(
+ path.join(videoDir, "metadata.info.json"),
+ JSON.stringify({ id: "cccc0000003", title: "a long stream" }),
+ );
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"action":{}}\n',
+ );
+
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]);
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ // NOT "we decided not to have it".
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]);
+ // Nothing will transcribe a video whose audio we chose not to fetch, and
+ // nothing will download it.
+ assert.ok(!snap.buckets.noTranscript.includes("cccc0000003"));
+ assert.ok(!snap.buckets.downloadedNoTranscript.includes("cccc0000003"));
+ assert.ok(!snap.undownloadedIds.includes("cccc0000003"));
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]);
+ // A video we deliberately have. `settledOnDisk` is subtracted from this and
+ // a chat-only dir must never be in it.
+ assert.equal(snap.totals.videos, 1);
+ assert.equal(snap.totals.downloaded, 0);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("before the chat lands it is chatOnlyPending — the download lane's bucket", async () => {
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep", rejectedLivestreams: "chat-only" },
+ listed: ["aaaa0000001", "cccc0000003"],
+ scanned: {
+ aaaa0000001: { title: "keep this one" },
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnlyPending, ["cccc0000003"]);
+ assert.deepEqual(snap.buckets.chatOnly, []);
+ // It cannot ride in undownloadedIds: that list means "fetch the media", and
+ // fetching the media is the one thing this video must not have done to it.
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]);
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, []);
+ // It IS the download lane's work, through the lane work list every lane
+ // reads its dispatch from.
+ assert.ok(snap.backfill?.download?.ids.includes("cccc0000003"));
+ // And LAST in it, behind the real download — the fold walks
+ // DOWNLOAD_BUCKETS in order.
+ const ids = snap.backfill?.download?.ids ?? [];
+ assert.equal(ids[ids.length - 1], "cccc0000003");
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
+test("turning the mode off re-decides the channel with nothing to migrate", async () => {
+ // Same store, same disk, no filter field: the livestream is an ordinary
+ // settled rejection again and the chat-only buckets are empty. Nothing about
+ // the verdict was ever stored per video.
+ const { dir, paths } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: ["cccc0000003"],
+ scanned: {
+ cccc0000003: { title: "a long stream", liveStatus: "was_live" },
+ },
+ });
+ try {
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.buckets.chatOnly, []);
+ assert.deepEqual(snap.buckets.chatOnlyPending, []);
+ assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -8,6 +8,7 @@ import {
isVideoFetched,
isVideoTranscribed,
readVideoFiles,
+ LIVE_CHAT_FILENAME,
VTT_FILENAME,
type VideoFiles,
} from "../lib/videoStatus";
@@ -48,6 +49,7 @@ import {
import { isExcludedFromTruncatedCheck } from "../lib/excludeTruncatedCheck-server";
import { loadDownloadOutcome } from "../lib/downloadOutcome-server";
import {
+ chatOnlyIdsFrom,
loadMetadataScan,
metadataScanWanted,
settledIdsFrom,
@@ -213,6 +215,25 @@ export type ChannelSnapshot = {
// re-appear as ordinary undownloaded videos on the next report.
// Optional: older snapshots lack it; readers must default to [].
skippedByTitleFilter: string[];
+ // CHAT-ONLY videos whose chat is on disk. A rejected livestream on a channel
+ // whose `downloadFilter.rejectedLivestreams` is "chat-only": the media was
+ // never fetched, the live chat was, and the directory holds
+ // metadata.info.json + transcript.live_chat.json and nothing else.
+ //
+ // A MEMBER OF THIS CORPUS, not a stub. It is in totals.videos, it is in the
+ // LMDB index, and the site publishes it as a chat track with no captions.
+ // It is deliberately NOT in skippedByTitleFilter (that bucket means "we
+ // decided not to have it"), NOT in noTranscript or downloadedNoTranscript
+ // (nothing is going to transcribe a video whose audio we chose not to
+ // fetch), and NOT in undownloadedIds. Optional: older snapshots lack it;
+ // readers must default to [] — which normalizeBuckets does, like every
+ // other bucket declared required here and absent from older files.
+ chatOnly: string[];
+ // The same population, chat NOT yet on disk — one of the download lane's
+ // buckets (DOWNLOAD_BUCKETS). It cannot ride in undownloadedIds: that list
+ // means "fetch the media", and fetching the media is the one thing this
+ // video must not have done to it. Absent from older files in the same way.
+ chatOnlyPending: string[];
// Videos whose transcript covers only a small fraction of the video's
// duration — the audio download silently truncated (yt-dlp exited "ok") so
// whisper transcribed just the first few minutes. The detection threshold
@@ -925,6 +946,11 @@ export async function generateChannelSnapshot(
// whole channel on the next report, with no rescan and nothing to migrate.
const metadataScanStore = await loadMetadataScan(paths, slug);
const settledIds = settledIdsFrom(metadataScanStore, config);
+ // A STRICT SUBSET of the settled set: the rejections the operator asked for
+ // the live chat of (downloadFilter.rejectedLivestreams === "chat-only").
+ // Settled means the MEDIA is not wanted, which is true of these too — what
+ // this adds is that the video is still a corpus member, as a chat track.
+ const chatOnlyIds = chatOnlyIdsFrom(metadataScanStore, config);
const noTranscript: string[] = [];
const downloadedNoTranscript: string[] = [];
@@ -942,6 +968,12 @@ export async function generateChannelSnapshot(
// their metadata before deciding). Tracked separately from the full settled
// set only so totals.videos can subtract exactly the dirs it counted.
const settledOnDisk: string[] = [];
+ // Chat-only videos whose chat is on disk — corpus members, counted in
+ // totals.videos, and out of every bucket that means "something is missing".
+ const chatOnly: string[] = [];
+ // Chat-only videos whose chat is NOT on disk yet: the download lane's work,
+ // and the reason this is a bucket rather than a derived count.
+ const chatOnlyPending: string[] = [];
const incompleteTranscript: string[] = [];
const shortAudio: string[] = [];
const autoSubsOnly: string[] = [];
@@ -1079,6 +1111,22 @@ export async function generateChannelSnapshot(
// it from transcribedWithAudio and the cleanup estimates while it still
// counted in totals.transcribed, i.e. a channel reporting more transcripts
// than videos. Settlement only ever decides what NOT to fetch.
+ // CHAT ONLY, and it is a member of this corpus rather than a stub. The dir
+ // holds metadata.info.json and transcript.live_chat.json, which is exactly
+ // what buildIndex needs to publish it as a chat track with no captions. It
+ // is short-circuited here for the same reason the settled branch below is —
+ // everything under this line classifies a video by what is MISSING, and
+ // nothing is missing from a chat-only video — but it is NOT settledOnDisk:
+ // that list is subtracted from totals.videos, and this one is a video we
+ // deliberately have.
+ if (chatOnlyIds.has(id) && !videoHasAnyArtifact(files)) {
+ if (files.entries.includes(LIVE_CHAT_FILENAME)) {
+ chatOnly.push(id);
+ continue;
+ }
+ // A directory with neither media nor chat is a stub like any other
+ // rejection's, and falls through to the settled branch below.
+ }
if (settledIds.has(id) && !videoHasAnyArtifact(files)) {
settledOnDisk.push(id);
continue;
@@ -1307,6 +1355,15 @@ export async function generateChannelSnapshot(
) {
metadataScanUnscanned++;
}
+ // CHAT ONLY: the media is settled (so this id never reaches
+ // undownloadedIds) but the CHAT may still be outstanding, and that is real
+ // download-lane work with no other home — the id has no artifact, so no
+ // artifact-derived bucket can carry it. Once the chat lands it drops out
+ // here and appears in `chatOnly` instead.
+ if (chatOnlyIds.has(dirId)) {
+ if (!f?.entries.includes(LIVE_CHAT_FILENAME)) chatOnlyPending.push(dirId);
+ continue;
+ }
// A settled video has no artifact and never will while the filter stands.
// This is the line that makes the settlement STICK: undownloadedIds is the
// auto-download runner's work queue, and it is derived from artifacts, so an
@@ -1374,13 +1431,20 @@ export async function generateChannelSnapshot(
nonStandardVtt: nonStandardVtt.sort(),
skippedByFilter: skippedByFilter.sort(),
// The settled set minus anything already on disk — same rule as the
- // short-circuit above, so the bucket and the classification cannot disagree.
+ // short-circuit above, so the bucket and the classification cannot disagree
+ // — and minus the chat-only subset, which has its own two buckets. A
+ // chat-only video IS settled (its media is not wanted) but reporting it as
+ // "skipped by the title filter" would say we decided not to have it, when
+ // we decided to have its chat.
skippedByTitleFilter: [...settledIds]
.filter((id) => {
+ if (chatOnlyIds.has(id)) return false;
const f = filesById.get(id);
return !f || !videoHasAnyArtifact(f);
})
.sort(),
+ chatOnly: chatOnly.sort(),
+ chatOnlyPending: chatOnlyPending.sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
autoSubsOnly: autoSubsOnly.sort(),
diff --git a/common/controller/metadataScanStore.ts b/common/controller/metadataScanStore.ts
@@ -30,7 +30,11 @@ import path from "node:path";
import { readFile, rename, writeFile } from "node:fs/promises";
import type { Paths } from "../lib/paths";
import type { ChannelConfig } from "../lib/channelConfig";
-import { compileDownloadFilter, titleFilterRejects } from "../lib/downloadFilters";
+import {
+ compileDownloadFilter,
+ titleFilterRejects,
+ titleFilterWantsChat,
+} from "../lib/downloadFilters";
export const METADATA_SCAN_FILENAME = "metadata-scan.json";
export const METADATA_SCAN_VERSION = 1;
@@ -282,6 +286,29 @@ export function settledIdsFrom(
return settled;
}
+// THE CHAT-ONLY SUBSET OF THE SETTLED SET — a strict subset, never a second
+// population. `titleFilterRejects` answers true for a chat-only video (see
+// DownloadFilterVerdict), so every id here is also in `settledIdsFrom`'s
+// answer; what this adds is which of them the operator wants the live chat of.
+//
+// Derived from the config every time, exactly like settledIdsFrom, so turning
+// `rejectedLivestreams` off again re-decides the channel with no rescan and
+// nothing stored per video to undo.
+export function chatOnlyIdsFrom(
+ scan: MetadataScan,
+ config: Pick<ChannelConfig, "downloadFilter"> | null | undefined,
+): Set<string> {
+ const ids = new Set<string>();
+ const compiled = compileDownloadFilter(config?.downloadFilter);
+ // The mode only ever means something alongside a real filter, and a channel
+ // that is not chat-only pays one field read for this question.
+ if (!compiled || compiled.rejectedLivestreams !== "chat-only") return ids;
+ for (const [id, entry] of Object.entries(scan.entries)) {
+ if (titleFilterWantsChat(compiled, entry)) ids.add(id);
+ }
+ return ids;
+}
+
// The loading form, for callers that have no scan in hand.
export async function settledByTitleFilterIds(
paths: Paths,
diff --git a/common/jobs/autoQueuePolicy.test.ts b/common/jobs/autoQueuePolicy.test.ts
@@ -329,9 +329,13 @@ test("bucketsForKind: per-kind ordered bucket lists", () => {
[...bucketsForKind("transcription")],
["downloadedNoTranscript", "failedListed"],
);
+ // `chatOnlyPending` is LAST: a chat fetch is the cheapest work on the lane and
+ // must never delay a real download. It is [] for every channel that has not
+ // set downloadFilter.rejectedLivestreams, so the order below is unchanged for
+ // all of them.
assert.deepEqual(
[...bucketsForKind("download")],
- ["partialDownloads", "undownloadedIds"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending"],
);
});
@@ -427,7 +431,7 @@ test("the default union is unchanged by the opt-in buckets", () => {
);
assert.deepEqual(
[...bucketsForKind("download")],
- ["partialDownloads", "undownloadedIds"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending"],
);
assert.deepEqual(
[...defaultBucketsForPolicy("transcription", { replaceAutoSubs: false })],
@@ -446,7 +450,7 @@ test("selectableBucketsForKind offers defaults plus the opt-in buckets", () => {
);
assert.deepEqual(
[...selectableBucketsForKind("download")],
- ["partialDownloads", "undownloadedIds", "autoSubsOnly"],
+ ["partialDownloads", "undownloadedIds", "chatOnlyPending", "autoSubsOnly"],
);
});
diff --git a/common/jobs/autoQueuePolicy.ts b/common/jobs/autoQueuePolicy.ts
@@ -79,7 +79,17 @@ export function sanitizeAutoQueueOrder(value: unknown): AutoQueueOrder {
// its internal priority. Single source of truth for the runner, the pending-
// count helper, and the editor's bucket picker.
export const TRANSCRIBE_BUCKETS = ["downloadedNoTranscript", "failedListed"] as const;
-export const DOWNLOAD_BUCKETS = ["partialDownloads", "undownloadedIds"] as const;
+// `chatOnlyPending` is LAST on purpose: it is the smallest and cheapest work on
+// the lane (one --skip-download pass per video, no media), and putting it ahead
+// of real downloads would let a chat backlog delay the corpus. Empty for every
+// channel that has not set `downloadFilter.rejectedLivestreams` — which is
+// every channel that predates the field — so the fold, the lane's work list and
+// the pick order are byte-identical for them.
+export const DOWNLOAD_BUCKETS = [
+ "partialDownloads",
+ "undownloadedIds",
+ "chatOnlyPending",
+] as const;
// Buckets a runner will NOT draw from unless asked. Replacing YouTube's
// auto-captions with our own transcript costs an audio download plus a
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -67,8 +67,31 @@ export type DownloadFilterConfig = {
// `{ includeLivestreams: true }` alone rejects plain uploads and passes
// livestreams. See titleFilterRejects.
includeLivestreams?: boolean;
+ // WHAT TO DO WITH A LIVESTREAM THE FILTER REJECTED. Absent = "skip", which
+ // is every channel that predates this field and is byte-for-byte what they
+ // did before it existed.
+ //
+ // "chat-only" is the middle answer that did not exist: a multi-hour stream
+ // whose TITLE says nothing about the subject is usually not worth its audio,
+ // but its live chat is text, it is small, and it is the only record of what
+ // the room said. So the video is not downloaded, its chat is, and it joins
+ // the corpus as a chat track with no captions — NOT as a downloaded video.
+ // See classifyAgainstFilter for why this is a third VERDICT rather than a
+ // flag read at download time.
+ rejectedLivestreams?: RejectedLivestreamMode;
};
+// Absent behaves as "skip". Spelled as a union rather than a boolean because a
+// third answer (keeping the audio at a lower quality, say) is a plausible next
+// one and a boolean would have to be migrated to make room for it.
+export type RejectedLivestreamMode = "skip" | "chat-only";
+
+export function isRejectedLivestreamMode(
+ v: unknown,
+): v is RejectedLivestreamMode {
+ return v === "skip" || v === "chat-only";
+}
+
export type ChannelConfig = {
handling: ChannelHandling;
// Omitted = "video" (every channel that predates the posts corpus).
@@ -365,11 +388,19 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null {
const include = typeof df.include === "string" ? df.include.trim() : "";
const exclude = typeof df.exclude === "string" ? df.exclude.trim() : "";
const includeLivestreams = df.includeLivestreams === true;
+ // ONLY ALONGSIDE A REAL FILTER, and only when it is not the default. The
+ // mode says what to do with a REJECTED livestream, so with nothing to
+ // reject it names a decision that can never be taken — storing it would put
+ // a setting on the Configure form that does nothing and explains nothing.
+ const rejectedLivestreams = isRejectedLivestreamMode(df.rejectedLivestreams)
+ ? df.rejectedLivestreams
+ : "skip";
if (include || exclude || includeLivestreams) {
config.downloadFilter = {
...(include ? { include } : {}),
...(exclude ? { exclude } : {}),
...(includeLivestreams ? { includeLivestreams: true } : {}),
+ ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
};
}
}
diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts
@@ -9,6 +9,7 @@ import {
downloadFilterText,
evaluateDownloadFilters,
titleFilterRejects,
+ titleFilterWantsChat,
type DownloadFilterContext,
} from "./downloadFilters";
import type { RawMetadata } from "./transcripts-server";
@@ -409,3 +410,91 @@ test("the matched description is capped", () => {
true,
);
});
+
+// --- rejectedLivestreams: the chat-only tier --------------------------------
+//
+// The whole design risk of this feature is one sentence: "chat-only" is a
+// REJECTION with an instruction attached, not a fourth way to pass. Everything
+// that asks "is this video's media wanted?" must keep answering no for it, or
+// the downloader fetches the very media the operator said not to.
+
+test("chat-only is a rejection, and titleFilterRejects still says so", () => {
+ const f = compileDownloadFilter({
+ include: "guest",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ const stream = liveMeta("was_live");
+ assert.equal(classifyAgainstFilter(f, stream), "chat-only");
+ // THE LOAD-BEARING ONE. settledIdsFrom and therefore undownloadedIds are
+ // built on this: a chat-only video is settled exactly like any other
+ // rejection, so the download queue never offers its media.
+ assert.equal(titleFilterRejects(f, stream), true);
+ assert.equal(titleFilterWantsChat(f, stream), true);
+});
+
+test("it applies to livestreams ONLY, and only when configured", () => {
+ const f = compileDownloadFilter({
+ include: "guest",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ // A rejected plain upload is a plain rejection: there is no chat to keep.
+ assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected");
+ assert.equal(titleFilterWantsChat(f, liveMeta("not_live")), false);
+ // A video the filter WANTS is unaffected either way.
+ assert.equal(
+ classifyAgainstFilter(f, liveMeta("was_live", "a guest appears")),
+ "text",
+ );
+
+ // The default, and every channel written before the field: absent = skip.
+ const plain = compileDownloadFilter({ include: "guest" });
+ assert.ok(plain);
+ assert.equal(plain.rejectedLivestreams, "skip");
+ assert.equal(classifyAgainstFilter(plain, liveMeta("was_live")), "rejected");
+ assert.equal(titleFilterWantsChat(plain, liveMeta("was_live")), false);
+});
+
+test("an EXCLUDE-matched livestream is chat-only too, because the mode is about the rejection", () => {
+ // The setting says what to do with a rejected livestream, not which selector
+ // did the rejecting. One rule, in one place, so the two cannot diverge.
+ const f = compileDownloadFilter({
+ exclude: "rerun",
+ rejectedLivestreams: "chat-only",
+ });
+ assert.ok(f);
+ assert.equal(
+ classifyAgainstFilter(f, liveMeta("was_live", "rerun of last night")),
+ "chat-only",
+ );
+ // Exclude-only is still "everything else passes" — the mode adds no filtering.
+ assert.equal(classifyAgainstFilter(f, liveMeta("was_live")), "text");
+});
+
+test("the mode alone is not a filter, and is never the reason a channel is filtered", () => {
+ // Nothing rejecting means nothing for the mode to say, so it does not make a
+ // channel filtered — which is also what keeps the download lane from offering
+ // a metadata scan to a channel that has no filter at all.
+ assert.equal(compileDownloadFilter({ rejectedLivestreams: "chat-only" }), null);
+});
+
+test("evaluateDownloadFilters marks the decision, and it is still a skip", () => {
+ const decision = evaluateDownloadFilters(
+ ctx({
+ metadata: meta({ title: "Synthetic plainvid0001", live_status: "was_live" }),
+ channelConfig: {
+ ...BASE_CONFIG,
+ downloadFilter: { include: "guest", rejectedLivestreams: "chat-only" },
+ },
+ }),
+ );
+ assert.ok(decision);
+ // `skip` is what every media-fetching path reads, and it is unchanged: the
+ // downloader is the one caller that asks the next question.
+ assert.equal(decision.skip, true);
+ assert.equal(decision.filter, "titleFilter");
+ assert.equal(decision.chatOnly, true);
+ assert.match(decision.reason, /live chat only/);
+});
+
diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts
@@ -10,7 +10,11 @@
// Adding a filter = append one entry to FILTERS. Each filter sees the same
// context (metadata + channel config + resolved settings).
-import type { ChannelConfig, DownloadFilterConfig } from "./channelConfig";
+import type {
+ ChannelConfig,
+ DownloadFilterConfig,
+ RejectedLivestreamMode,
+} from "./channelConfig";
import type { RawMetadata } from "./transcripts-server";
export type DownloadFilterSettings = {
@@ -32,6 +36,11 @@ export type DownloadFilterDecision = {
skip: boolean;
filter: string;
reason: string;
+ // The video's MEDIA is skipped and its live chat is wanted — see
+ // DownloadFilterConfig.rejectedLivestreams. `skip` is still true: everything
+ // that decides whether to fetch the media reads that and is unchanged. The
+ // downloader is the one caller that asks the next question.
+ chatOnly?: boolean;
};
type DownloadFilter = {
@@ -80,6 +89,9 @@ export type CompiledDownloadFilter = {
include: RegExp | null;
exclude: RegExp | null;
includeLivestreams: boolean;
+ // See DownloadFilterConfig.rejectedLivestreams. "skip" is the default and is
+ // what every channel written before the field did.
+ rejectedLivestreams: RejectedLivestreamMode;
};
// Compile a channel's filter. Returns null when there is no filter AND when a
@@ -96,11 +108,17 @@ export function compileDownloadFilter(
const exclude = filter?.exclude?.trim() ?? "";
const includeLivestreams = filter?.includeLivestreams === true;
if (!include && !exclude && !includeLivestreams) return null;
+ // The mode is NOT a positive selector and deliberately does not make a
+ // channel "filtered" on its own: it says what to do with a rejection, and
+ // with nothing rejecting there is nothing for it to say.
+ const rejectedLivestreams: RejectedLivestreamMode =
+ filter?.rejectedLivestreams === "chat-only" ? "chat-only" : "skip";
try {
return {
include: include ? new RegExp(include, "i") : null,
exclude: exclude ? new RegExp(exclude, "i") : null,
includeLivestreams,
+ rejectedLivestreams,
};
} catch {
return null;
@@ -192,7 +210,18 @@ export function downloadFilterPatternProblem(pattern: string): string | null {
// Why a video passed, or that it didn't. "livestream" exists so the UI can say
// how many videos a channel is keeping for a reason other than their name.
-export type DownloadFilterVerdict = "text" | "livestream" | "rejected";
+//
+// "chat-only" IS A REJECTION with an instruction attached, not a fourth way to
+// pass. The video is not wanted; its live chat is. Everything that asks "is
+// this video wanted?" — `titleFilterRejects`, and therefore the settled set and
+// the download queue — must keep answering yes-it-is-rejected for it, or the
+// downloader would fetch the media the operator asked it not to. Only the
+// callers that ask the NEXT question ("and then what?") look for this value.
+export type DownloadFilterVerdict =
+ | "text"
+ | "livestream"
+ | "rejected"
+ | "chat-only";
// PURE, and the SINGLE matcher. Both the download-time registry entry below and
// the derived settled set (controller/metadataScanStore.ts) call exactly this,
@@ -217,19 +246,50 @@ export function classifyAgainstFilter(
meta: FilterableVideo,
): DownloadFilterVerdict {
const text = downloadFilterText(meta);
- if (compiled.exclude && compiled.exclude.test(text)) return "rejected";
+ if (compiled.exclude && compiled.exclude.test(text)) {
+ return rejection(compiled, meta);
+ }
const hasPositive = Boolean(compiled.include || compiled.includeLivestreams);
if (!hasPositive) return "text";
if (compiled.include && compiled.include.test(text)) return "text";
if (compiled.includeLivestreams && isLivestream(meta)) return "livestream";
- return "rejected";
+ return rejection(compiled, meta);
}
+// HOW a rejection is spelled — the ONE place, so `exclude`-matched and
+// unmatched livestreams cannot get different answers. The operator's setting is
+// about what to do with a rejected livestream, not about which selector did the
+// rejecting.
+function rejection(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): DownloadFilterVerdict {
+ return compiled.rejectedLivestreams === "chat-only" && isLivestream(meta)
+ ? "chat-only"
+ : "rejected";
+}
+
+// IS THIS VIDEO'S MEDIA UNWANTED? True for "chat-only" too — see
+// DownloadFilterVerdict. This is what `settledByTitleFilterIds` and therefore
+// `undownloadedIds` are built on, so a chat-only video is settled exactly like
+// any other rejection and the download queue never offers its media.
export function titleFilterRejects(
compiled: CompiledDownloadFilter,
meta: FilterableVideo,
): boolean {
- return classifyAgainstFilter(compiled, meta) === "rejected";
+ const verdict = classifyAgainstFilter(compiled, meta);
+ return verdict === "rejected" || verdict === "chat-only";
+}
+
+// Is this one the operator wants the CHAT of? A separate question from the one
+// above, asked by exactly the two places that act on the answer: the downloader,
+// which fetches the chat instead of skipping, and the snapshot, which puts the
+// video in its own bucket rather than in `skippedByTitleFilter`.
+export function titleFilterWantsChat(
+ compiled: CompiledDownloadFilter,
+ meta: FilterableVideo,
+): boolean {
+ return classifyAgainstFilter(compiled, meta) === "chat-only";
}
// Why a rejection happened, for the log and the outcome record.
@@ -288,10 +348,14 @@ const titleFilter: DownloadFilter = {
const m = ctx.metadata;
if (!m) return null; // fail open — see above
if (!titleFilterRejects(compiled, m)) return null;
+ const chatOnly = titleFilterWantsChat(compiled, m);
return {
skip: true,
filter: "titleFilter",
- reason: titleFilterReason(compiled, m),
+ reason: chatOnly
+ ? `${titleFilterReason(compiled, m)} — fetching its live chat only`
+ : titleFilterReason(compiled, m),
+ ...(chatOnly ? { chatOnly: true } : {}),
};
},
};
diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts
@@ -27,7 +27,15 @@ export type DownloadOutcomeStatus =
// The app-level filter pass (e.g. skip-live) declined to download this video.
// Not a failure and not archived — the next sync/download-missing retries it
// once the filter no longer matches (e.g. a live stream becomes a VOD).
- | "skipped-filtered";
+ | "skipped-filtered"
+ // The download filter rejected this livestream and the channel's
+ // `rejectedLivestreams` mode is "chat-only": the MEDIA was not fetched, the
+ // live chat was. The directory holds metadata.info.json and
+ // transcript.live_chat.json and nothing else, so the video is indexable (a
+ // chat track, no captions) while every "is it downloaded?" predicate — all of
+ // which test whisper/VTT/audio — still answers no. Terminal on the operator's
+ // terms, like skipped-filtered, and equally re-decided by editing the filter.
+ | "chat-only";
export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus> = [
"ok",
@@ -39,6 +47,7 @@ export const DOWNLOAD_OUTCOME_STATUS_VALUES: ReadonlyArray<DownloadOutcomeStatus
"corrupt-full-source",
"failed-short-audio",
"skipped-filtered",
+ "chat-only",
];
export type DownloadAttemptKind =
@@ -52,7 +61,11 @@ export type DownloadAttemptKind =
// A cookie re-run of a metadata prefetch that failed with an auth/age error
// (cookie mode "always"/"when-required" with a cookie value configured).
// Recorded with n: 0 alongside the failed prefetch it retries.
- | "metadata-prefetch-auth-retry";
+ | "metadata-prefetch-auth-retry"
+ // The live-chat pass a "chat-only" filter verdict runs INSTEAD of a download:
+ // --skip-download --write-subs --sub-langs live_chat, no media, no archive
+ // line. Recorded with n: 1, since it is the only real attempt there is.
+ | "live-chat-only";
export type AudioCheckProbeVerdict = "clean" | "partial" | "malformed";
diff --git a/common/views/pipeline/channelFlow.ts b/common/views/pipeline/channelFlow.ts
@@ -413,6 +413,17 @@ export function computeChannelFlow(
"diagnostics",
"Declined as currently live or upcoming; retried on a later sync.",
),
+ // THE GAP IS REAL AND IT IS THIS STATION'S. A chat-only video is not
+ // settled out of the listing (it is not in skippedByTitleFilter), so it
+ // counts as expected here — and until its chat lands it has no directory,
+ // so it is missing from totals.videos. That is exactly the gap a siding
+ // exists to name.
+ ...siding(
+ "chat only, not fetched",
+ buckets.chatOnlyPending.length,
+ "diagnostics",
+ "Livestreams the filter rejected on a channel set to keep the chat. The lane fetches the live chat last — after every real download — because it costs no media.",
+ ),
...siding(
"need cookies",
diff --git a/common/views/pipeline/stageStatus.ts b/common/views/pipeline/stageStatus.ts
@@ -39,6 +39,8 @@ export function normalizeBuckets(
nonStandardVtt: raw?.nonStandardVtt ?? [],
skippedByFilter: raw?.skippedByFilter ?? [],
skippedByTitleFilter: raw?.skippedByTitleFilter ?? [],
+ chatOnly: raw?.chatOnly ?? [],
+ chatOnlyPending: raw?.chatOnlyPending ?? [],
incompleteTranscript: raw?.incompleteTranscript ?? [],
shortAudio: raw?.shortAudio ?? [],
autoSubsOnly: raw?.autoSubsOnly ?? [],
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -54,6 +54,8 @@ import {
isLivestreamMetadata,
} from "../lib/transcripts-server";
import { evaluateDownloadFilters } from "../lib/downloadFilters";
+import { normalizeLiveChat } from "../controller/normalizeLiveChat";
+import { LIVE_CHAT_FILENAME } from "../lib/videoStatus";
import { CLIPS_DIR_NAME } from "../lib/clipWindow";
import {
upsertMetadataScan,
@@ -404,6 +406,58 @@ function runOneYtdlp(
);
}
+// ONE EXTRA yt-dlp PASS THAT FETCHES A LIVESTREAM'S CHAT AND NOTHING ELSE.
+//
+// Reached only from the filter branch below, for a channel whose
+// `rejectedLivestreams` is "chat-only". The media is not wanted; the chat is.
+//
+// WHAT THE ARGUMENT ORDER IS FOR. yt-dlp takes the LAST occurrence of an
+// option, so the refusals go AFTER the channel's own extra args — a channel
+// that configured `--write-thumbnail` or `--write-auto-subs` must not be able
+// to turn this pass back into a partial download of things nobody asked for.
+// Same rule fetchWindowManaged relies on, and for the same reason.
+//
+// --no-write-info-json, NOT the absence of --write-info-json: the metadata is
+// ALREADY on disk from the prefetch, and it has to stay there — a video dir
+// with no metadata.info.json is invisible to buildIndex, so the chat would be
+// fetched and then never published. This pass must neither rewrite it nor
+// delete it.
+//
+// No archive line is appended. An archive id means "downloaded" to
+// verifyTranscripts and to the sync walk, and this video is not.
+async function fetchLiveChatOnly(
+ opts: ManagedDownloadOpts,
+ channelDir: string,
+ videoDir: string,
+ cookies: string | undefined,
+): Promise<AttemptOutcome> {
+ return runOneYtdlp(opts, channelDir, [
+ "--ignore-config",
+ "--restrict-filenames",
+ ...channelConfigArgs(opts.channelConfig, cookies),
+ "--skip-download",
+ "--write-subs",
+ "--no-write-auto-subs",
+ "--sub-langs",
+ "live_chat",
+ "--no-write-info-json",
+ "--no-write-description",
+ "--no-write-thumbnail",
+ "--no-download-archive",
+ ...outputArgsForUrl(opts.videoUrl),
+ "--",
+ opts.videoUrl,
+ ]);
+}
+
+// Did the chat pass actually leave a chat behind? A stream with chat replay
+// disabled makes yt-dlp exit 0 and write nothing, and a directory holding only
+// a metadata.info.json is the leftover this slice exists to stop creating.
+async function hasLiveChatOnDisk(videoDir: string): Promise<boolean> {
+ const entries = await readdir(videoDir).catch(() => [] as string[]);
+ return entries.includes(LIVE_CHAT_FILENAME);
+}
+
function attemptSucceeded(exitCode: number | null): boolean {
// yt-dlp: 0 = clean, 101 = break-on-existing / max-downloads (clean stop).
return exitCode === 0 || exitCode === 101;
@@ -701,11 +755,74 @@ async function runManagedDownload(
}
}
}
+ // ── CHAT ONLY ────────────────────────────────────────────────────
+ //
+ // The operator asked for this livestream's chat and not its media, so
+ // this is the one rejection that FETCHES something — and the one that
+ // keeps its prefetch directory, deliberately. A chat-only dir holds
+ // metadata.info.json (without it buildIndex never sees the video and the
+ // chat is published nowhere) plus transcript.live_chat.json, and nothing
+ // else. Every "is it downloaded?" predicate tests whisper, an English VTT
+ // or audio, so it still answers no — which is exactly right: this video
+ // is a chat track, not a download.
+ let chatOnlyFetched = false;
+ if (decision.chatOnly && !opts.signal.aborted) {
+ opts.onLog(
+ `Fetching the live chat for ${canonicalId} (media skipped by the download filter).\n`,
+ );
+ const chatRes = await fetchLiveChatOnly(
+ opts,
+ channelDir,
+ videoDir,
+ alwaysCookies(cookiePolicy) ??
+ (prefetchNeededCookies ? authRetryCookies(cookiePolicy) : undefined),
+ );
+ attempts.push({
+ n: 1,
+ kind: "live-chat-only",
+ handling: opts.channelConfig.handling,
+ usedCookies: Boolean(alwaysCookies(cookiePolicy)),
+ ytdlpExitCode: chatRes.exitCode,
+ error: attemptSucceeded(chatRes.exitCode)
+ ? undefined
+ : trimError(chatRes.stderrTail),
+ });
+ // The CUES sidecar, not just the raw file: buildIndex prefers
+ // live_chat.cues.json when it is fresh, and normalizing here means a
+ // chat-only video is index-ready the moment it lands rather than
+ // waiting for the corpus-wide normalize pass.
+ if (attemptSucceeded(chatRes.exitCode)) {
+ chatOnlyFetched = await hasLiveChatOnDisk(videoDir);
+ if (chatOnlyFetched) {
+ try {
+ await normalizeLiveChat({
+ videoDir,
+ channelSlug: opts.channelSlug,
+ ...(opts.channelConfig.name
+ ? { configName: opts.channelConfig.name }
+ : {}),
+ log: (m: string) => opts.onLog(`${m}\n`),
+ });
+ } catch (err) {
+ opts.onLog(
+ `Could not normalize the live chat: ${(err as Error).message}\n`,
+ );
+ }
+ } else {
+ // The source served no chat for this stream. Nothing was fetched,
+ // so the directory is not a corpus member and goes the way every
+ // other rejection's does.
+ opts.onLog(
+ `No live chat was available for ${canonicalId}; nothing to keep.\n`,
+ );
+ }
+ }
+ }
const finishedAt = new Date().toISOString();
const record: DownloadOutcomeRecord = {
videoId: canonicalId,
webpageUrl: opts.videoUrl,
- status: "skipped-filtered",
+ status: chatOnlyFetched ? "chat-only" : "skipped-filtered",
startedAt,
finishedAt,
attempts,
@@ -717,7 +834,7 @@ async function runManagedDownload(
// sidecar is what `skippedByFilter` is derived from, so only this one
// discards.
const discarded =
- decision.filter === "titleFilter" && metadata
+ decision.filter === "titleFilter" && metadata && !chatOnlyFetched
? await discardPrefetchDir(videoDir, opts.onLog)
: false;
if (discarded) {
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -948,6 +948,12 @@ async function runManagedDownloads(
}
if (outcome.status === "skipped-filtered") {
skippedCount++;
+ } else if (outcome.status === "chat-only") {
+ // The media was skipped and the live chat was fetched. Counted as a
+ // SKIP, because the batch's `okCount` means "videos downloaded" and
+ // this one deliberately was not — the chat is a different artifact
+ // and the report's own chatOnly bucket is where it is counted.
+ skippedCount++;
} else if (outcome.status === "corrupt-full-source") {
// Completed but stayed malformed after one re-download: terminal and
// kept on disk, but not a usable download. Count as skipped (not ok,
diff --git a/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx b/editor/app/channels/[slug]/components/stages/DiagnosticsStage.tsx
@@ -55,6 +55,8 @@ type Props = {
nonStandardVttIds: string[];
skippedByFilterIds: string[];
skippedByTitleFilterIds: string[];
+ chatOnlyIds: string[];
+ chatOnlyPendingIds: string[];
totals: { videos: number; transcribed: number; downloaded: number };
availability: AvailabilitySnapshot;
maybeMissing: { ids: string[]; checkedAt: string };
@@ -73,6 +75,8 @@ export function DiagnosticsStage({
nonStandardVttIds,
skippedByFilterIds,
skippedByTitleFilterIds,
+ chatOnlyIds,
+ chatOnlyPendingIds,
totals,
availability,
maybeMissing,
@@ -131,6 +135,23 @@ export function DiagnosticsStage({
"The metadata scan read these titles and this channel's download filter (Configure > Advanced) rejects them, so nothing downloads or re-attempts them. Change the include/exclude patterns and they are re-decided on the next report — no rescan needed.",
ariaLabel: "filtered out",
},
+ {
+ ids: chatOnlyIds,
+ label: "Chat only (filtered livestreams)",
+ // No Retry either, and for the same reason as the bucket above: these are
+ // settled, and the thing to change is the mode in Configure. A retry
+ // would offer to download the media the operator said not to.
+ description:
+ "Livestreams this channel's download filter rejected while \u201cFiltered-out livestreams\u201d is set to keep the chat (Configure > Advanced). The media was never fetched; the live chat was, and it is published as a chat track with no captions. Set the mode back to Skip and they stop being fetched \u2014 what is already on disk stays.",
+ ariaLabel: "chat only",
+ },
+ {
+ ids: chatOnlyPendingIds,
+ label: "Chat only: chat not fetched yet",
+ description:
+ "Filtered-out livestreams whose live chat has not been fetched yet. The download lane picks these up last \u2014 after every real download \u2014 because a chat pass costs one metadata-sized request and no media.",
+ ariaLabel: "chat only pending",
+ },
];
const populated = buckets.filter((b) => b.ids.length > 0);
const total = populated.reduce((acc, b) => acc + b.ids.length, 0);
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -559,6 +559,8 @@ export default async function ChannelDetailPage({
nonStandardVttIds={buckets.nonStandardVtt}
skippedByFilterIds={buckets.skippedByFilter}
skippedByTitleFilterIds={buckets.skippedByTitleFilter}
+ chatOnlyIds={buckets.chatOnly ?? []}
+ chatOnlyPendingIds={buckets.chatOnlyPending ?? []}
totals={snapshot.totals}
availability={normalizeAvailability(snapshot.availability)}
maybeMissing={normalizeMaybeMissing(snapshot.maybeMissing)}
diff --git a/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx b/editor/app/channels/[slug]/videos/[id]/components/VideoPanel.tsx
@@ -1603,6 +1603,11 @@ function DownloadOutcomeBadge({
? `Filtered out by the download filter${why}`
: `Skipped by filter${why}`;
}
+ // NOT a download, and the badge has to say so plainly: this video's media
+ // was never fetched and never will be while the filter stands. What is
+ // here is its live chat, which is why the video is in the corpus at all.
+ case "chat-only":
+ return "Chat only (filtered livestream) — live chat kept, no media and no transcript";
default:
return outcome.status;
}
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -731,6 +731,27 @@ export function ChannelForm({
</span>
</span>
</label>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Filtered-out livestreams</span>
+ <select
+ name="downloadFilterRejectedLivestreams"
+ defaultValue={c?.downloadFilter?.rejectedLivestreams ?? "skip"}
+ aria-label="filtered-out livestreams"
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="skip">Skip them — download nothing</option>
+ <option value="chat-only">
+ Keep the live chat — no audio, no transcript
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ What to do with a livestream the filter above rejected. A multi-hour
+ stream whose title says nothing is rarely worth its audio, but its
+ live chat is text, it is small, and it is the only record of what the
+ room said. Chat-only videos join the corpus as a chat track with no
+ captions; they are never downloaded and never transcribed.
+ </span>
+ </label>
<Field
label="Cookies from browser"
name="cookiesFromBrowser"
diff --git a/editor/app/channels/components/channelConfigToForm.ts b/editor/app/channels/components/channelConfigToForm.ts
@@ -48,6 +48,7 @@ export const CHANNEL_FORM_VALUES = [
"cookieMode",
"downloadFilterInclude",
"downloadFilterExclude",
+ "downloadFilterRejectedLivestreams",
"audioCheckIntervalSeconds",
"audioCheckMaxRollbacks",
"audioCheckCopyTimeoutSeconds",
@@ -103,6 +104,14 @@ export function channelConfigToFormData(config: ChannelConfig): FormData {
if (config.downloadFilter?.includeLivestreams) {
fd.set("downloadFilterIncludeLivestreams", "on");
}
+ // Omitted when it is "skip" — the parser reads an absent field as "skip", so
+ // emitting it would be a value with no effect, and a patch clearing it ("")
+ // has to land on the same default the form's own select does.
+ put(
+ fd,
+ "downloadFilterRejectedLivestreams",
+ config.downloadFilter?.rejectedLivestreams,
+ );
if (config.audioCheck?.enabled) {
fd.set("audioCheckEnabled", "on");
putNum(fd, "audioCheckIntervalSeconds", config.audioCheck.intervalSeconds);
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -22,6 +22,7 @@ import { handleFromAccountUrl } from "yt-dlp-transcript-common/social/fetchers";
import { isCookieMode } from "yt-dlp-transcript-common/lib/cookiePolicy";
import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import { downloadFilterPatternProblem } from "yt-dlp-transcript-common/lib/downloadFilters";
+import { isRejectedLivestreamMode } from "yt-dlp-transcript-common/lib/channelConfig";
export type ParsedChannelForm = {
name: string;
@@ -241,12 +242,33 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
}
const includeLivestreams =
formData.get("downloadFilterIncludeLivestreams") != null;
+ // WHAT TO DO WITH A REJECTED LIVESTREAM. Only ever stored alongside a real
+ // filter and only when it is not the default — with nothing rejecting, the
+ // mode names a decision that can never be taken, and a stored "skip" would be
+ // a key on disk that changes nothing.
+ const rejectedLivestreamsRaw = String(
+ formData.get("downloadFilterRejectedLivestreams") ?? "",
+ ).trim();
+ if (
+ rejectedLivestreamsRaw &&
+ !isRejectedLivestreamMode(rejectedLivestreamsRaw)
+ ) {
+ throw new Error(
+ `Filtered-out livestreams must be "skip" or "chat-only" (got ${JSON.stringify(
+ rejectedLivestreamsRaw,
+ )})`,
+ );
+ }
+ const rejectedLivestreams = isRejectedLivestreamMode(rejectedLivestreamsRaw)
+ ? rejectedLivestreamsRaw
+ : "skip";
const downloadFilter: ChannelConfig["downloadFilter"] | undefined =
downloadFilterInclude || downloadFilterExclude || includeLivestreams
? {
...(downloadFilterInclude ? { include: downloadFilterInclude } : {}),
...(downloadFilterExclude ? { exclude: downloadFilterExclude } : {}),
...(includeLivestreams ? { includeLivestreams: true } : {}),
+ ...(rejectedLivestreams !== "skip" ? { rejectedLivestreams } : {}),
}
: undefined;
diff --git a/editor/e2e/chat-only.spec.ts b/editor/e2e/chat-only.spec.ts
@@ -0,0 +1,352 @@
+import { readFile, rm, writeFile } from "node:fs/promises";
+import { test, expect, type Page } from "@playwright/test";
+import {
+ buildIndex,
+ channelStage,
+ generateReport,
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSite,
+} from "./helpers";
+import { baseUrl } from "./baseUrl";
+
+// THE THIRD ANSWER FOR A FILTERED-OUT LIVESTREAM.
+//
+// A multi-hour stream whose title says nothing about the subject is rarely
+// worth its audio, but its live chat is text, it is small, and it is the only
+// record of what the room said. `downloadFilter.rejectedLivestreams:
+// "chat-only"` is that answer: the media is never fetched, the chat is, and the
+// video joins the corpus as a chat track with no captions.
+//
+// What every test here is really guarding is one risk. A chat-only directory
+// makes "metadata, no transcript" a LEGITIMATE shape — and that is exactly what
+// buildIndex.ts:309-315 and deriveChannelSets read as "this video was fetched".
+// Get the bucket rules wrong and filtered livestreams silently leave
+// missingNeverFetched and enter the published site as blank pages.
+
+const CHANNEL = "test-filter";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+const GUEST = ["guestvid0001", "guestvid0002"];
+const PLAIN = ["plainvid0001", "plainvid0002"];
+// The fixture's FINISHED livestream VOD: `livevid` makes the fake report
+// live_status "was_live", and its title contains neither "guest" nor "plain",
+// so the include pattern rejects it and only the livestream rules decide it.
+const LIVE = "livevid000001";
+
+type Snapshot = {
+ generatedAt: string;
+ totals: { videos: number; transcribed: number; downloaded: number };
+ buckets: {
+ skippedByTitleFilter?: string[];
+ chatOnly?: string[];
+ chatOnlyPending?: string[];
+ noTranscript?: string[];
+ downloadedNoTranscript?: string[];
+ };
+ undownloadedIds: string[];
+};
+
+async function writeFilterConfig(filter: Record<string, unknown> | null) {
+ await writeFile(
+ resolvePath(`${ROOT}/config.json`),
+ JSON.stringify(
+ {
+ handling: "youtube",
+ name: "Test Title Filter",
+ url: "https://www.youtube.com/@example/videos",
+ ...(filter ? { downloadFilter: filter } : {}),
+ },
+ null,
+ 2,
+ ),
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+}
+
+async function refreshReport(
+ page: Page,
+ after: string,
+ until?: (snapshot: Snapshot) => boolean,
+): Promise<Snapshot> {
+ await page.goto("/channels");
+ const refresh = page.getByRole("button", {
+ name: `refresh report ${CHANNEL}`,
+ });
+ await refresh.waitFor({ state: "visible" });
+ await expect
+ .poll(
+ async () => {
+ await refresh.click({ timeout: 5_000 }).catch(() => {});
+ for (let i = 0; i < 20; i++) {
+ const cur = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch(
+ () => null,
+ );
+ if (cur && cur.generatedAt > after && (!until || until(cur))) {
+ return true;
+ }
+ await new Promise((r) => setTimeout(r, 250));
+ }
+ return false;
+ },
+ { timeout: 60_000, intervals: [1000] },
+ )
+ .toBe(true);
+ return readJson<Snapshot>(`${ROOT}/snapshot.json`);
+}
+
+async function download(page: Page): Promise<void> {
+ await page.goto(channelStage(CHANNEL, "download"));
+ await page.getByRole("button", { name: "Download videos" }).click();
+ await expect(page.getByLabel("Download videos output")).toContainText(
+ "Managed download complete",
+ { timeout: 60_000 },
+ );
+}
+
+test("a chat-only livestream gets its chat and stays undownloaded", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+
+ await download(page);
+
+ // The chat, and the metadata that makes it publishable. Nothing else.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.live_chat.json`)).toBe(
+ true,
+ );
+ // metadata.info.json HAS to survive: buildIndex stats it per video dir and
+ // skips the directory when it throws, so without it the chat would be fetched
+ // and then published nowhere.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/metadata.info.json`)).toBe(true);
+ // Normalized on the spot, so the video is index-ready when it lands rather
+ // than waiting for the corpus-wide normalize pass.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/live_chat.cues.json`)).toBe(true);
+ // NOT a download, by every definition the app has.
+ expect(await pathExists(`${ROOT}/data/${LIVE}/transcript.en.vtt`)).toBe(false);
+ expect(await pathExists(`${ROOT}/data/${LIVE}/audio.mp3`)).toBe(false);
+ const outcome = await readJson<{ status: string; attempts: unknown[] }>(
+ `${ROOT}/data/${LIVE}/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("chat-only");
+ // No archive line: an archive id means "downloaded" to the sync walk and to
+ // verifyTranscripts, and this video is not.
+ const archive = await readFile(resolvePath(`${ROOT}/archive`), "utf8").catch(
+ () => "",
+ );
+ expect(archive).not.toContain(LIVE);
+
+ // The ordinary rejections still leave nothing behind, and the matches still
+ // download — the mode changes what happens to LIVESTREAMS and nothing else.
+ for (const id of PLAIN) {
+ expect(await pathExists(`${ROOT}/data/${id}`)).toBe(false);
+ }
+ for (const id of GUEST) {
+ expect(await pathExists(`${ROOT}/data/${id}/transcript.en.vtt`)).toBe(true);
+ }
+});
+
+test("a chat-only video is in chatOnly, and in no bucket that means work", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+ await download(page);
+
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const snap = await refreshReport(
+ page,
+ before.generatedAt,
+ (s) => (s.buckets.chatOnly ?? []).length > 0,
+ );
+
+ expect(snap.buckets.chatOnly ?? []).toEqual([LIVE]);
+ // NOT "we decided not to have it" — we decided to have its chat.
+ expect(snap.buckets.skippedByTitleFilter ?? []).toEqual([...PLAIN].sort());
+ // Nothing is going to transcribe a video whose audio we chose not to fetch,
+ // so it must be out of both transcription-facing buckets AND the download
+ // queue, or the lanes would fight the filter forever.
+ expect(snap.buckets.noTranscript ?? []).not.toContain(LIVE);
+ expect(snap.buckets.downloadedNoTranscript ?? []).not.toContain(LIVE);
+ expect(snap.undownloadedIds).not.toContain(LIVE);
+ // Its chat is on disk, so there is no outstanding chat work either.
+ expect(snap.buckets.chatOnlyPending ?? []).toEqual([]);
+ // A MEMBER OF THIS CORPUS, not a stub: it has a directory and it is counted.
+ // (guestvid0001, guestvid0002 and the chat-only livestream; needsauthvid01
+ // leaves a failure directory too, so this is a floor rather than an equality.)
+ expect(snap.totals.videos).toBeGreaterThanOrEqual(3);
+ expect(snap.totals.downloaded).toBe(GUEST.length);
+});
+
+test("before the chat lands it is chatOnlyPending, which is download-lane work", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("title-filter-channel");
+ await writeFilterConfig({ include: "guest", rejectedLivestreams: "chat-only" });
+ await generateReport(page, CHANNEL);
+
+ // Seed the scan store directly: the scan is what turns a listed id into a
+ // chat-only verdict without fetching anything, and this is the state the
+ // download lane is meant to find.
+ await writeFile(
+ resolvePath(`${ROOT}/metadata-scan.json`),
+ JSON.stringify({
+ version: 1,
+ entries: {
+ [LIVE]: {
+ title: `Synthetic ${LIVE}`,
+ description: "",
+ uploadDate: "20240101",
+ liveStatus: "was_live",
+ scannedAt: "2026-01-01T00:00:00.000Z",
+ },
+ },
+ errors: {},
+ lastRun: null,
+ }),
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+
+ const before = await readJson<Snapshot>(`${ROOT}/snapshot.json`);
+ const snap = await refreshReport(
+ page,
+ before.generatedAt,
+ (s) => (s.buckets.chatOnlyPending ?? []).length > 0,
+ );
+ expect(snap.buckets.chatOnlyPending ?? []).toEqual([LIVE]);
+ // It is NOT in undownloadedIds — fetching the media is the one thing this
+ // video must not have done to it — and not in skippedByTitleFilter either.
+ expect(snap.undownloadedIds).not.toContain(LIVE);
+ expect(snap.buckets.skippedByTitleFilter ?? []).not.toContain(LIVE);
+});
+
+// The index half, on the cheapest fixture that has one. The claim: a directory
+// holding metadata.info.json and transcript.live_chat.json and NO transcript is
+// admitted, published with its chat track, and carries no cues.
+const INDEX_CHANNEL = "test-youtube";
+const INDEX_DIR = "20240101_test1234567";
+const INDEX_DATA = `test-transcripts/channels/${INDEX_CHANNEL}/data/${INDEX_DIR}`;
+
+test("the export publishes a chat-only video as a chat track with no captions", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData("one-youtube-channel-with-data");
+ // Make it chat-only-shaped: the chat, and no transcript at all.
+ await rm(resolvePath(`${INDEX_DATA}/transcript.en.vtt`), { force: true });
+ await writeFile(
+ resolvePath(`${INDEX_DATA}/transcript.live_chat.json`),
+ '{"clientId":"fake","action":{"addChatItemAction":{"item":{"liveChatTextMessageRenderer":' +
+ '{"authorName":{"simpleText":"viewer"},"message":{"runs":[{"text":"platypus"}]},' +
+ '"timestampUsec":"11000000","videoOffsetTimeMsec":"11000"}}}}}\n',
+ );
+ await writeSite("testsite", {
+ channels: [{ slug: INDEX_CHANNEL, groupId: "default" }],
+ });
+ await buildIndex(page);
+
+ const subsPage = await readJson<
+ Array<{ id: string; tracks?: Record<string, unknown[]> }>
+ >(`test-transcripts/.export-index/shared/subs/${INDEX_CHANNEL}/page-0000.json`);
+ const detail = subsPage.find((d) => d.id === INDEX_DIR);
+ expect(detail).toBeDefined();
+ expect(Object.keys(detail?.tracks ?? {})).toEqual(["live_chat"]);
+ expect((detail?.tracks?.live_chat ?? []).length).toBeGreaterThan(0);
+
+ // And no captions: the transcripts shard carries the video (it has metadata)
+ // with no cues at all.
+ const transcriptsPage = await readJson<Array<{ id: string; cues?: unknown[] }>>(
+ `test-transcripts/.export-index/shared/transcripts/${INDEX_CHANNEL}/page-0000.json`,
+ );
+ const t = transcriptsPage.find((d) => d.id === INDEX_DIR);
+ expect(t).toBeDefined();
+ expect(t?.cues?.length ?? 0).toBe(0);
+});
+
+test("the form round-trips the chat-only tier and the ops API sets it", async ({
+ page,
+ request,
+}) => {
+ test.setTimeout(120_000);
+ await resetData("title-filter-channel");
+ await generateReport(page, CHANNEL);
+ await page.goto(channelStage(CHANNEL, "configure"));
+
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ const select = page.getByLabel("filtered-out livestreams");
+ // Absent on disk reads as "skip", which is what every channel written before
+ // the field did.
+ await expect(select).toHaveValue("skip");
+
+ await select.selectOption("chat-only");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () =>
+ (
+ await readJson<{
+ downloadFilter?: { rejectedLivestreams?: string };
+ }>(`${ROOT}/config.json`)
+ ).downloadFilter?.rejectedLivestreams ?? null,
+ { timeout: 30_000 },
+ )
+ .toBe("chat-only");
+
+ // Back to the default, and the key is REMOVED rather than stored as "skip":
+ // a stored default is a key on disk that changes nothing.
+ await page.goto(channelStage(CHANNEL, "configure"));
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await expect(page.getByLabel("filtered-out livestreams")).toHaveValue(
+ "chat-only",
+ );
+ await page.getByLabel("filtered-out livestreams").selectOption("skip");
+ await page.getByRole("button", { name: "Save changes" }).click();
+ await expect
+ .poll(
+ async () =>
+ "rejectedLivestreams" in
+ ((
+ await readJson<{ downloadFilter?: Record<string, unknown> }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter ?? {}),
+ { timeout: 30_000 },
+ )
+ .toBe(false);
+
+ // The ops API goes through the same parser, so it gets the same validation.
+ const ok = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ data: {
+ slug: CHANNEL,
+ patch: { downloadFilterRejectedLivestreams: "chat-only" },
+ },
+ });
+ expect(ok.ok()).toBeTruthy();
+ expect(
+ (
+ await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter?.rejectedLivestreams,
+ ).toBe("chat-only");
+
+ const bad = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ data: { slug: CHANNEL, patch: { downloadFilterRejectedLivestreams: "maybe" } },
+ });
+ expect(bad.status()).toBeGreaterThanOrEqual(400);
+ // Refused, and the stored value is untouched.
+ expect(
+ (
+ await readJson<{ downloadFilter?: { rejectedLivestreams?: string } }>(
+ `${ROOT}/config.json`,
+ )
+ ).downloadFilter?.rejectedLivestreams,
+ ).toBe("chat-only");
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -772,6 +772,50 @@ async function main() {
return;
}
+ // CHAT ONLY: --skip-download --write-subs --sub-langs live_chat, with
+ // auto-subs and the info json explicitly refused. Must come BEFORE the
+ // youtube single-URL branch below, which matches --write-auto-subs (this
+ // invocation passes --no-write-auto-subs, so it would fall through to the
+ // final error instead).
+ //
+ // It writes ONLY transcript.live_chat.json — no metadata.info.json (the
+ // prefetch already wrote it and this pass must not touch it), no transcript,
+ // no audio, and no DLOM_ARCHIVE line. A fake that wrote any of those would
+ // make the spec pass for the wrong reason: the whole claim is that a
+ // chat-only video is NOT downloaded.
+ if (
+ has("--skip-download") &&
+ has("--write-subs") &&
+ has("--no-write-auto-subs") &&
+ arg("--sub-langs") === "live_chat" &&
+ !has("--flat-playlist")
+ ) {
+ const url = lastNonFlag();
+ const id = urlIdYouTube(url ?? "");
+ if (!id) {
+ process.stderr.write(`[fake-ytdlp] chat-only mode missing URL\n`);
+ process.exit(2);
+ }
+ const videoDir = path.join("data", id);
+ await ensureDir(videoDir);
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `live-chat-only:${url} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
+ // `nochat` is the stream whose chat replay is off: yt-dlp exits 0 and
+ // writes nothing, which is the case the downloader has to notice (it keeps
+ // no directory for it).
+ if (!(url ?? "").toLowerCase().includes("nochat")) {
+ await writeFile(
+ path.join(videoDir, "transcript.live_chat.json"),
+ '{"clientId":"fake","action":{"addChatItemAction":{}}}\n',
+ );
+ }
+ process.stdout.write(`[fake-ytdlp] live chat only ${id}\n`);
+ return;
+ }
+
// Managed YouTube-handling per-URL invocation: --skip-download with
// --write-auto-subs, no -a. The real download reuses the prefetched
// metadata via --load-info-json (no positional URL) — derive the id from
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -4207,6 +4207,56 @@ rejects.
intact: `runMetadataScanAction` checks the platform cooldown and no gate at
all, so the operator can still scan while downloads are paused — which is
precisely when they want to decide what the lane should fetch on resume.
+- **`downloadFilter.rejectedLivestreams: "skip" | "chat-only"`, absent = "skip"**
+ (`lib/channelConfig.ts`), and the sanitizer stores it only alongside a real
+ filter and only when it is not the default. A channel without the field
+ behaves byte-for-byte as it did.
+- **"chat-only" IS A REJECTION with an instruction attached, not a fourth way to
+ pass.** `DownloadFilterVerdict` gains it; `classifyAgainstFilter` returns it
+ from ONE place (`rejection()`), so an `exclude`-matched livestream and an
+ unmatched one cannot get different answers. **`titleFilterRejects` still
+ answers TRUE for it** — that is what keeps `settledIdsFrom`, `undownloadedIds`
+ and the sync walk correct with no other change. `titleFilterWantsChat` is the
+ next question, asked only by the downloader and the snapshot.
+ `chatOnlyIdsFrom` (metadataScanStore) is the derived STRICT SUBSET of the
+ settled set.
+- **The download path runs ONE extra yt-dlp pass** (`fetchLiveChatOnly`):
+ `--skip-download --write-subs --no-write-auto-subs --sub-langs live_chat`, with
+ the refusals AFTER `channelConfigArgs` and `-o` after those — yt-dlp takes the
+ LAST occurrence, so a channel's own `--write-thumbnail` cannot turn this into a
+ partial download. `--no-write-info-json`, NOT the absence of
+ `--write-info-json`: the prefetch's `metadata.info.json` must SURVIVE, or
+ `buildIndex.ts:309-315` never sees the video and the chat is published nowhere.
+ Then `normalizeLiveChat` writes `live_chat.cues.json` on the spot. No archive
+ line — an archive id means "downloaded" to the sync walk and to
+ `verifyTranscripts`.
+- **A chat pass that fetched nothing keeps no directory.** A stream with chat
+ replay off makes yt-dlp exit 0 and write nothing; `hasLiveChatOnDisk` is what
+ sends that dir to `discardPrefetchDir` like any other rejection's.
+- **`DownloadOutcomeStatus` gains `"chat-only"`** and `DownloadAttemptKind` gains
+ `"live-chat-only"`. The batch counts it as a SKIP (`okCount` means videos
+ downloaded); the runner counts it as a SUCCESS (the unit did the work it was
+ picked for, and the platform's cooldown clears).
+- **Two buckets, and neither is `skippedByTitleFilter`.** `chatOnly` = the
+ verdict plus `transcript.live_chat.json` on disk — a corpus MEMBER, counted in
+ `totals.videos`, never in `settledOnDisk` (which is subtracted from it), and
+ short-circuited out of every bucket that means something is missing.
+ `chatOnlyPending` = the verdict without the file, appended LAST to
+ `DOWNLOAD_BUCKETS` so a chat backlog can never delay a real download. Both are
+ `[]` for every channel without the field, so the fold, the lane work list and
+ the pick order are byte-identical for them.
+- **It cannot ride in `undownloadedIds`**, and that is the reason `chatOnlyPending`
+ is a bucket rather than a derived count: that list means "fetch the media", and
+ fetching the media is the one thing this video must not have done to it.
+- **THE RISK THE TESTS EXIST FOR: a chat-only dir makes "metadata, no
+ transcript" a legitimate shape** — exactly what `buildIndex.ts:309-315` and
+ `deriveChannelSets` (`channelSets.ts:58-67`) read as "ever fetched". Get the
+ bucket rules wrong and filtered livestreams silently leave
+ `missingNeverFetched` and enter the published site as blank pages.
+ `channelSnapshot.test.ts` runs all three states (chat landed / chat pending /
+ mode turned off) against a real `generateChannelSnapshot`, and
+ `editor/e2e/chat-only.spec.ts` builds the index and asserts the published
+ record carries ONLY a `live_chat` track and zero cues.
- **`common/controller/metadataScanJob.ts` is the one definition of the scan as a
JOB** — kind, per-platform queue key, replay spec and the shared
`onPlatformBackoff` wiring. The editor action keeps only what a CLICK is owed