import { test } from "node:test"; import assert from "node:assert/strict"; import { mkdir, mkdtemp, rm, symlink, writeFile } from "node:fs/promises"; import { tmpdir } from "node:os"; import path from "node:path"; import { attributeAudioHold, diarizationWillNeverClear, digestWorkOf, emptyHeldAudio, foldBackfillEntry, foldBucketLaneEntry, generateChannelSnapshot, } from "./channelSnapshot"; import { ChannelTextUnreadableError } from "../lib/channelMedia"; import type { Paths } from "../lib/paths"; import { emptyOperationCounts, presentOperationWork, reachableOperationWork, type OperationClassification, } from "../lib/operations"; // The snapshot's accounting, tested where it can actually be reached. // // generateChannelSnapshot needs lmdb, an archive reader and a corpus on disk, so // the fold inside it was never exercised by anything. These two functions are // the parts every downstream surface trusts, and both have a failure mode that // looks exactly like success: a work list that disagrees with its own count, and // a coverage number that reads "all done" over a channel nothing has looked at. function fold(states: OperationClassification[]) { return foldBackfillEntry(states.map((state, i) => ({ id: `v${i}`, state }))); } test("ids are EXACTLY the reachable set — the invariant nothing else checks", () => { // A policy leaf hands `ids` out as the work to do while the cards render the // count. The two drifting apart is a progress bar that stalls one short of // complete forever, and this pair has drifted once already (countBackfillWork // silently dropped `blocked`). const entry = fold([ "missing", "stale", "partial", "present", "blocked", "deferred", "missing-input", "not-applicable", ]); assert.equal(entry.ids.length, reachableOperationWork(entry)); assert.deepEqual(entry.ids, ["v0", "v1", "v2"]); // And the three that must never be reachable are still counted, just not there. assert.equal(entry.blocked, 1); assert.equal(entry.deferred, 1); assert.equal(entry.missingInput, 1); }); test("eligible counts every video the operation had an opinion about", () => { // not-applicable is the ONLY exclusion: an untranscribable video is out of // scope, everything else is either done or outstanding. Without this the // `present` half is unrecoverable, because addOperationState discards it. const entry = fold([ "present", "present", "missing", "blocked", "not-applicable", "not-applicable", ]); assert.equal(entry.eligible, 4); // 4 eligible - 1 missing - 1 blocked = 2 present, which is the number no // counter stores directly. assert.equal(digestWorkOf({ backfill: { digest: entry } }).present, 2); }); test("a video no kind classified is skipped, not counted as done", () => { const entry = foldBackfillEntry([ { id: "a", state: "missing" }, { id: "b", state: undefined }, ]); assert.equal(entry.eligible, 1); assert.deepEqual(entry.ids, ["a"]); }); test("ids are sorted, because a snapshot is compared byte-for-byte", () => { const entry = foldBackfillEntry( ["zz", "aa", "mm"].map((id) => ({ id, state: "missing" as const })), ); assert.deepEqual(entry.ids, ["aa", "mm", "zz"]); }); // --------------------------------------------------------------------------- // foldBucketLaneEntry: the same entry for a lane that has no classifier. // // Slice 1.5. Extracted for exactly the reason foldBackfillEntry was: inside // generateChannelSnapshot it could only be reached with lmdb, an archive reader // and a corpus on disk, and its failure mode — a work list that disagrees with // its own count — looks like success. const SOURCE = { buckets: { partialDownloads: ["p1"], downloadedNoTranscript: ["t2", "t1"], failedListed: ["t1", "t3"], noTranscript: ["n1", "n2"], downloadedAutoSubsOnly: ["a1"], }, undownloadedIds: ["zz9", "aa1"], }; test("a bucket lane's entry keeps the reachable invariant too", () => { for (const lane of ["download", "transcription"] as const) { const entry = foldBucketLaneEntry(lane, { ...SOURCE, present: 100 }); assert.equal(entry.ids.length, reachableOperationWork(entry)); assert.equal(entry.stale, 0); assert.equal(entry.partial, 0); assert.equal(entry.blocked, 0); assert.equal(entry.deferred, 0); } }); test("a bucket lane's ids are the default union, unsorted", () => { // partialDownloads before undownloadedIds, and undownloadedIds in PLAYLIST // order — sorting it would silently reorder the auto-download queue. assert.deepEqual( foldBucketLaneEntry("download", { ...SOURCE, present: 100 }).ids, ["p1", "zz9", "aa1"], ); // downloadedNoTranscript then failedListed, `t1` claimed once, and the opt-in // auto-captions bucket nowhere in it. assert.deepEqual( foldBucketLaneEntry("transcription", { ...SOURCE, present: 100 }).ids, ["t2", "t1", "t3"], ); }); test("missingInput is transcription's alone, and present round-trips", () => { // Download's input is the channel listing, which is never missing; a video // with no audio is the TRANSCRIPTION lane's missing input, and the download // lane's ordinary work. const dl = foldBucketLaneEntry("download", { ...SOURCE, present: 40 }); assert.equal(dl.missingInput, 0); assert.equal(presentOperationWork(dl), 40); const tr = foldBucketLaneEntry("transcription", { ...SOURCE, present: 7 }); assert.equal(tr.missingInput, 2); assert.equal(presentOperationWork(tr), 7); }); // --------------------------------------------------------------------------- // digestWorkOf: one reader, one definition. test("digestWorkOf reads the registry entry, the split included", () => { // The registry's classification is the one the RUNNER dispatches from, and it // is now the only one the snapshot carries: the `buckets.noDigest` list that // used to sit beside it had no cues-staleness or transcript gate, so the two // genuinely disagreed (11,777 videos corpus-wide on 2026-08-26). Two // definitions of "digested" is the failure this whole change exists to end. const work = digestWorkOf({ backfill: { digest: { ...emptyOperationCounts(), missing: 2, partial: 1, blocked: 4, deferred: 3, ids: ["a", "b", "c"], eligible: 10, }, }, }); assert.equal(work.reachable, 3); assert.deepEqual(work.ids, ["a", "b", "c"]); assert.equal(work.blocked, 4); assert.equal(work.deferred, 3); assert.equal(work.partial, 1); }); test("digestWorkOf reports UNKNOWN, not zero coverage, when it cannot tell", () => { for (const snapshot of [null, undefined, {}, { backfill: {} }]) { const work = digestWorkOf(snapshot); assert.equal(work.reachable, 0); // The distinction that matters: no work outstanding AND no idea how much is // done. A 0 here would render as "0% digested" on every surface. assert.equal(work.present, null); assert.equal(work.eligible, null); } }); test("digestWorkOf survives an entry written before `partial` and `eligible`", () => { const work = digestWorkOf({ backfill: { digest: { missing: 4, stale: 1, missingInput: 0, deferred: 0, blocked: 0, ids: ["a", "b", "c", "d", "e"], } as never, }, }); assert.equal(work.reachable, 5); assert.equal(work.partial, 0); assert.equal(work.present, null); }); // --- The audio-hold partition ------------------------------------------------ // // The /cleanup sieve says what is holding each byte of audio, and its five // figures are only trustworthy as a PARTITION: a video that is both pinned and // undiarized must be counted once, under the pin, or the page reports a bigger // hold than exists. attributeAudioHold is where that rule lives, and it has to // stay in step with cleanAudioFromTranscribed's discover loop — which is a // cascade of `continue`s, so first-gate-wins is the whole of the semantics. type Vid = Parameters[0] & { bytes: number }; function video(over: Partial = {}): Vid { // A plain reclaimable video: transcribed, unprotected, already diarized. return { bytes: 100, hasWhisper: true, inKeepLatestWindow: false, doNotClean: false, diarizationGuardOn: true, hasDiarization: true, ...over, }; } // The same fold generateChannelSnapshot performs, over a synthetic corpus. function partition(videos: Vid[]) { const held = emptyHeldAudio(); let total = 0; let reclaimable = 0; for (const v of videos) { total += v.bytes; const gate = attributeAudioHold(v); if (gate === "reclaimable") reclaimable += v.bytes; else held[gate] += v.bytes; } return { held, total, reclaimable }; } test("every audio byte lands in exactly one gate", () => { const { held, total, reclaimable } = partition([ video({ bytes: 2800, hasWhisper: false }), video({ bytes: 184, inKeepLatestWindow: true }), video({ bytes: 31, doNotClean: true }), video({ bytes: 1100, hasDiarization: false }), video({ bytes: 2300 }), ]); assert.deepEqual(held, { noTranscript: 2800, keepLatest: 184, doNotClean: 31, awaitingDiarization: 1100, diarizationNeverClears: 0, }); assert.equal(reclaimable, 2300); // THE INVARIANT. This is what stops the sieve's remainder column from // drifting away from the reclaimable figure the page already leads with. const sumHeld = held.noTranscript + held.keepLatest + held.doNotClean + held.awaitingDiarization; assert.equal(sumHeld + reclaimable, total); }); test("first gate wins — a pinned, undiarized video is counted ONCE", () => { // Under the PIN, because the sweep leaves at the first `continue` it hits. // Counting it at both gates would inflate the hold and prescribe a diarize run // that would not release it anyway. const { held } = partition([ video({ bytes: 50, doNotClean: true, hasDiarization: false }), ]); assert.equal(held.doNotClean, 50); assert.equal(held.awaitingDiarization, 0); }); test("gate order matches the sweep: no-transcript beats every protection", () => { const { held } = partition([ video({ bytes: 7, hasWhisper: false, inKeepLatestWindow: true, doNotClean: true, hasDiarization: false, }), ]); assert.equal(held.noTranscript, 7); assert.equal(held.keepLatest, 0); assert.equal(held.doNotClean, 0); assert.equal(held.awaitingDiarization, 0); }); test("the diarization hold exists only while the guard is on", () => { // The sweep reads settings.diarization.enabled and nothing else. With it off, // an undiarized video is reclaimable TODAY — the estimate must say so. assert.equal( attributeAudioHold( video({ diarizationGuardOn: false, hasDiarization: false }), ), "reclaimable", ); assert.equal( attributeAudioHold( video({ diarizationGuardOn: true, hasDiarization: false }), ), "awaitingDiarization", ); }); test("only the reachable diarization states will ever clear a hold", () => { // The one number nothing else in the app can produce. `undefined` is the kind // being disabled outright (models unconfigured): the guard still holds the // audio, and no queued job will ever release it. for (const state of ["missing", "stale", "partial"] as const) { assert.equal(diarizationWillNeverClear(state), false); } for (const state of [ "deferred", "missing-input", "not-applicable", "blocked", undefined, ] as const) { assert.equal(diarizationWillNeverClear(state), true); } }); test("generateChannelSnapshot refuses a LEGACY channel rather than writing an empty snapshot", async () => { // The readdir inside it swallows ENOENT as "no videos", so without the guard a // channel whose text is on an unmounted drive would publish a snapshot saying // every video is undownloaded — and all four lanes read that as work to do. // The throw is what makes the scheduler keep the last good snapshot.json. // Since release 17 only the retired whole-directory layout puts the text on // another drive, and the text guard refuses it whatever the drive is doing. const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-media-")); try { const paths = { channelsDir: path.join(dir, "channels") } as Paths; const channelDir = path.join(paths.channelsDir, "alpha"); await mkdir(channelDir, { recursive: true }); const target = path.join(dir, "platter", "alpha", "data"); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ url: "https://example.com/c", dataDir: target }), ); // A link with no target: an unmounted drive, exactly. await symlink(target, path.join(channelDir, "data")); await assert.rejects( () => generateChannelSnapshot(paths, "alpha"), (err: unknown) => err instanceof ChannelTextUnreadableError && /migrate-tier alpha/.test(err.message), ); } finally { await rm(dir, { recursive: true, force: true }); } }); test("release 17: an unmounted MEDIA drive does not stop the snapshot; its media bytes are unknown", async () => { const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-tier-")); try { const paths = { channelsDir: path.join(dir, "channels") } as Paths; const channelDir = path.join(paths.channelsDir, "alpha"); const videoDir = path.join(channelDir, "data", "v1"); await mkdir(videoDir, { recursive: true }); const target = path.join(dir, "platter", "alpha", "media"); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ handling: "youtube", url: "https://example.com/c", mediaDir: target }), ); await writeFile(path.join(channelDir, "playlist"), ""); const meta = JSON.stringify({ id: "v1" }); await writeFile(path.join(videoDir, "metadata.info.json"), meta); await writeFile(path.join(videoDir, "transcript.json"), "{}"); // The audio is tiered; the media link dangles (an unmounted drive). await symlink(path.join("..", "..", "media", "v1", "audio.mp3"), path.join(videoDir, "audio.mp3")); await symlink(target, path.join(channelDir, "media")); const snap = await generateChannelSnapshot(paths, "alpha"); assert.equal(snap.totals.downloaded, 1, "the link is a name in the listing"); assert.equal(snap.totalMediaBytes, undefined, "unknown, never 0"); assert.equal(snap.totalAudioBytes, undefined); assert.equal(snap.totalTextBytes, meta.length + 2); // The drive back: the link's bytes are counted as media. await mkdir(path.join(target, "v1"), { recursive: true }); await writeFile(path.join(target, "v1", "audio.mp3"), Buffer.alloc(300)); const back = await generateChannelSnapshot(paths, "alpha"); assert.equal(back.totalMediaBytes, 300); assert.equal(back.totalAudioBytes, 300); assert.equal(back.totalTextBytes, meta.length + 2); } finally { await rm(dir, { recursive: true, force: true }); } }); // --------------------------------------------------------------------------- // The filtered channel, end to end through generateChannelSnapshot. // // It needs nothing but a directory: a config, a playlist and the metadata-scan // store. That matters here because the whole point of the two facts below is // that they are derived from the STORE and not from anything on a video's // disk — a claim only a run with no video dirs at all can actually pin. type FilterFixture = { filter?: Record; listed: string[]; scanned?: Record< string, { title?: string; liveStatus?: string; noLiveChat?: boolean } >; // Ids the channel has EVER been seen to contain. Seeded so the // missingNeverFetched rule — "we were told about this, never got it, and it // is gone" — can be exercised: it is derived from the roster, not the disk. roster?: string[]; // Ids the scan failed on, by class — what it writes for a members-only video. scanErrors?: Record; }; async function filteredChannel( fixture: FilterFixture, ): Promise<{ dir: string; paths: Paths; channelDir: string }> { const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-filter-")); const paths = { channelsDir: path.join(dir, "channels"), transcriptsDir: dir, savedVideosDir: path.join(dir, "saved-videos"), } as Paths; const channelDir = path.join(paths.channelsDir, "alpha"); await mkdir(path.join(channelDir, "data"), { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ handling: "youtube", url: "https://www.youtube.com/@alpha/videos", ...(fixture.filter ? { downloadFilter: fixture.filter } : {}), }), ); await writeFile( path.join(channelDir, "playlist"), fixture.listed .map((id) => `https://www.youtube.com/watch?v=${id}`) .join("\n"), ); const entries: Record = {}; for (const [id, e] of Object.entries(fixture.scanned ?? {})) { entries[id] = { title: e.title ?? `Synthetic ${id}`, description: "", uploadDate: "20240101", ...(e.liveStatus ? { liveStatus: e.liveStatus } : {}), ...(e.noLiveChat ? { noLiveChat: true } : {}), scannedAt: "2026-01-01T00:00:00.000Z", }; } if (fixture.roster) { await writeFile( path.join(channelDir, "roster.json"), JSON.stringify({ version: 1, entries: Object.fromEntries( fixture.roster.map((id) => [ id, { url: `https://www.youtube.com/watch?v=${id}`, firstSeenAt: "2026-01-01T00:00:00.000Z", }, ]), ), }), ); } await writeFile( path.join(channelDir, "metadata-scan.json"), JSON.stringify({ version: 1, entries, errors: Object.fromEntries( Object.entries(fixture.scanErrors ?? {}).map(([id, cls]) => [ id, { class: cls, message: "synthetic", at: new Date().toISOString() }, ]), ), lastRun: null, }), ); return { dir, paths, channelDir }; } test("an id the scan read as members-only or private is out of the download queue", async () => { // THE BUG THIS PINS. The scan classified 28 timcast-irl videos members-only; // the download lane then spent a cookie-authed ~30 s attempt on each to learn // the same thing, because only a download's availability.json excluded them. // `deleted` stays queued: the scan infers it from a bare "Video unavailable", // which is also how a soft block reads. const { dir, paths, channelDir } = await filteredChannel({ filter: { include: "keep" }, listed: [ "aaaa0000001", "bbbb0000002", "cccc0000003", "dddd0000004", "eeee0000005", ], scanned: { aaaa0000001: { title: "keep this one" } }, scanErrors: { bbbb0000002: "members_only", cccc0000003: "private", dddd0000004: "deleted", eeee0000005: "members_only", }, }); try { // A cancelled download attempt leaves `error` on disk — inconclusive, so // the scan's verdict still stands. const cancelled = path.join(channelDir, "data", "eeee0000005"); await mkdir(cancelled, { recursive: true }); await writeFile( path.join(cancelled, "availability.json"), JSON.stringify({ checkedAt: "2026-01-01T00:00:00.000Z", availability: "error", history: [ { availability: "error", observedAt: "2026-01-01T00:00:00.000Z", source: "download", }, ], }), ); const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.undownloadedIds, ["aaaa0000001", "dddd0000004"]); assert.deepEqual(snap.excludedFromDownload?.membersOnly, [ "bbbb0000002", "eeee0000005", ]); assert.deepEqual(snap.excludedFromDownload?.private, ["cccc0000003"]); // Still offered to the manual cookie run, as a download-observed one is. assert.ok(snap.buckets.needsCookies.includes("bbbb0000002")); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a settled video is in skippedByTitleFilter with NO directory of its own", async () => { // THE BUCKET IS SOURCED FROM THE SCAN STORE, not from a download-outcome // sidecar — which is what lets a title-filter rejection delete the prefetch // directory it made (ytdlp/downloadOneManaged.ts, discardPrefetchDir) without // the count moving. This fixture has no data/ dirs at all; if the bucket // depended on one, it would be empty here. const { dir, paths } = await filteredChannel({ filter: { include: "keep" }, listed: ["aaaa0000001", "bbbb0000002"], scanned: { aaaa0000001: { title: "keep this one" }, bbbb0000002: { title: "drop this one" }, }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); // And it is out of the download lane's queue, which is the point of it. assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); // A settled id was never a video: it has no directory, so it was never in // totals.videos either. assert.equal(snap.totals.videos, 0); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a chat-only livestream is a corpus member, not a settled stub", async () => { // THE RISK THIS PINS. A chat-only dir makes "metadata, no transcript" a // LEGITIMATE shape — which is exactly what buildIndex and deriveChannelSets // read as "this video was fetched". If the bucket rules are wrong, filtered // livestreams either vanish from the report or reappear as download work // forever. const { dir, paths, channelDir } = await filteredChannel({ filter: { include: "keep", rejectedLivestreams: "chat-only" }, listed: ["aaaa0000001", "bbbb0000002", "cccc0000003"], scanned: { aaaa0000001: { title: "keep this one" }, bbbb0000002: { title: "drop this one" }, cccc0000003: { title: "a long stream", liveStatus: "was_live" }, }, }); try { // The chat has landed for the livestream: metadata + the raw chat, and // nothing else on disk. const videoDir = path.join(channelDir, "data", "cccc0000003"); await mkdir(videoDir, { recursive: true }); await writeFile( path.join(videoDir, "metadata.info.json"), JSON.stringify({ id: "cccc0000003", title: "a long stream" }), ); await writeFile( path.join(videoDir, "transcript.live_chat.json"), '{"action":{}}\n', ); const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]); assert.deepEqual(snap.buckets.chatOnlyPending, []); // NOT "we decided not to have it". assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); // Nothing will transcribe a video whose audio we chose not to fetch, and // nothing will download it. assert.ok(!snap.buckets.noTranscript.includes("cccc0000003")); assert.ok(!snap.buckets.downloadedNoTranscript.includes("cccc0000003")); assert.ok(!snap.undownloadedIds.includes("cccc0000003")); assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); // A video we deliberately have. `settledOnDisk` is subtracted from this and // a chat-only dir must never be in it. assert.equal(snap.totals.videos, 1); assert.equal(snap.totals.downloaded, 0); } finally { await rm(dir, { recursive: true, force: true }); } }); test("before the chat lands it is chatOnlyPending — the download lane's bucket", async () => { const { dir, paths } = await filteredChannel({ filter: { include: "keep", rejectedLivestreams: "chat-only" }, listed: ["aaaa0000001", "cccc0000003"], scanned: { aaaa0000001: { title: "keep this one" }, cccc0000003: { title: "a long stream", liveStatus: "was_live" }, }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.chatOnlyPending, ["cccc0000003"]); assert.deepEqual(snap.buckets.chatOnly, []); // It cannot ride in undownloadedIds: that list means "fetch the media", and // fetching the media is the one thing this video must not have done to it. assert.deepEqual(snap.undownloadedIds, ["aaaa0000001"]); assert.deepEqual(snap.buckets.skippedByTitleFilter, []); // It IS the download lane's work, through the lane work list every lane // reads its dispatch from. assert.ok(snap.backfill?.download?.ids.includes("cccc0000003")); // And LAST in it, behind the real download — the fold walks // DOWNLOAD_BUCKETS in order. const ids = snap.backfill?.download?.ids ?? []; assert.equal(ids[ids.length - 1], "cccc0000003"); } finally { await rm(dir, { recursive: true, force: true }); } }); test("turning the mode off re-decides the channel with nothing to migrate", async () => { // Same store, same disk, no filter field: the livestream is an ordinary // settled rejection again and the chat-only buckets are empty. Nothing about // the verdict was ever stored per video. const { dir, paths } = await filteredChannel({ filter: { include: "keep" }, listed: ["cccc0000003"], scanned: { cccc0000003: { title: "a long stream", liveStatus: "was_live" }, }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.chatOnly, []); assert.deepEqual(snap.buckets.chatOnlyPending, []); assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a settled video that left the listing is NOT reported as never fetched", async () => { // THE LOUDEST ALARM THIS REPORT RAISES, pointed at the wrong video. Both // missingNeverFetched and undownloaded are defined by the ABSENCE of a // directory, and a settled video has none since a rejection stopped leaving // its prefetch dir behind — so without subtracting the settled set, the sweep // tells the operator to attempt a direct-link recovery of a video they // configured us not to fetch. const { dir, paths } = await filteredChannel({ filter: { include: "keep" }, // `bbbb0000002` is in the roster and NOT in the listing any more. listed: ["aaaa0000001"], roster: ["aaaa0000001", "bbbb0000002"], scanned: { aaaa0000001: { title: "keep this one" }, bbbb0000002: { title: "drop this one" }, }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual( (snap.missingNeverFetched ?? []).map((m) => m.id), [], ); // It is still settled, and still reported as such — it left the alarm, not // the report. assert.deepEqual(snap.buckets.skippedByTitleFilter, ["bbbb0000002"]); } finally { await rm(dir, { recursive: true, force: true }); } }); test("an unsettled video that left the listing still raises the alarm", async () => { // The control for the test above: subtracting the settled set must not have // made the category unreachable. const { dir, paths } = await filteredChannel({ filter: { include: "keep" }, listed: [], roster: ["aaaa0000001"], scanned: { aaaa0000001: { title: "keep this one" } }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual( (snap.missingNeverFetched ?? []).map((m) => m.id), ["aaaa0000001"], ); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a stream with no chat replay leaves chatOnlyPending for good", async () => { // A clean pass that found nothing is an ANSWER: chat replay was off, and // asking again tomorrow gets the same nothing. Without the flag the id sits // in chatOnlyPending forever and every runner restart re-prefetches it to run // a pass that will never return anything. const { dir, paths } = await filteredChannel({ filter: { include: "keep", rejectedLivestreams: "chat-only" }, listed: ["cccc0000003"], scanned: { cccc0000003: { title: "a long stream", liveStatus: "was_live", noLiveChat: true, }, }, }); try { const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.chatOnlyPending, []); assert.deepEqual(snap.buckets.chatOnly, []); // It settles as the ordinary filtered-out livestream it is. assert.deepEqual(snap.buckets.skippedByTitleFilter, ["cccc0000003"]); assert.ok(!snap.undownloadedIds.includes("cccc0000003")); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a chat already on disk stays a member after the mode is turned off", async () => { // buildIndex has already published it as a chat track, and flipping the mode // back to "skip" does not un-publish it. Deciding this off `chatOnlyIds` // would make the same directory a settled STUB — and settledOnDisk is // subtracted from totals.videos, so the site would serve a video the report // had stopped counting. const { dir, paths, channelDir } = await filteredChannel({ filter: { include: "keep" }, listed: ["cccc0000003"], scanned: { cccc0000003: { title: "a long stream", liveStatus: "was_live" }, }, }); try { const videoDir = path.join(channelDir, "data", "cccc0000003"); await mkdir(videoDir, { recursive: true }); await writeFile( path.join(videoDir, "metadata.info.json"), JSON.stringify({ id: "cccc0000003", title: "a long stream" }), ); await writeFile( path.join(videoDir, "transcript.live_chat.json"), '{"action":{}}\n', ); const snap = await generateChannelSnapshot(paths, "alpha"); assert.deepEqual(snap.buckets.chatOnly, ["cccc0000003"]); assert.deepEqual(snap.buckets.skippedByTitleFilter, []); assert.equal(snap.totals.videos, 1); // And nothing outstanding: the mode is off, so there is no chat to fetch. assert.deepEqual(snap.buckets.chatOnlyPending, []); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a video dir's clips/ counts into totalClipsBytes, apart from the media and text tiers", async () => { // `clips/` is the one subdirectory a video dir has, and until this it was // counted by nothing: the loop did `if (!st.isFile()) continue` under a // comment saying a video dir is flat, which the clip-window feature made // untrue. The bytes were on the platter and `rsync -a` carried them, so // /storage and every relocation estimate under-reported a channel that had // been walked by a report. // // Release 17: the three are SIBLINGS. totalMediaBytes is the media tier's // (the tierable names — audio, the raw live chat), totalTextBytes the rest of // data//, totalClipsBytes the clip cache, which is never tiered. const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-clips-")); try { const paths = { channelsDir: path.join(dir, "channels") } as Paths; const channelDir = path.join(paths.channelsDir, "alpha"); const videoDir = path.join(channelDir, "data", "v1"); await mkdir(path.join(videoDir, "clips"), { recursive: true }); // A directory that is NOT clips/ still counts nothing — no blanket recursion. await mkdir(path.join(videoDir, "scratch"), { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ url: "https://example.com/c" }), ); await writeFile(path.join(channelDir, "playlist"), ""); await writeFile(path.join(videoDir, "audio.m4a"), Buffer.alloc(100)); await writeFile(path.join(videoDir, "scratch", "junk.bin"), Buffer.alloc(999)); const meta = JSON.stringify({ id: "v1" }); await writeFile(path.join(videoDir, "metadata.info.json"), meta); await writeFile(path.join(videoDir, "clips", "1.00-3.00.mp4"), Buffer.alloc(50)); await writeFile(path.join(videoDir, "clips", "1.00-3.00.json"), Buffer.alloc(7)); const snap = await generateChannelSnapshot(paths, "alpha"); assert.equal(snap.totalClipsBytes, 57); assert.equal(snap.totalMediaBytes, 100); assert.equal(snap.totalTextBytes, meta.length); } finally { await rm(dir, { recursive: true, force: true }); } }); test("a channel with no clips/ reports 0 clip bytes, not undefined", async () => { // A fresh snapshot always carries the field; only a snapshot written BEFORE // the field existed lacks it, which is the case readers render as "unknown". const dir = await mkdtemp(path.join(tmpdir(), "ttb-snap-noclips-")); try { const paths = { channelsDir: path.join(dir, "channels") } as Paths; const channelDir = path.join(paths.channelsDir, "alpha"); const videoDir = path.join(channelDir, "data", "v1"); await mkdir(videoDir, { recursive: true }); await writeFile( path.join(channelDir, "config.json"), JSON.stringify({ url: "https://example.com/c" }), ); await writeFile(path.join(channelDir, "playlist"), ""); await writeFile(path.join(videoDir, "audio.m4a"), Buffer.alloc(100)); const snap = await generateChannelSnapshot(paths, "alpha"); assert.equal(snap.totalClipsBytes, 0); assert.equal(snap.totalMediaBytes, 100); } finally { await rm(dir, { recursive: true, force: true }); } });