Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 14c2c7e5bb8785d0974240972a58bbdf256e114c
parent de710a7fc6aee9043f3bd2451e56bda79e06701a
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sun, 20 Sep 2026 03:16:01 -0400

download filter: a scheduled stream is a livestream, or it settles forever

`LIVESTREAM_STATUSES` omitted `is_upcoming`, which was a trap rather than a
nicety. The scan never re-reads an id it already has, so a stream scanned
before it airs keeps that status in the store for good — and an
includeLivestreams channel would settle it permanently and never download the
broadcast it was configured to want.

Accepting it costs nothing: skip-live, the next entry in the registry, still
declines the download, and declines it RETRYABLY, which is the correct state
for a video that will exist tomorrow. The two filters answer two different
questions and the unit test now pins that hand-off.

Not done, deliberately: making entries with a transient status eligible for
re-scan. It would keep the backlog permanently above zero on any channel with a
standing scheduled stream, and the status is only load-bearing for this one
flag.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mcommon/lib/downloadFilters.test.ts | 29++++++++++++++++++++++++++++-
Mcommon/lib/downloadFilters.ts | 17+++++++++++++----
2 files changed, 41 insertions(+), 5 deletions(-)

diff --git a/common/lib/downloadFilters.test.ts b/common/lib/downloadFilters.test.ts @@ -286,8 +286,13 @@ test("every live status that counts, and one that does not", () => { for (const status of ["was_live", "post_live", "is_live"]) { assert.equal(classifyAgainstFilter(f, liveMeta(status)), "livestream", status); } - assert.equal(classifyAgainstFilter(f, liveMeta("is_upcoming")), "rejected"); + // A SCHEDULED stream counts. Leaving it out settled it forever: the scan + // never re-reads an id it already has, so the stored "is_upcoming" would + // outlive the broadcast and the channel would never download the stream it + // was configured to want. skip-live still declines the download, retryably. + assert.equal(classifyAgainstFilter(f, liveMeta("is_upcoming")), "livestream"); assert.equal(classifyAgainstFilter(f, liveMeta("")), "rejected"); + assert.equal(classifyAgainstFilter(f, liveMeta("not_live")), "rejected"); // yt-dlp's raw spelling is read as well as the store's. assert.equal( classifyAgainstFilter(f, { @@ -338,3 +343,25 @@ test("the registry names both selectors when it declines", () => { assert.ok(d); assert.match(d.reason, /neither include \/guest\/i nor a livestream/); }); + +test("an upcoming stream is accepted by the filter and still deferred by skip-live", () => { + // The two filters answering the two questions, in the order the registry runs + // them: the title filter says "yes, this channel wants livestreams", and + // skip-live says "not yet — it has not aired". + const d = evaluateDownloadFilters( + ctx({ + channelConfig: { + ...BASE_CONFIG, + downloadFilter: { includeLivestreams: true }, + }, + metadata: meta({ + title: "Tomorrow's stream", + description: "", + live_status: "is_upcoming", + }), + }), + ); + assert.ok(d); + assert.equal(d.filter, "skipLive"); + assert.match(d.reason, /upcoming/); +}); diff --git a/common/lib/downloadFilters.ts b/common/lib/downloadFilters.ts @@ -108,14 +108,23 @@ export function compileDownloadFilter( } // What counts as a livestream for `includeLivestreams`. "was_live"/"post_live" -// are finished VODs (the normal case); "is_live" is included so the FILTER -// accepts a stream in progress — skip-live, the next entry in the registry, -// still declines to download it until it ends. Two different questions, and -// they are answered by two different filters on purpose. +// are finished VODs (the normal case); "is_live" and "is_upcoming" are included +// so the FILTER accepts a stream that has not finished — skip-live, the next +// entry in the registry, still declines to DOWNLOAD it until it ends. Two +// different questions, answered by two different filters on purpose. +// +// "is_upcoming" is here because leaving it out was a trap rather than a +// nicety: a scheduled stream scanned before it airs is stored with that status +// forever (the scan never re-reads an id it already has), so an +// includeLivestreams channel would settle it permanently and never download the +// stream it was configured to want. Accepting it costs nothing — skip-live +// declines the download as RETRYABLE, which is exactly the right state for a +// video that will exist tomorrow. const LIVESTREAM_STATUSES: ReadonlySet<string> = new Set([ "was_live", "post_live", "is_live", + "is_upcoming", ]); // The metadata-scan store spells it `liveStatus`; yt-dlp's raw metadata spells