commit 76f67173648eeebc3eb17834b89917ff6059ce21
parent bfcf9918b7e39acd16e77f58d56037d4e4758bc7
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 5 Oct 2026 01:16:26 -0400
Merge fix/scan-excludes-members-only (a metadata-scan members-only or private verdict keeps a video out of the auto-download queue)
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
3 files changed, 88 insertions(+), 1 deletion(-)
diff --git a/common/controller/channelSnapshot.test.ts b/common/controller/channelSnapshot.test.ts
@@ -419,6 +419,8 @@ type FilterFixture = {
// missingNeverFetched rule — "we were told about this, never got it, and it
// is gone" — can be exercised: it is derived from the roster, not the disk.
roster?: string[];
+ // Ids the scan failed on, by class — what it writes for a members-only video.
+ scanErrors?: Record<string, string>;
};
async function filteredChannel(
@@ -476,11 +478,77 @@ async function filteredChannel(
}
await writeFile(
path.join(channelDir, "metadata-scan.json"),
- JSON.stringify({ version: 1, entries, errors: {}, lastRun: null }),
+ JSON.stringify({
+ version: 1,
+ entries,
+ errors: Object.fromEntries(
+ Object.entries(fixture.scanErrors ?? {}).map(([id, cls]) => [
+ id,
+ { class: cls, message: "synthetic", at: new Date().toISOString() },
+ ]),
+ ),
+ lastRun: null,
+ }),
);
return { dir, paths, channelDir };
}
+test("an id the scan read as members-only or private is out of the download queue", async () => {
+ // THE BUG THIS PINS. The scan classified 28 timcast-irl videos members-only;
+ // the download lane then spent a cookie-authed ~30 s attempt on each to learn
+ // the same thing, because only a download's availability.json excluded them.
+ // `deleted` stays queued: the scan infers it from a bare "Video unavailable",
+ // which is also how a soft block reads.
+ const { dir, paths, channelDir } = await filteredChannel({
+ filter: { include: "keep" },
+ listed: [
+ "aaaa0000001",
+ "bbbb0000002",
+ "cccc0000003",
+ "dddd0000004",
+ "eeee0000005",
+ ],
+ scanned: { aaaa0000001: { title: "keep this one" } },
+ scanErrors: {
+ bbbb0000002: "members_only",
+ cccc0000003: "private",
+ dddd0000004: "deleted",
+ eeee0000005: "members_only",
+ },
+ });
+ try {
+ // A cancelled download attempt leaves `error` on disk — inconclusive, so
+ // the scan's verdict still stands.
+ const cancelled = path.join(channelDir, "data", "eeee0000005");
+ await mkdir(cancelled, { recursive: true });
+ await writeFile(
+ path.join(cancelled, "availability.json"),
+ JSON.stringify({
+ checkedAt: "2026-01-01T00:00:00.000Z",
+ availability: "error",
+ history: [
+ {
+ availability: "error",
+ observedAt: "2026-01-01T00:00:00.000Z",
+ source: "download",
+ },
+ ],
+ }),
+ );
+ const snap = await generateChannelSnapshot(paths, "alpha");
+ assert.deepEqual(snap.undownloadedIds, ["aaaa0000001", "dddd0000004"]);
+ assert.deepEqual(snap.excludedFromDownload?.membersOnly, [
+ "bbbb0000002",
+ "eeee0000005",
+ ]);
+ assert.deepEqual(snap.excludedFromDownload?.private, ["cccc0000003"]);
+ // Still offered to the manual cookie run, as a download-observed one is.
+ assert.ok(snap.buckets.needsCookies.includes("bbbb0000002"));
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
+
test("a settled video is in skippedByTitleFilter with NO directory of its own", async () => {
// THE BUCKET IS SOURCED FROM THE SCAN STORE, not from a download-outcome
// sidecar — which is what lets a title-filter rejection delete the prefetch
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -1125,6 +1125,24 @@ export async function generateChannelSnapshot(
// Settled means the MEDIA is not wanted, which is true of these too — what
// this adds is that the video is still a corpus member, as a chat track.
const chatOnlyIds = chatOnlyIdsFrom(metadataScanStore, config);
+ // A scan that already read an id as members-only or private is the
+ // same answer a failed download records in availability.json — without this
+ // the download lane re-learned it, one ~30 s cookie-authed attempt per id
+ // (timcast-irl: 28 members-only ids the scan had classified an hour before).
+ // Derived like settledIds: a later scan that reads the video deletes the
+ // error, and the exclusion lifts with nothing to undo. A conclusive on-disk
+ // availability is newer evidence and wins; `error` (a cancelled or throttled
+ // attempt) is not conclusive. Not the scan's `deleted`: that is
+ // inferred from a bare "Video unavailable", which is also how a soft block
+ // reads, so it is not evidence enough to stop asking.
+ for (const [id, err] of Object.entries(metadataScanStore.errors)) {
+ const onDisk = effectiveById.get(id);
+ if (onDisk && onDisk !== "error") continue;
+ const cls = err.class as Availability;
+ if (cls !== "members_only" && cls !== "private") continue;
+ effectiveById.set(id, cls);
+ excludedById.set(id, cls);
+ }
const noTranscript: string[] = [];
const downloadedNoTranscript: string[] = [];
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,7 @@
# Changelog
## [Unreleased]
+- **Auto-download no longer tries a video the metadata scan already found members-only or private.** The scan records why it could not read a video, but only a failed download used to take a video out of the auto-download queue, so each members-only video the scan had found was still downloaded once: four yt-dlp requests, two with browser cookies, about 30 seconds each. A members-only or private answer from the scan now keeps the video out of the queue and counts it under the channel's members-only or private exclusions, and it stays in **Needs cookies** for a manual cookie run. A scan that reads the video later lifts this. A video the scan saw only as "Video unavailable" is still tried, since YouTube gives that answer when it is throttling too.
- **X post fetches stop on Drain, and an account with no posts is not searched.** Draining a `fetch-posts` job used to do nothing until gallery-dl finished its whole run. Now the timeline fetch stops at the next page boundary (at once when gallery-dl is between pages or waiting out a rate limit), the older-posts walk stops its current window's search at once and never starts the 45–120 second pause between windows, and both keep their resume point: the job ends done, not failed, and the log says "Drained; the next run resumes …". An older-posts walk is refused when nothing is archived and the last timeline fetch finished having read no posts, since it would only repeat empty searches; `"force": true` (`--force` on `archilyzer posts fetch`) walks anyway. A walk with nothing archived that finds nothing ends after two empty three-month windows instead of four, and records why; a walk that has posts keeps the year-of-empty-windows rule. Capture-posts already stopped between posts on Drain.
- **Capturing an X post that is an Article also saves the article.** An X Article (a long-form post) is archived as nothing but its link, and gallery-dl cannot read its body. When `pnpm ops capture-posts` meets a post whose archived text, or whose card on the page, links to an article, it now opens the article in the same X profile, after the same 4–10 second pause, and saves beside the post's capture: `article.json` (the title, author, date and every heading, paragraph, quote, list item, image, link and embedded post in reading order, an embedded post by its URL), `article.md` (the same as readable text), `article.png` (the whole article as shown, cut off at 16,000 pixels tall and marked `trimmed` when longer), `article.html` (the article as X served it, so it can be read again without going back to X) and the article's pictures as `article-img-1.jpg`, `article-img-2.png`, … at full size, fetched through the same browser session. `capture.json` records the article's state, title, block count and every file's size and SHA-256. `"articles": false` leaves articles alone; `"shots": false, "media": false, "articles": true` reads only the articles. An article already captured, deleted or unavailable is not opened again unless `"force": true`; one that failed is tried on the next run. If X asks to log in on the article page, the job stops there as it does for a post. The post viewer's capture panel in the editor shows the article's title with a link to `article.md`.
- **A report video can play a clip that has only sound, and a clip can be a file beside the manifest.** When a clip's source has no picture, `build-video.mjs` plays it under a poster: a card with the clip's channel, title and date, the size of the picture area, with the sound's waveform moving along its foot (`render.audioPoster.waveform: false` keeps it still). The segment matches every other one in size, frame rate and sound, and the header, footer and on-screen deck are drawn over it as over footage. A video's saved sound (`audio.mp3` and the like in its folder) is now a source the build can cut from, after every saved picture: before any download when the clip has no picture to fetch (`"audioOnly": true` on the clip, `"preferLocalAudio": true` in `render`, a podcast or feed record, or a record with no page). A clip that should have a picture is not quietly played from its sound: with `--no-network` or `--skip-fetch`, one whose picture is missing still stops the build, listed as needing a download with a note that its sound is on disk, so `--no-network` still proves every picture is there. Add `--audio-fallback` to play such clips from their sound under the poster instead; they are listed apart (and logged as `audio-fallback`), and clips that have no picture to fetch are listed as playing from audio only rather than refused. A clip may also give `"src"` (a video or audio file) and `"cues"` (its transcript, either a `transcript.cues.json` or a `parakeet-stitch` transcript), both relative to the manifest, instead of a channel and video: it plays the whole file unless `start`/`end` cut inside it, `resolve-windows.mjs` widens it with those cues, it gets no QR unless it has a `citeUrl`, and a path that leaves the manifest's folder (or an absolute one, without `"allowAbsoluteSrc": true` in `render`), a missing file or an unreadable transcript stops the build before anything runs, naming the clip.