commit 0facbdfbfdbffacb9e06b7d59e96af28b0d3740e
parent 4aec83721570422e4309d490df0134c98f2c3ed2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 21 Sep 2026 02:12:02 -0400
e2e: four expectations that were wrong about the fixture, not the code
The features passed; these assertions did not. Each one is a fact about the
fixture the first run taught us:
- `needsauthvid01` is settled by a DOWNLOAD-time rejection. The fake refuses
it only on the metadata-scan invocation, so the scan-first test records an
error for it and settles nothing, while a prefetch reads it fine. That is
the difference the no-scan test exists to exercise, so it belongs in the
expected set rather than out of it.
- "Download videos" prefilters on the DESTINATION file, so a video that
already has transcript.en.vtt is never attempted and the discard is never
asked. The refusal worth pinning through this button is a clip window:
data/<id>/clips/ is media another tool asked this editor for, it is not a
destination, and it is invisible to every video-dir enumerator — so a
discard that walked past it would delete bytes nothing would mention again.
- The auto-queue test asserted the non-matches are never prefetched AFTER the
scan. They usually are not, but the scan only REQUESTS a snapshot regen and
the runner picks up the new work list on a later tick, so that is a race
with the regen rather than a fact about the feature. The claim that matters
— the scan runs before any prefetch, in one batch — is unchanged and passed.
- The live_chat fixture line was not yt-dlp's shape. parseLiveChat reads a
`replayChatItemAction` envelope; a line without one yields zero cues, the
channel's subs dir is then removed as empty, and the assertion died on
ENOENT three steps from the actual cause.
Plus the ops-API calls needed the bearer the route has always required.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
3 files changed, 69 insertions(+), 23 deletions(-)
diff --git a/editor/e2e/auto-queue.spec.ts b/editor/e2e/auto-queue.spec.ts
@@ -886,7 +886,12 @@ test("the download lane scans a filtered channel before it downloads from it", a
expect(scanAt).toBeGreaterThanOrEqual(0);
expect(firstPrefetchAt).toBeGreaterThan(scanAt);
- // And it settled the two non-matches, so they were never prefetched at all.
+ // And it read the whole listing in ONE batch — which is the saving. What is
+ // deliberately NOT asserted here is that the non-matches are never prefetched
+ // afterwards: the scan asks for a snapshot regen and the runner picks up the
+ // new work list on a later tick, so whether a plain video slips through in
+ // that window is a race with the regen, not a fact about the feature. The
+ // settlement itself is pinned in title-filter.spec.ts.
const scan = await readJson<{ entries: Record<string, unknown> }>(
"test-transcripts/channels/filtered/metadata-scan.json",
);
@@ -895,11 +900,6 @@ test("the download lane scans a filtered channel before it downloads from it", a
"plainvid0001",
"plainvid0002",
]);
- for (const id of ["plainvid0001", "plainvid0002"]) {
- expect(invocations).not.toContain(
- `prefetch:https://www.youtube.com/watch?v=${id}`,
- );
- }
// ONE SCAN PER CHANNEL PER RUNNER. The loop keeps ticking every three
// seconds; a second scan would mean the source is re-asked for the whole
diff --git a/editor/e2e/chat-only.spec.ts b/editor/e2e/chat-only.spec.ts
@@ -30,6 +30,12 @@ const CHANNEL = "test-filter";
const ROOT = `test-transcripts/channels/${CHANNEL}`;
const GUEST = ["guestvid0001", "guestvid0002"];
const PLAIN = ["plainvid0001", "plainvid0002"];
+// The fixture's cookie-gated video. Its METADATA SCAN is refused without
+// cookies, but a download-time prefetch reads it fine — so on this path (no
+// scan first) the filter decides it like any other non-match.
+const NEEDS_AUTH = "needsauthvid01";
+const TOKEN = "test-worker-token";
+const AUTH = { authorization: `Bearer ${TOKEN}` };
// The fixture's FINISHED livestream VOD: `livevid` makes the fake report
// live_status "was_live", and its title contains neither "guest" nor "plain",
// so the include pattern rejects it and only the livestream rules decide it.
@@ -168,7 +174,9 @@ test("a chat-only video is in chatOnly, and in no bucket that means work", async
expect(snap.buckets.chatOnly ?? []).toEqual([LIVE]);
// NOT "we decided not to have it" — we decided to have its chat.
- expect(snap.buckets.skippedByTitleFilter ?? []).toEqual([...PLAIN].sort());
+ expect(snap.buckets.skippedByTitleFilter ?? []).toEqual(
+ [...PLAIN, NEEDS_AUTH].sort(),
+ );
// Nothing is going to transcribe a video whose audio we chose not to fetch,
// so it must be out of both transcription-facing buckets AND the download
// queue, or the lanes would fight the filter forever.
@@ -178,9 +186,10 @@ test("a chat-only video is in chatOnly, and in no bucket that means work", async
// Its chat is on disk, so there is no outstanding chat work either.
expect(snap.buckets.chatOnlyPending ?? []).toEqual([]);
// A MEMBER OF THIS CORPUS, not a stub: it has a directory and it is counted.
- // (guestvid0001, guestvid0002 and the chat-only livestream; needsauthvid01
- // leaves a failure directory too, so this is a floor rather than an equality.)
- expect(snap.totals.videos).toBeGreaterThanOrEqual(3);
+ // Exactly three — the two matches plus the chat-only livestream. Every other
+ // rejection took its prefetch directory with it.
+ expect(snap.totals.videos).toBe(GUEST.length + 1);
+ // And it is not a DOWNLOAD: only the two matches are.
expect(snap.totals.downloaded).toBe(GUEST.length);
});
@@ -241,11 +250,27 @@ test("the export publishes a chat-only video as a chat track with no captions",
await resetData("one-youtube-channel-with-data");
// Make it chat-only-shaped: the chat, and no transcript at all.
await rm(resolvePath(`${INDEX_DATA}/transcript.en.vtt`), { force: true });
+ // yt-dlp's live_chat shape: one `replayChatItemAction` envelope per line,
+ // carrying the offset and the renderer parseLiveChat reads.
await writeFile(
resolvePath(`${INDEX_DATA}/transcript.live_chat.json`),
- '{"clientId":"fake","action":{"addChatItemAction":{"item":{"liveChatTextMessageRenderer":' +
- '{"authorName":{"simpleText":"viewer"},"message":{"runs":[{"text":"platypus"}]},' +
- '"timestampUsec":"11000000","videoOffsetTimeMsec":"11000"}}}}}\n',
+ JSON.stringify({
+ replayChatItemAction: {
+ videoOffsetTimeMsec: "11000",
+ actions: [
+ {
+ addChatItemAction: {
+ item: {
+ liveChatTextMessageRenderer: {
+ authorName: { simpleText: "viewer" },
+ message: { runs: [{ text: "platypus in chat" }] },
+ },
+ },
+ },
+ },
+ ],
+ },
+ }) + "\n",
);
await writeSite("testsite", {
channels: [{ slug: INDEX_CHANNEL, groupId: "default" }],
@@ -323,6 +348,7 @@ test("the form round-trips the chat-only tier and the ops API sets it", async ({
// The ops API goes through the same parser, so it gets the same validation.
const ok = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ headers: AUTH,
data: {
slug: CHANNEL,
patch: { downloadFilterRejectedLivestreams: "chat-only" },
@@ -338,6 +364,7 @@ test("the form round-trips the chat-only tier and the ops API sets it", async ({
).toBe("chat-only");
const bad = await request.post(`${baseUrl}/api/ops/channel-config`, {
+ headers: AUTH,
data: { slug: CHANNEL, patch: { downloadFilterRejectedLivestreams: "maybe" } },
});
expect(bad.status()).toBeGreaterThanOrEqual(400);
diff --git a/editor/e2e/title-filter.spec.ts b/editor/e2e/title-filter.spec.ts
@@ -245,30 +245,49 @@ test("a title-filter rejection leaves no video directory behind", async ({
before.generatedAt,
(s) => (s.buckets.skippedByTitleFilter ?? []).length > 0,
);
- expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(SETTLED_BY_GUEST);
+ // FOUR, not three, and the extra one is the point of this path rather than a
+ // fixture quirk: `needsauthvid01` is the video the METADATA SCAN cannot read
+ // (the scan's fake refuses it without cookies, so the scan-first test above
+ // records an ERROR for it and settles nothing). A download-time prefetch
+ // reads it fine, so the filter decides it here — which is exactly the
+ // difference this test exists to exercise.
+ const settledByDownload = [...SETTLED_BY_GUEST, NEEDS_AUTH].sort();
+ expect(snapshot.buckets.skippedByTitleFilter ?? []).toEqual(settledByDownload);
const scan = await readJson<MetadataScan>(`${ROOT}/metadata-scan.json`);
- expect(Object.keys(scan.entries).sort()).toEqual(SETTLED_BY_GUEST);
+ expect(Object.keys(scan.entries).sort()).toEqual(settledByDownload);
});
-test("a rejection never deletes a directory that already holds artifacts", async ({
+test("a rejection never deletes a directory that holds somebody else's data", async ({
page,
}) => {
test.setTimeout(180_000);
- // A video downloaded BEFORE the filter was written is downloaded — a fact,
- // not a preference. The discard refuses anything but its own prefetch, and
- // the retryable outcome sidecar is written for it exactly as before.
+ // The discard refuses anything but its own prefetch. A CLIP WINDOW is the
+ // case worth pinning: `data/<id>/clips/` is media another tool asked this
+ // editor for (umtool's POST /api/media/fetch-window), it is not a
+ // "destination" so the run still attempts the video, and it is invisible to
+ // every video-dir enumerator — so a discard that walked past it would delete
+ // bytes nothing else would ever mention again.
+ //
+ // (A DOWNLOADED video cannot be tested through this button: "Download videos"
+ // prefilters on the destination file, so one that already has a
+ // transcript.en.vtt is never attempted at all. That refusal is the
+ // isVideoDownloaded branch, covered by the unit-level guard.)
await resetData("title-filter-channel");
const kept = PLAIN[0];
- await mkdir(resolvePath(`${ROOT}/data/${kept}`), { recursive: true });
+ await mkdir(resolvePath(`${ROOT}/data/${kept}/clips`), { recursive: true });
await writeFile(
- resolvePath(`${ROOT}/data/${kept}/transcript.en.vtt`),
- "WEBVTT\n\n00:00:00.000 --> 00:00:01.000\nold.\n",
+ resolvePath(`${ROOT}/data/${kept}/clips/0.00-30.00.mp3`),
+ "fake clip audio\n",
);
await generateReport(page, CHANNEL);
await download(page);
- expect(await pathExists(`${ROOT}/data/${kept}/transcript.en.vtt`)).toBe(true);
+ expect(await pathExists(`${ROOT}/data/${kept}/clips/0.00-30.00.mp3`)).toBe(
+ true,
+ );
+ // The directory stayed, so the retryable outcome sidecar is written for it
+ // exactly as it was before this rule existed.
const outcome = await readJson<{ status: string }>(
`${ROOT}/data/${kept}/download-outcome.json`,
);