commit 8a34d340a63f625381acdcc8c58a118bbaa81667
parent c5b8afaf99c09b3ec2068dd032abeec7dbbc51fd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 27 Aug 2026 10:31:40 -0400
backfill: the subtitle-channel re-acquire is pinned end to end
e2e case (11) in backfill.spec.ts seeds a handling: "youtube" channel with
captions and no audio — the shape every video on a subtitle channel has, and the
shape that makes diarization missing-input — arms allowRedownload, and runs the
speaker lane. It asserts all four properties the ~16,000 wasted fetches lacked:
diarization.json lands (so audio was downloaded), the audio does not survive the
item that fetched it, transcript.en.vtt is byte-identical to what was seeded (no
subtitle re-fetch), the last download attempt records handling "transcribe", and
transcript.cues.json exists with an mtime at or after metadata.info.json's.
The fake yt-dlp needed no changes: its prefetch branch rewrites metadata for any
invocation and its -x branch writes audio.mp3, which is the same path
auto-subs-replace.spec.ts already drives for the same override.
And the four places whose prose said "re-acquire" without saying "audio":
settings.ts's allowRedownload doc comment and the operator-facing hints on the
lane settings checkbox and the speakers stage card now say the fetch is audio,
that a subtitle-only channel gets a per-video transcribe override, and that the
channel's stored config is not changed. operations.ts's "backfillReacquire
answers it by fetching AUDIO" is now true as written and is left alone.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Diffstat:
4 files changed, 143 insertions(+), 3 deletions(-)
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -328,6 +328,13 @@ export type BackfillSettings = {
// reachable work, against 45 GB free at 97% full. When on, each re-fetched
// file is removed in a `finally` as soon as the backfill has used it, unless
// the video is marked do-not-clean.
+ //
+ // WHAT IT DOWNLOADS IS AUDIO, on every channel. On a `handling: "youtube"`
+ // channel — which normally only fetches subtitles — the re-acquire applies a
+ // PER-VIDEO transcribe override so yt-dlp lands audio a diarizer can read;
+ // the channel's stored config is not changed. Without that override the fetch
+ // re-downloads the captions the video already has and lands nothing, which is
+ // what happened to ~16,000 videos on eight channels in 2026-08.
allowRedownload: boolean;
};
diff --git a/editor/app/channels/[slug]/components/stages/SpeakersStage.tsx b/editor/app/channels/[slug]/components/stages/SpeakersStage.tsx
@@ -163,7 +163,7 @@ export function SpeakersStage({
{missingInput === 1 ? "video needs" : "videos need"} their
media re-acquired first
{allowRedownload
- ? " — re-download is on, so this run will fetch and then delete it, bounded by the free-disk floor."
+ ? " — re-download is on, so this run will fetch audio (even on a subtitle-only channel) and then delete it, bounded by the free-disk floor."
: " — re-download is off, so this run skips them."}
</>
),
diff --git a/editor/app/operations/components/settings/LaneSettingsForm.tsx b/editor/app/operations/components/settings/LaneSettingsForm.tsx
@@ -109,7 +109,10 @@ export function LaneSettingsForm({
the reachable work, against 45 GB free. With this on, each file is
fetched, used, and <strong>deleted again immediately</strong>{" "}
(unless the video is marked "do not clean"), and nothing
- starts at all when free space is under the disk floor.
+ starts at all when free space is under the disk floor. What it
+ fetches is <strong>audio</strong>, including on channels that
+ normally only download subtitles — those use a per-video
+ transcribe override, and their stored config is not changed.
</span>
</span>
</label>
diff --git a/editor/e2e/backfill.spec.ts b/editor/e2e/backfill.spec.ts
@@ -1,4 +1,4 @@
-import { mkdir, readdir, rm, writeFile } from "node:fs/promises";
+import { mkdir, readdir, readFile, rm, stat, writeFile } from "node:fs/promises";
import { test, expect } from "@playwright/test";
import type { APIRequestContext } from "@playwright/test";
import { baseUrl } from "./baseUrl";
@@ -814,6 +814,136 @@ test("the dashboard pauses and resumes the backfill lane", async ({ page }) => {
await expect.poll(laneEnabled, { timeout: 30_000 }).toBe(true);
});
+// (11) THE RE-DOWNLOAD ON A SUBTITLE CHANNEL: audio, not captions.
+//
+// A `handling: "youtube"` channel downloads with --skip-download --write-subs
+// --write-auto-subs. Run a re-acquire with that config and yt-dlp re-fetches the
+// captions the video already has, rewrites metadata.info.json, and lands nothing
+// a diarizer can read — which is what happened to ~16,000 videos on eight
+// channels between 2026-08-22 and 08-26: zero diarizations, and every touched
+// video left reading `deferred` to the digest lane because the metadata rewrite
+// had made transcript.cues.json stale.
+//
+// Both halves are pinned here: the transcribe override that makes there be audio
+// at all, and the re-normalize that keeps the cues fresh afterwards.
+const YT_SLUG = "test-yt-subs";
+const YT_ROOT = `test-transcripts/channels/${YT_SLUG}`;
+
+function ytRel(videoId: string, file: string): string {
+ return `${YT_ROOT}/data/${videoId}/${file}`;
+}
+
+// A youtube-handling channel with captions and NO audio — the shape every video
+// on a subtitle channel has, and the shape that makes diarization missing-input.
+// audioFormat is pinned so the forced transcribe-handling download lands on
+// audio.mp3, exactly as auto-subs-replace.spec.ts seeds it.
+const SEEDED_VTT =
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nSeeded caption line one.\n\n" +
+ "00:00:05.000 --> 00:00:10.000\nSeeded caption line two.\n";
+
+async function seedSubtitleChannel(videoId: string): Promise<void> {
+ const dir = resolvePath(`${YT_ROOT}/data/${videoId}`);
+ await mkdir(dir, { recursive: true });
+ await writeFile(
+ resolvePath(`${YT_ROOT}/config.json`),
+ JSON.stringify({
+ handling: "youtube",
+ name: "Subtitle test channel",
+ url: "https://www.youtube.com/@subs/videos",
+ audioFormat: "mp3",
+ }) + "\n",
+ );
+ // findVideoSourceUrl resolves the id's URL from the stored playlist.
+ await writeFile(
+ resolvePath(`${YT_ROOT}/playlist`),
+ `https://www.youtube.com/watch?v=${videoId}\n`,
+ );
+ await writeFile(`${dir}/transcript.en.vtt`, SEEDED_VTT);
+ await writeFile(
+ `${dir}/metadata.info.json`,
+ JSON.stringify({
+ id: videoId,
+ title: `Synthetic ${videoId}`,
+ upload_date: "20240101",
+ duration: 60,
+ extractor_key: "Youtube",
+ webpage_url: `https://www.youtube.com/watch?v=${videoId}`,
+ subtitles: {},
+ automatic_captions: { en: [{ ext: "vtt", url: "fake://subs" }] },
+ }) + "\n",
+ );
+ await fetch(`${baseUrl}/api/test/invalidate-cache`).catch(() => {});
+}
+
+test("re-acquiring on a subtitle channel downloads audio and re-normalizes the cues", async ({
+ page,
+}) => {
+ test.setTimeout(SLOW);
+ const VID = "ytsubs000001";
+ await resetData();
+ await writeSettings(
+ backfillSettings({ backfill: { allowRedownload: true } }),
+ );
+ await seedSubtitleChannel(VID);
+ expect(await pathExists(ytRel(VID, "audio.mp3"))).toBe(false);
+
+ await generateReport(page, YT_SLUG);
+ await page.goto(channelStage(YT_SLUG, "speakers"));
+ await page
+ .getByRole("button", { name: "Run speaker work", exact: true })
+ .click();
+
+ // The lane got what it needed — which it can only have done by downloading
+ // AUDIO. With the channel's own config this assertion is what failed silently
+ // 16,000 times as "nothing usable landed".
+ await expect
+ .poll(async () => pathExists(ytRel(VID, "diarization.json")), {
+ timeout: 60_000,
+ })
+ .toBe(true);
+
+ // …and the audio still does not survive the item that fetched it (case (4)'s
+ // property, on a channel that never had audio to begin with).
+ await expect
+ .poll(
+ async () => {
+ const entries = await readdir(
+ resolvePath(`${YT_ROOT}/data/${VID}`),
+ ).catch(() => [] as string[]);
+ return entries.filter(
+ (e) => e.startsWith("audio.") && !e.endsWith(".info.json"),
+ );
+ },
+ { timeout: 30_000 },
+ )
+ .toEqual([]);
+
+ // NO SUBTITLE RE-FETCH. The captions on disk are byte-identical to what was
+ // seeded; the fake yt-dlp's youtube-handling mode writes its own transcript,
+ // so any drift here means the override did not apply.
+ expect(await readFile(resolvePath(ytRel(VID, "transcript.en.vtt")), "utf8"))
+ .toBe(SEEDED_VTT);
+
+ // The override, as the download itself recorded it.
+ const outcome = await readJson<{
+ attempts: { kind: string; handling: string }[];
+ }>(ytRel(VID, "download-outcome.json"));
+ const attempts = outcome.attempts ?? [];
+ expect(attempts.length).toBeGreaterThan(0);
+ expect(attempts[attempts.length - 1]?.handling).toBe("transcribe");
+
+ // THE SECOND HALF: the fetch rewrote metadata.info.json, and the cues were
+ // re-normalized after it. Without this the next lane in the same sweep skips
+ // the video as no-transcript and the digest lane defers it — 16,081 videos
+ // corpus-wide, until an operator runs Normalize.
+ expect(await pathExists(ytRel(VID, "transcript.cues.json"))).toBe(true);
+ const [cuesStat, metaStat] = await Promise.all([
+ stat(resolvePath(ytRel(VID, "transcript.cues.json"))),
+ stat(resolvePath(ytRel(VID, "metadata.info.json"))),
+ ]);
+ expect(cuesStat.mtimeMs).toBeGreaterThanOrEqual(metaStat.mtimeMs);
+});
+
// (N) THE REGRESSION THIS STEP'S DESIGN EXISTS TO PREVENT.
//
// The channel snapshot now carries a work-list entry for EVERY catalog