commit c5b8afaf99c09b3ec2068dd032abeec7dbbc51fd
parent 8920be42b8489d4eda66688cb3ad4a5962da61c2
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Thu, 27 Aug 2026 10:28:29 -0400
backfill: re-acquire fetches audio on a subtitle channel and keeps the cues fresh
Two defects in reacquireMediaFor, one mechanism.
With backfill.reach corpus / order newest / allowRedownload true, the sweep
walked eight handling: "youtube" channels between 2026-08-22 and 08-26. Every
video on a subtitle-downloading channel is diarization missing-input by design
(no audio); allowRedownload turns missing-input into a dispatch; reacquireMediaFor
then called downloadOneManaged with the channel's UNMODIFIED config, so yt-dlp
ran --skip-download --write-subs --write-auto-subs, rewrote metadata (plus a VTT
where YouTube served one), landed no audio, logged "nothing usable landed", and
moved on at ~12/min. ~16,000 fetches, zero diarizations.
1. reacquireConfigFor() forces handling: "transcribe" for the one video and is
used for findVideoSourceUrl, downloadOneManaged and resolveCookiePolicy alike.
The channel's stored config is untouched. Same override, same reason, as
autoRunner's replaceAutoSubs unit (autoRunner.ts:996-1010) and
downloadOneManaged's own no-subs fallback (:932-950).
2. Every re-acquire rewrites metadata.info.json — the metadata prefetch runs
before any attempt, and it runs even when the attempt then fails — which
leaves transcript.cues.json stale by mtime. A census with the real
isCuesJsonFresh over all 78,128 transcribed video dirs: 16,081 stale, 75
missing, 1 no-meta; corpus-wide digest `deferred` is 16,156. Stale cues make
the very next lane in the same sweep (attribution-diarized) skip the video as
no-transcript and the digest lane defer it, until an operator runs Normalize.
So the download is now followed by refreshCues() on both the success and the
failure path, the same step transcribeOne takes after a transcription
(transcribeOne.ts:236-243). Cheap — `fresh` when nothing moved — and it
invalidates no digest: isSectionFresh is provenance-keyed (app/model/prompt/
contextHash), not cues-mtime.
Also, the diarization cap before the download. state() returned missing-input
before reading the duration cap, deliberately, because countBackfillWork calls
state() per video per job start. But with re-download armed missing-input IS a
download and diarizeOneVideo does not enforce the cap, so a 10-hour VOD over the
cap would be fetched in full and diarized anyway. The cap is now read in that
branch too, but only when allowRedownload is on. This moves no count on this
corpus: maxAudioHours is 0 (cap off).
The 16,081 are an mtime false positive, not lost content — 230 of 240 sampled
re-fetched VTTs parse to byte-identical cues, and 8,231 of the cases never
touched the transcript at all. One Normalize pass clears them; that is an
operator action, not this commit.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Diffstat:
4 files changed, 295 insertions(+), 10 deletions(-)
diff --git a/common/controller/backfillReacquire.test.ts b/common/controller/backfillReacquire.test.ts
@@ -0,0 +1,138 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import path from "node:path";
+import { mkdtemp, rm, writeFile, utimes } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import { reacquireConfigFor, refreshCues } from "./backfillReacquire";
+import { isCuesJsonFresh } from "./normalizeTranscript";
+import {
+ CUES_JSON_FILENAME,
+ META_FILENAME,
+ VTT_FILENAME,
+} from "../lib/videoStatus";
+import type { ChannelConfig } from "../lib/channelConfig";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/backfillReacquire.test.ts
+//
+// Only the two pure-ish surfaces: reacquireMediaFor itself needs a network and
+// a disk gate, and this repo does not module-mock. The e2e in
+// editor/e2e/backfill.spec.ts covers the whole path.
+
+// ---------------------------------------------------------------------------
+// reacquireConfigFor — the fix for ~16,000 caption re-fetches that landed no
+// audio (eight channels, 2026-08-22 -> 08-26).
+
+const YOUTUBE_CHANNEL: ChannelConfig = {
+ handling: "youtube",
+ name: "The Channel",
+ platform: "youtube",
+ audioFormat: "mp3",
+ url: "https://www.youtube.com/@channel",
+ cookiesFromBrowser: "firefox",
+ cookieMode: "when-required",
+ subLangs: "en",
+ keepLatest: 3,
+};
+
+test("a subtitle channel re-acquires with transcribe handling, everything else intact", () => {
+ const out = reacquireConfigFor(YOUTUBE_CHANNEL);
+ assert.equal(out.handling, "transcribe");
+ // Every OTHER field survives: the override is one key, not a synthetic
+ // config. Cookies especially — a re-acquire on a members-gated channel still
+ // needs them.
+ for (const [key, value] of Object.entries(YOUTUBE_CHANNEL)) {
+ if (key === "handling") continue;
+ assert.deepEqual(
+ (out as Record<string, unknown>)[key],
+ value,
+ `field ${key} was not carried over`,
+ );
+ }
+});
+
+test("the channel's stored config is never mutated", () => {
+ const config: ChannelConfig = { ...YOUTUBE_CHANNEL };
+ const out = reacquireConfigFor(config);
+ assert.notEqual(out, config);
+ // The load-bearing one: readChannelConfig hands back the object the rest of
+ // the app reads, and a mutation here would turn one video's override into a
+ // permanent change of what the channel downloads.
+ assert.equal(config.handling, "youtube");
+});
+
+test("a transcribe channel is passed through unchanged, same object", () => {
+ const config: ChannelConfig = { handling: "transcribe", name: "Other" };
+ // Identity matters: reacquireMediaFor logs the override only when the object
+ // changed, so a copy here would announce an override on every video.
+ assert.equal(reacquireConfigFor(config), config);
+});
+
+// ---------------------------------------------------------------------------
+// refreshCues — the second half of the defect. Every re-acquire rewrites
+// metadata.info.json (the prefetch runs before any attempt), which leaves
+// transcript.cues.json stale by mtime; that is what made 16,081 videos read
+// `deferred` to the digest lane and `no-transcript` to attribution.
+
+const CONFIG: ChannelConfig = { handling: "youtube", name: "The Channel" };
+
+// mtimes set EXPLICITLY, never left to write order — freshness is pure mtime
+// math and two files written in the same millisecond compare equal on
+// filesystems with coarse timestamps. Same discipline as normalizeAll.test.ts.
+async function staleVideoFixture(): Promise<{
+ dir: string;
+ cleanup: () => Promise<void>;
+}> {
+ const dir = await mkdtemp(path.join(tmpdir(), "reacquire-cues-"));
+ await writeFile(
+ path.join(dir, META_FILENAME),
+ JSON.stringify({ id: "vid1", title: "A video", duration: 60 }),
+ );
+ await writeFile(
+ path.join(dir, VTT_FILENAME),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nhello\n",
+ );
+ await writeFile(
+ path.join(dir, CUES_JSON_FILENAME),
+ JSON.stringify({ version: 1, id: "vid1", title: "A video", cues: [] }),
+ );
+ const base = Date.now() / 1000 - 1000;
+ // The exact shape a re-acquire leaves behind: cues.json older than the
+ // metadata the fetch just rewrote.
+ await utimes(path.join(dir, CUES_JSON_FILENAME), base, base);
+ await utimes(path.join(dir, VTT_FILENAME), base + 5, base + 5);
+ await utimes(path.join(dir, META_FILENAME), base + 10, base + 10);
+ return { dir, cleanup: () => rm(dir, { recursive: true, force: true }) };
+}
+
+test("refreshCues makes a re-acquired video's cues fresh again", async () => {
+ const { dir, cleanup } = await staleVideoFixture();
+ try {
+ assert.equal((await isCuesJsonFresh(dir)).fresh, false);
+ await refreshCues(dir, "the-channel", CONFIG, "vid1", () => {});
+ // The property the 16,081 lost: without this the very next lane in the same
+ // sweep defers the video until an operator runs Normalize.
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ } finally {
+ await cleanup();
+ }
+});
+
+test("refreshCues never throws — it runs on the download-failure path too", async () => {
+ const dir = await mkdtemp(path.join(tmpdir(), "reacquire-cues-empty-"));
+ try {
+ const logged: string[] = [];
+ // An empty dir is `skipped: no-metadata`, not an error — but the guarantee
+ // being pinned is that NOTHING escapes: a normalize problem must never
+ // replace the real outcome of the re-acquire.
+ await refreshCues(dir, "the-channel", CONFIG, "vid1", (m) =>
+ logged.push(m),
+ );
+ assert.equal(
+ logged.some((m) => m.includes("Re-normalized")),
+ false,
+ );
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+});
diff --git a/common/controller/backfillReacquire.ts b/common/controller/backfillReacquire.ts
@@ -6,7 +6,7 @@
// below is arranged around one property — A RE-FETCHED FILE DOES NOT SURVIVE THE
// ITEM THAT FETCHED IT.
//
-// Four guards, in the order they fire:
+// Five guards, in the order they fire:
//
// 1. OPT-IN. The caller only reaches here when settings.backfill.allowRedownload
// (or an explicit per-run flag) is set. Default off.
@@ -21,6 +21,19 @@
// fetch created — diffed against a listing taken BEFORE it ran, so nothing
// that was already on disk can ever be deleted by it — and the caller runs
// it whether the backfill succeeded, failed or threw.
+// 5. AUDIO, NOT THE CHANNEL'S DEFAULT. A `handling: "youtube"` channel's own
+// download is --skip-download --write-subs --write-auto-subs: run with the
+// channel's stored config, a re-acquire re-fetches the captions the video
+// already has and lands nothing a diarizer can read. That is measured, not
+// hypothetical — between 2026-08-22 and 08-26 it spent ~16,000 fetches
+// across eight channels for zero diarizations. reacquireConfigFor forces
+// transcribe-handling for the one video; the channel's stored config is
+// untouched.
+//
+// And one thing that is not a guard: EVERY fetch rewrites metadata.info.json
+// (the prefetch runs before any attempt, and it runs even when the attempt then
+// fails), which leaves transcript.cues.json stale by mtime. So the download is
+// followed by a re-normalize on both paths — see refreshCues at its call site.
//
// The one exception to (4) is a video marked do-not-clean. That marker is the
// operator saying "this media is archived, keep it", and it is honoured here for
@@ -37,10 +50,12 @@ import { isPermanentlyGone } from "../lib/availability";
import { resolveEffectiveAvailability } from "../lib/availability-server";
import { isDoNotClean } from "../lib/doNotClean-server";
import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import type { ChannelConfig } from "../lib/channelConfig";
import { isRealAudioFile, isSourceMediaFile } from "../lib/videoStatus";
import { readChannelConfig } from "./channels";
import { findVideoSourceUrl } from "./undownloadedVideos";
import { downloadOneManaged } from "../ytdlp/downloadOneManaged";
+import { normalizeTranscript } from "./normalizeTranscript";
import { resolveDiarizableMedia } from "./diarizeOne";
export type ReacquireOutcome =
@@ -58,6 +73,19 @@ export type ReacquireOutcome =
const NOTHING_TO_CLEAN = async (): Promise<boolean> => false;
+// A backfill wants AUDIO. A `handling: "youtube"` channel's own download is
+// --skip-download --write-subs --write-auto-subs: it would re-fetch the captions
+// the video already has and land nothing a diarizer can read — which is exactly
+// what happened to ~16,000 videos on eight channels, 2026-08-22 -> 08-26. Same
+// override, same reason, as autoRunner's replaceAutoSubs unit and
+// downloadOneManaged's own no-subs fallback: transcribe-handling for this one
+// video, the channel's stored config untouched.
+export function reacquireConfigFor(config: ChannelConfig): ChannelConfig {
+ return config.handling === "transcribe"
+ ? config
+ : { ...config, handling: "transcribe" };
+}
+
export async function reacquireMediaFor(opts: {
paths: Paths;
channelSlug: string;
@@ -93,11 +121,21 @@ export async function reacquireMediaFor(opts: {
log(`Not re-acquiring ${opts.videoId}: channel config unreadable.`);
return { status: "failed", cleanup: NOTHING_TO_CLEAN };
}
+ // Guard (5): the config this re-acquire actually downloads with. Passed to
+ // findVideoSourceUrl as well as to downloadOneManaged, the way autoRunner
+ // passes its override to both. resolveCookiePolicy reads cookie fields only,
+ // so it is unaffected either way — same object, for consistency.
+ const downloadConfig = reacquireConfigFor(config);
+ if (downloadConfig !== config) {
+ log(
+ `Re-acquiring media for ${opts.videoId} — subtitle channel, downloading audio (handling override: transcribe).`,
+ );
+ }
const url = await findVideoSourceUrl(
opts.paths,
opts.channelSlug,
opts.videoId,
- config,
+ downloadConfig,
);
if (!url) {
log(`Not re-acquiring ${opts.videoId}: no resolvable source URL.`);
@@ -112,27 +150,47 @@ export async function reacquireMediaFor(opts: {
);
log(`Re-acquiring media for ${opts.videoId} (${url})…`);
+ let failed = false;
try {
await downloadOneManaged({
channelSlug: opts.channelSlug,
- channelConfig: config,
+ channelConfig: downloadConfig,
paths: opts.paths,
videoUrl: url,
onLog: (s) => log(s.trimEnd()),
signal: opts.signal ?? new AbortController().signal,
- cookiePolicy: resolveCookiePolicy(settings, config),
+ cookiePolicy: resolveCookiePolicy(settings, downloadConfig),
inlineTranscribeOnFallback: false,
globalSkipLiveDownloads: settings.skipLiveDownloads,
// The archive already lists this video — it was downloaded once. A second
// line would just be noise.
appendArchive: false,
// Audio now, container discarded: a backfill wants the smallest thing that
- // can be read, and this is the flag the save-disk path already uses.
+ // can be read, and this is the flag the save-disk path already uses. On a
+ // subtitle channel it is guard (5)'s transcribe override that makes there
+ // BE audio to extract in the first place.
extractImmediately: true,
keepSourceVideoOverride: false,
});
} catch (err) {
+ failed = true;
log(`Re-acquire ${opts.videoId} failed: ${(err as Error).message}`);
+ }
+
+ // The fetch rewrote metadata.info.json, which the cues embed and which
+ // isCuesJsonFresh compares against. Without this the very next lane in the
+ // same sweep (attribution) skips the video as no-transcript and the digest
+ // lane defers it, until an operator runs Normalize. Same step transcribeOne
+ // takes after a transcription, for the same reason. Cheap: `fresh` when
+ // nothing moved. Digest freshness is provenance-keyed, so this invalidates
+ // nothing.
+ //
+ // On the FAILURE path too: the metadata prefetch runs before any attempt, so
+ // a download that threw has already moved the mtime the cues are compared
+ // against. That is the 16,081.
+ await refreshCues(opts.videoDir, opts.channelSlug, config, opts.videoId, log);
+
+ if (failed) {
// Fall through to the cleanup builder anyway: a failed download can still
// have left a partial file, and that is exactly the leak this exists to stop.
return {
@@ -155,6 +213,39 @@ export async function reacquireMediaFor(opts: {
};
}
+// Re-parse the raw transcript into transcript.cues.json after a fetch has moved
+// metadata.info.json under it.
+//
+// NEVER RETHROWS: this runs on the failure path as well, and a normalize problem
+// must not replace the real outcome of the re-acquire. It is also deliberately
+// NOT part of cleanup(): cues.json is a record, not media, and buildCleanup
+// never touches non-media files.
+//
+// Exported for its unit only — reacquireMediaFor's own path needs a network.
+export async function refreshCues(
+ videoDir: string,
+ channelSlug: string,
+ config: ChannelConfig,
+ videoId: string,
+ log: (msg: string) => void,
+): Promise<void> {
+ try {
+ const outcome = await normalizeTranscript({
+ videoDir,
+ channelSlug,
+ configName: config.name,
+ log,
+ });
+ if (outcome.status === "wrote") {
+ log(`Re-normalized transcript.cues.json for ${videoId}.`);
+ }
+ } catch (err) {
+ log(
+ `Warning: could not re-normalize transcript.cues.json for ${videoId}: ${(err as Error).message}`,
+ );
+ }
+}
+
// Remove exactly the media this fetch added. Returns true when it removed
// something, false when it deliberately kept it (do-not-clean) or there was
// nothing new.
diff --git a/common/lib/operations.test.ts b/common/lib/operations.test.ts
@@ -496,6 +496,49 @@ test("the cap does not resurrect a video whose input is gone", async () => {
// missing-input is checked FIRST. A 12-hour video with nothing to diarize from
// is unreachable, not deferred — deferred promises "we could do this if you
// raised the cap", and that would be a lie here.
+ //
+ // With re-download ARMED it stops being a lie, and the answer changes; the
+ // three cases below are that branch.
+ assert.equal(
+ await classify({ durationSec: 12 * 3600 }, settingsWithCap(4)),
+ "missing-input",
+ );
+});
+
+// The cap in the missing-input branch. `missing-input` is not inert once
+// settings.backfill.allowRedownload is on: it DISPATCHES A DOWNLOAD, and
+// diarizeOneVideo does not enforce the cap itself — so without this a 10-hour
+// VOD over the cap would be fetched in full and then diarized anyway. One
+// metadata read is nothing next to a download.
+function settingsWithRedownload(hours: number): SiteSettings {
+ const s = settingsWithCap(hours);
+ return { ...s, backfill: { ...s.backfill, allowRedownload: true } };
+}
+
+test("re-download armed: over the cap is deferred BEFORE a download is spent", async () => {
+ assert.equal(
+ await classify({ durationSec: 12 * 3600 }, settingsWithRedownload(4)),
+ "deferred",
+ );
+});
+
+test("re-download armed: under the cap still asks for its media", async () => {
+ assert.equal(
+ await classify({ durationSec: 3 * 3600 }, settingsWithRedownload(4)),
+ "missing-input",
+ );
+ // Unknown duration gets the benefit of the doubt here too — the same rule as
+ // the `missing` branch, and for the same reason.
+ assert.equal(
+ await classify({}, settingsWithRedownload(4)),
+ "missing-input",
+ );
+});
+
+test("the cap read stays off the hot path when re-download is off", async () => {
+ // countBackfillWork calls state() for every video on every job start, so the
+ // metadata read is bought only by the branch that can spend a download. With
+ // the flag off the answer is missing-input whatever the duration says.
assert.equal(
await classify({ durationSec: 12 * 3600 }, settingsWithCap(4)),
"missing-input",
diff --git a/common/lib/operations.ts b/common/lib/operations.ts
@@ -676,11 +676,24 @@ const diarization: Operation = {
: "stale";
}
}
- if (!(await hasDiarizableInput(videoDir, files))) return "missing-input";
- // The duration cap, read ONLY here. This is the would-be-`missing` branch,
- // which is ~835 videos corpus-wide rather than 77,000, and that gating is
- // not optional: countBackfillWork calls state() for every video on every job
- // start, and readVideoDurationSec reads and parses a file.
+ if (!(await hasDiarizableInput(videoDir, files))) {
+ // Read the cap here ONLY when this branch can dispatch a download: with
+ // re-download armed, missing-input is a fetch, and diarizeOneVideo does not
+ // enforce the cap. Otherwise stay cheap — this runs per video per job start.
+ if (
+ settings.backfill.allowRedownload &&
+ (await isOverDiarizationCap(videoDir, settings.diarization))
+ ) {
+ return "deferred";
+ }
+ return "missing-input";
+ }
+ // The duration cap, read in the would-be-`missing` branch, which is ~835
+ // videos corpus-wide rather than 77,000 — and in the missing-input branch
+ // above only when re-download is armed and the branch therefore spends a
+ // download. That gating is not optional: countBackfillWork calls state() for
+ // every video on every job start, and readVideoDurationSec reads and parses
+ // a file.
//
// Unknown duration is NOT deferred — an absent or unparseable
// metadata.info.json must not silently remove a video from the work list.