Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit f08840763fb5fa396f379cb7067941918e3885a0
parent 94ac368b020149163edbc855d46d15ec054f33af
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Sat,  8 Aug 2026 22:52:14 -0400

Cap what the OOM kills, and give the refusal its own number

6 of 10 videos over 6 hours are killed by the kernel mid-diarization on this
16 GB box, burning ~40 minutes each and producing nothing. This is the stopgap
that stops the burning; Part 3 (windowing) is the fix that makes it removable.

settings.diarization.maxAudioHours, default 4 hours, 0 = off. Duration is a
PROXY and the code says so: the blowup is O(n^2) in speech-SEGMENT count and
turn density varies 40x here (33-1364 turns/hour), so a sparse 7h42m video
succeeded while a dense 6h12m one died. Duration is used because it is the only
predictor available for free, from metadata already on disk, BEFORE spending
the 40 minutes to find out.

THE LINE THAT MATTERED was backfillBatch's candidate pull: it handled
not-applicable / present / missing-input explicitly and FELL THROUGH TO
DISPATCHING everything else. A new state without a guard there is not skipped,
it is diarized -- the exact OOM the cap exists to prevent -- and no
TypeScript error would have said so, because nothing in this repo checked
BackfillState exhaustively.

So rather than adding a fourth `if`, the decision is extracted as a pure
`candidateAction(state, {force, allowRedownload})` with a `never` default.
The hazard is now a COMPILE error, and the single most consequential branch in
the file is testable without a registry, a pool or a corpus -- it had no
coverage at all before.

Duration is read ONLY inside the would-be-`missing` branch (~835 videos, not
77,000): VideoFiles deliberately carries no duration, because reusing the
caller's listing is what makes classification free across the corpus, and
countBackfillWork calls state() for every video on every job start. Unknown
duration is NOT deferred -- an unreadable metadata.info.json must not silently
remove work from the list, and the existing fixtures carry no duration.

reachableBackfillWork is DELIBERATELY untouched: not adding `deferred` there is
the entire guard that stops a capped corpus reading as finished. Snapshots
predating the field need `?? 0` at every read site or .toLocaleString() throws;
those are marked as load-bearing rather than defensive. `force` does not
overrule the cap -- force means "redo work that looks done", not "ignore the
limit"; raising the limit is one edit in Settings.

BackfillProbe now carries `settings` rather than the cap riding on `target`:
target is what isDiarizationFresh compares, so a cap in there would mark every
sidecar on disk stale the moment the cap moved.

Verified: common 619/619 (605 + 14 new: 6 cap classifications, 6 dispatch
decisions incl. an exhaustiveness backstop, 1 counts split, 1 sanitize
round-trip); tsc clean in common and editor.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/backfillBatch.test.ts | 108+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++--
Mcommon/controller/backfillBatch.ts | 0
Mcommon/controller/channelSnapshot.ts | 6++++--
Mcommon/lib/backfillKinds.test.ts | 107++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mcommon/lib/backfillKinds.ts | 79+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++------
Mcommon/lib/settings.ts | 30++++++++++++++++++++++++++++++
Mcommon/lib/videoStatus.ts | 28++++++++++++++++++++++++++++
Meditor/CHANGELOG.md | 1+
Meditor/app/channels/[slug]/backfillActions.ts | 3+++
Meditor/app/channels/[slug]/components/stages/BackfillStage.tsx | 18++++++++++++++++++
Meditor/app/channels/[slug]/lib/stageStatus.ts | 12++++++++++++
Meditor/app/channels/[slug]/page.tsx | 4++++
Meditor/app/settings/actions.ts | 1+
Meditor/app/settings/components/SettingsForm.tsx | 8++++++++
14 files changed, 394 insertions(+), 11 deletions(-)

diff --git a/common/controller/backfillBatch.test.ts b/common/controller/backfillBatch.test.ts @@ -1,7 +1,13 @@ import { test } from "node:test"; import assert from "node:assert/strict"; -import { backfillLimit } from "./backfillBatch"; -import { defaultBackfill, sanitizeBackfill } from "../lib/settings"; +import { backfillLimit, candidateAction } from "./backfillBatch"; +import { + defaultBackfill, + defaultDiarization, + sanitizeBackfill, + sanitizeDiarization, +} from "../lib/settings"; +import type { BackfillClassification } from "../lib/backfillKinds"; // Run with: // pnpm --filter yt-dlp-transcript-common exec tsx --test common/controller/backfillBatch.test.ts @@ -84,3 +90,101 @@ test("a stale sweep scope survives sanitization as a list of slugs", () => { }, ); }); + +// --------------------------------------------------------------------------- +// The candidate pull's dispatch decision. +// +// Untested until the duration cap needed it, and it is the single most +// consequential branch in the file: it used to fall through to DISPATCHING any +// state it did not recognize, so "add a state" and "run that state anyway" were +// the same edit. + +const DISPATCH = { force: false, allowRedownload: false }; + +test("only reachable work is dispatched by default", () => { + assert.equal(candidateAction("missing", DISPATCH), "dispatch"); + assert.equal(candidateAction("stale", DISPATCH), "dispatch"); +}); + +test("deferred is NEVER dispatched", () => { + // The whole point of the cap. Without this the classification would be + // computed, ignored, and the six-hour video handed to the engine — 40 minutes + // of CPU and a kernel OOM kill, producing nothing. + assert.equal(candidateAction("deferred", DISPATCH), "deferred"); + assert.equal( + candidateAction("deferred", { force: true, allowRedownload: true }), + "deferred", + ); +}); + +test("force redoes present work but does not overrule the cap", () => { + assert.equal(candidateAction("present", DISPATCH), "fresh"); + assert.equal( + candidateAction("present", { ...DISPATCH, force: true }), + "dispatch", + ); + // `force` means "redo work that looks done", not "ignore the cap". + assert.equal( + candidateAction("deferred", { ...DISPATCH, force: true }), + "deferred", + ); +}); + +test("missing-input is counted unless re-download is armed", () => { + assert.equal(candidateAction("missing-input", DISPATCH), "missing-input"); + assert.equal( + candidateAction("missing-input", { ...DISPATCH, allowRedownload: true }), + "dispatch", + ); +}); + +test("not-applicable is counted as nothing at all", () => { + assert.equal(candidateAction("not-applicable", DISPATCH), "skip"); + assert.equal( + candidateAction("not-applicable", { force: true, allowRedownload: true }), + "skip", + ); +}); + +test("every classification has an explicit decision", () => { + // The list is written out rather than derived so that adding a state to the + // union without deciding what the pull does with it fails HERE as well as at + // the compile step — a runtime backstop for the `never` check, since the + // hazard this replaces was precisely a silent fall-through. + const ALL: BackfillClassification[] = [ + "present", + "stale", + "missing", + "missing-input", + "deferred", + "not-applicable", + ]; + for (const state of ALL) { + assert.doesNotThrow(() => candidateAction(state, DISPATCH), state); + } + assert.throws( + () => candidateAction("invented" as BackfillClassification, DISPATCH), + /unhandled backfill state/, + ); +}); + +test("the duration cap round-trips, and 0 means off", () => { + // 0 is a MEANINGFUL value here, not an empty one — it is how the cap is turned + // off once windowed diarization makes it unnecessary. clampPositiveInt would + // have floored it to 1, i.e. a one-hour cap, which is why this knob does not + // use it. + assert.equal(sanitizeDiarization({ maxAudioHours: 0 }).maxAudioHours, 0); + assert.equal(sanitizeDiarization({ maxAudioHours: 6.5 }).maxAudioHours, 6.5); + // Junk falls back to the default rather than to "no cap" — a typo in a + // hand-edited settings.json must not silently re-arm the OOM. + assert.equal( + sanitizeDiarization({ maxAudioHours: "four" }).maxAudioHours, + defaultDiarization().maxAudioHours, + ); + assert.equal( + sanitizeDiarization({ maxAudioHours: -2 }).maxAudioHours, + defaultDiarization().maxAudioHours, + ); + // The shipped default: every video that OOM-killed this box was over 6h. + assert.equal(defaultDiarization().maxAudioHours, 4); +}); diff --git a/common/controller/backfillBatch.ts b/common/controller/backfillBatch.ts Binary files differ. diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts @@ -381,10 +381,11 @@ export async function generateChannelSnapshot( // per-video probe below is a comparison rather than a derivation, exactly as // digestTarget is above. An empty list is the default (every backfill feature // ships off), and it makes the whole indicator free. - const backfillKinds = laneBackfillKinds(getSettings()); + const backfillSettings = getSettings(); + const backfillKinds = laneBackfillKinds(backfillSettings); const backfillTargets: Record<string, unknown> = {}; for (const kind of backfillKinds) { - backfillTargets[kind.id] = kind.resolveTarget(getSettings()); + backfillTargets[kind.id] = kind.resolveTarget(backfillSettings); } const limit = pLimit(SNAPSHOT_VIDEO_CONCURRENCY); @@ -460,6 +461,7 @@ export async function generateChannelSnapshot( videoId: id, files, target: backfillTargets[kind.id], + settings: backfillSettings, }); } return { diff --git a/common/lib/backfillKinds.test.ts b/common/lib/backfillKinds.test.ts @@ -70,6 +70,9 @@ async function fixture(opts: { container?: boolean; savedPointer?: boolean; sidecar?: string; + // Written into metadata.info.json. Absent means NO metadata file at all, + // which is the "duration unknown" case the cap has to get right. + durationSec?: number; }): Promise<{ dir: string; cleanup: () => Promise<void> }> { const root = await mkdtemp(path.join(os.tmpdir(), "backfill-kinds-")); const dir = path.join(root, "vid1"); @@ -95,6 +98,12 @@ async function fixture(opts: { if (opts.sidecar !== undefined) { await writeFile(path.join(dir, DIARIZATION_FILENAME), opts.sidecar); } + if (opts.durationSec !== undefined) { + await writeFile( + path.join(dir, META_FILENAME), + JSON.stringify({ id: "vid1", duration: opts.durationSec }), + ); + } return { dir, cleanup: () => rm(root, { recursive: true, force: true }) }; } @@ -129,6 +138,7 @@ async function classify( videoId: "vid1", files, target: diarization.resolveTarget(settings), + settings, }); } finally { await cleanup(); @@ -262,6 +272,73 @@ test("missing-input: no sidecar and nothing to diarize from", async () => { assert.equal(await classify({ savedPointer: true }), "missing-input"); }); +// --------------------------------------------------------------------------- +// The duration cap. A stopgap for an OOM that kills 6 of 10 videos over 6 hours +// on this box, burning ~40 minutes each and producing nothing. + +test("deferred: over the cap, with the input right there", async () => { + // 5 hours against the 4-hour default. The input EXISTS — that is the whole + // point of a third state: this is not missing-input (nothing to work from) and + // not missing (work to do); it is work deliberately not attempted. + assert.equal( + await classify({ audio: true, durationSec: 5 * 3600 }), + "deferred", + ); +}); + +test("under the cap is ordinary missing work", async () => { + assert.equal( + await classify({ audio: true, durationSec: 3 * 3600 }), + "missing", + ); + // Exactly at the cap is not over it. + assert.equal( + await classify({ audio: true, durationSec: 4 * 3600 }), + "missing", + ); +}); + +test("unknown duration is NOT deferred", async () => { + // No metadata.info.json at all. Deferring here would quietly remove a video + // from the work list on the strength of a file that could not be read, and it + // would break every fixture in this file that predates the cap. + assert.equal(await classify({ audio: true }), "missing"); + // Present but useless — the same answer, for the same reason. + assert.equal(await classify({ audio: true, durationSec: 0 }), "missing"); +}); + +test("maxAudioHours 0 turns the cap off", async () => { + // Where this setting goes once windowed diarization lands. + assert.equal( + await classify( + { audio: true, durationSec: 12 * 3600 }, + settingsWithDiarization({ maxAudioHours: 0 }), + ), + "missing", + ); +}); + +test("the cap never overrules a sidecar that is already there", async () => { + // A long video ALREADY diarized stays `present`: the cap decides what to + // attempt, not what counts as done. Otherwise raising the cap would look like + // work appearing and lowering it would look like work being undone. + assert.equal( + await classify({ + audio: true, + durationSec: 12 * 3600, + sidecar: sidecar(CURRENT), + }), + "present", + ); +}); + +test("the cap does not resurrect a video whose input is gone", async () => { + // missing-input is checked FIRST. A 12-hour video with nothing to diarize from + // is unreachable, not deferred — deferred promises "we could do this if you + // raised the cap", and that would be a lie here. + assert.equal(await classify({ durationSec: 12 * 3600 }), "missing-input"); +}); + test("a malformed sidecar reads as absent, never as done", async () => { // The same rule diarization-server.ts's hasDiarization() encodes: a // half-written file must not be what convinces anything the work is captured @@ -398,6 +475,7 @@ async function classifyAttr( videoId: "vid1", files, target: kind.resolveTarget(settings), + settings, }); } finally { await cleanup(); @@ -655,8 +733,35 @@ test("counts keep reachable work and needs-re-acquiring apart", () => { ] as BackfillClassification[]) { addBackfillState(counts, state); } - assert.deepEqual(counts, { missing: 2, stale: 1, missingInput: 3 }); + assert.deepEqual(counts, { + missing: 2, + stale: 1, + missingInput: 3, + deferred: 0, + }); // 3, not 6. Measured on the real corpus the difference is 835 vs 77,105, and // reporting the larger number is what would make every surface useless. assert.equal(reachableBackfillWork(counts), 3); }); + +test("deferred is counted, and is NOT reachable work", async () => { + const counts = emptyBackfillCounts(); + for (const state of [ + "missing", + "deferred", + "deferred", + "stale", + ] as BackfillClassification[]) { + addBackfillState(counts, state); + } + assert.deepEqual(counts, { + missing: 1, + stale: 1, + missingInput: 0, + deferred: 2, + }); + // 2, not 4. This is the assertion that keeps a capped corpus from ever reading + // as finished, and the one that fails if someone "tidies up" by folding + // deferred into the total. + assert.equal(reachableBackfillWork(counts), 2); +}); diff --git a/common/lib/backfillKinds.ts b/common/lib/backfillKinds.ts @@ -76,6 +76,7 @@ import { loadAttribution } from "./attribution-server"; import { findSourceMedia, isVideoTranscribed, + readVideoDurationSec, type VideoFiles, } from "./videoStatus"; import { SAVED_VIDEO_POINTER_FILENAME } from "./savedVideo"; @@ -99,7 +100,20 @@ import type { AttributeOneOutcome } from "../controller/attributeOne"; // missing — does not have it, and the input to produce it is HERE. // missing-input — does not have it, and the input is gone. Reachable only by // re-acquiring the media, which is opt-in and bounded. -export type BackfillState = "present" | "stale" | "missing" | "missing-input"; +// deferred — does not have it, the input is here, and the kind refuses to +// attempt it under the current configuration. Today that is +// only the diarization duration cap. NEVER summed into +// reachable work, so a capped corpus cannot read as finished. +// +// `deferred` follows the house rule BackfillRunOutcome = "skipped" already sets +// on the run side: a deliberate non-action gets its own counter, and is never +// folded into the work total nor reported as a failure. +export type BackfillState = + | "present" + | "stale" + | "missing" + | "missing-input" + | "deferred"; // Videos this backfill has no opinion about (not transcribed, marked // untranscribable). Kept out of BackfillState so it can never be counted. @@ -120,6 +134,13 @@ export type BackfillProbe = { // the channel snapshot's per-video classification free — see VideoFiles.entries. files: VideoFiles; target: unknown; + // The live settings. Passed in rather than read here so a classification stays + // a pure function of what the caller already has, and so a kind can consult a + // knob that must NOT become part of its freshness identity — the diarization + // duration cap is exactly that: `target` is compared by isDiarizationFresh, so + // putting the cap there would mark every sidecar on disk stale the moment the + // cap moved. + settings: SiteSettings; }; export type BackfillRunOptions = { @@ -184,7 +205,7 @@ const diarization: BackfillKind = { // let the counter and the runner disagree about what is stale. resolveTarget: (settings): DiarizationFreshnessTarget => diarizationTarget(settings.diarization), - async state({ videoDir, files, target }) { + async state({ videoDir, files, target, settings }) { // Same eligibility as diarizeAll's transcribedOnly default: the capture lane // exists to pair speaker turns with a transcript, and an untranscribed // video's audio is not at risk from the cleanup sweep yet. @@ -203,9 +224,17 @@ const diarization: BackfillKind = { : "stale"; } } - return (await hasDiarizableInput(videoDir, files)) - ? "missing" - : "missing-input"; + if (!(await hasDiarizableInput(videoDir, files))) return "missing-input"; + // The duration cap, read ONLY here. This is the would-be-`missing` branch, + // which is ~835 videos corpus-wide rather than 77,000, and that gating is + // not optional: countBackfillWork calls state() for every video on every job + // start, and readVideoDurationSec reads and parses a file. + // + // Unknown duration is NOT deferred — an absent or unparseable + // metadata.info.json must not silently remove a video from the work list. + return (await isOverDiarizationCap(videoDir, settings.diarization)) + ? "deferred" + : "missing"; }, async run(opts) { // diarizeOneVideo re-reads settings when none is passed, which is what we @@ -247,6 +276,32 @@ async function hasDiarizableInput( return (await resolveSavedVideo(videoDir)) !== null; } +// Is this video longer than the diarization duration cap? +// +// THE CAP IS A STOPGAP FOR AN OOM, and this function is where its two honest +// limitations live. First, duration is a PROXY: the memory blowup is O(n^2) in +// speech-SEGMENT count, and turn density varies 40x across this corpus, so a +// sparse 7h42m video is cheaper than a dense 6h12m one. Duration is used anyway +// because it is the only predictor available from metadata already on disk, for +// free, before committing 45 minutes of CPU to find out the hard way. Second, +// duration is the CONTAINER's, so a video whose metadata is missing or lies gets +// the benefit of the doubt. +// +// Unknown duration therefore returns false — not deferred. Deferring on an +// unreadable metadata.info.json would quietly delete work from the list on the +// strength of a file that could not be parsed, which is the opposite of what a +// third counter is for. +async function isOverDiarizationCap( + videoDir: string, + diarization: SiteSettings["diarization"], +): Promise<boolean> { + const capHours = diarization.maxAudioHours; + if (!capHours || capHours <= 0) return false; // cap off + const seconds = await readVideoDurationSec(videoDir); + if (seconds === null) return false; + return seconds > capHours * 3600; +} + // --------------------------------------------------------------------------- // Attribution — the second and third entries, and the ones that make this a // registry rather than a wrapper around diarization. @@ -471,10 +526,15 @@ export type BackfillCounts = { missing: number; stale: number; missingInput: number; + // Work the kind is refusing to attempt under the current configuration (the + // diarization duration cap). A THIRD number, alongside the other two that are + // never summed. Snapshots written before this field existed do not carry it, + // so every read site needs `?? 0` — `.toLocaleString()` on undefined throws. + deferred: number; }; export function emptyBackfillCounts(): BackfillCounts { - return { missing: 0, stale: 0, missingInput: 0 }; + return { missing: 0, stale: 0, missingInput: 0, deferred: 0 }; } // Fold one classification into a counts record. Central so no surface invents @@ -487,10 +547,17 @@ export function addBackfillState( if (state === "missing") counts.missing++; else if (state === "stale") counts.stale++; else if (state === "missing-input") counts.missingInput++; + else if (state === "deferred") counts.deferred++; } // What the lane can act on WITHOUT re-acquiring media. The number every "how // much is left?" surface should lead with. +// +// DELIBERATELY UNCHANGED by the addition of `deferred`. This function is the +// guard: adding a fourth state to the union raises no TypeScript error anywhere +// (there is no exhaustiveness check over BackfillState in this repo), so the only +// thing keeping capped videos out of the work total is that they are not added +// here. If a future state belongs in the total, it goes in on purpose. export function reachableBackfillWork(counts: BackfillCounts): number { return counts.missing + counts.stale; } diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -363,6 +363,24 @@ export type DiarizationSettings = { // default: diarization is CPU-bound and competes with GPU feeding and the // digest sweep for the same 8 threads. concurrency: number; + // Videos longer than this are DEFERRED rather than diarized: reported as a + // third number that is never summed into reachable work, so a capped corpus + // can never read as finished. + // + // THIS IS A STOPGAP AND IT IS NOT THE FIX. sherpa-onnx's clustering holds a + // pairwise distance matrix over speech-segment embeddings — O(n^2) in SEGMENT + // count — and speaker-turn density varies 40x across this corpus (33-1364 + // turns/hour), so duration does not actually predict the blowup: a sparse + // 7h42m video completed while a dense 6h12m one was OOM-killed. Duration is + // merely the only predictor available for free, from metadata already on disk, + // BEFORE spending 45 minutes to find out. n^2 at 30k segments is 6.7 GiB and + // at 40k is 11.9 GiB, which brackets the 10.6 GB and 9.6 GB peaks measured on + // this 16 GB box. + // + // 0 disables the cap. That is where this goes once windowed diarization lands: + // windowing divides per-window n by the window count, so the matrix falls by + // its square, and the cap stops being needed rather than being tuned. + maxAudioHours: number; }; // Configuration for the derived-corpus digest layer. Local-first by decision: @@ -1083,6 +1101,10 @@ export function defaultDiarization(): DiarizationSettings { segModel: "", embModel: "", concurrency: 1, + // 4 hours. Every video that OOM-killed this box was over 6h; the shortest + // was 6h12m. 4 leaves headroom for the density variance duration cannot see. + // Temporary — see DiarizationSettings.maxAudioHours. + maxAudioHours: 4, }; } @@ -1106,6 +1128,14 @@ export function sanitizeDiarization(value: unknown): DiarizationSettings { segModel: str(r.segModel, d.segModel), embModel: str(r.embModel, d.embModel), concurrency: clampPositiveInt(r.concurrency, d.concurrency, 16), + // 0 is meaningful here (cap off), so this cannot use clampPositiveInt. + // Fractional hours are allowed — the knob is a duration, not a count. + maxAudioHours: + typeof r.maxAudioHours === "number" && + Number.isFinite(r.maxAudioHours) && + r.maxAudioHours >= 0 + ? r.maxAudioHours + : d.maxAudioHours, }; } diff --git a/common/lib/videoStatus.ts b/common/lib/videoStatus.ts @@ -224,3 +224,31 @@ export function isVideoDownloaded(files: VideoFiles): boolean { files.audioFiles.length > 0 ); } + +// A video's runtime in seconds from metadata.info.json, or null when there is no +// metadata, it will not parse, or it carries no usable duration. +// +// DELIBERATELY NOT PART OF VideoFiles. That type carries no duration on purpose +// — reusing the caller's readdir listing is what makes classification free +// across 77,000 videos — and this reads and JSON-parses a file, which is several +// orders of magnitude more expensive. Call it only after a cheap check has +// already narrowed the population; the diarization cap calls it inside the +// would-be-`missing` branch, where there are ~835 videos rather than 77,000. +// +// NULL IS "UNKNOWN", NEVER "ZERO". Every caller has to decide what to do without +// an answer, and for the cap that decision is "do not defer" — the same +// cheap-first rule verifyBeforeClean follows, where anything unresolved is left +// alone rather than acted on. +export async function readVideoDurationSec( + videoDir: string, +): Promise<number | null> { + try { + const raw = await readFile(path.join(videoDir, META_FILENAME), "utf8"); + const meta = JSON.parse(raw) as { duration?: unknown }; + const d = meta?.duration; + if (typeof d !== "number" || !Number.isFinite(d) || d <= 0) return null; + return d; + } catch { + return null; + } +} diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md @@ -1,6 +1,7 @@ # Changelog ## [Unreleased] +- **Very long recordings are now set aside instead of being attempted and killed.** Speaker diarization on a multi-hour stream can exhaust this machine's memory and be killed by the kernel partway through — six of ten videos over six hours died that way, each after about forty minutes of work, producing nothing. The cause is not length as such: the engine holds a comparison of every detected speaker turn against every other, so its memory grows with the *square* of how many turns there are, and turn density varies fortyfold across this corpus (a sparse seven-hour reaction video finished; a dense six-hour stream did not). Length is simply the only predictor available for free, from information already on disk, before spending the forty minutes. So there is now a **Max audio hours** setting (default **4**, `0` turns it off), and a video over it is reported as **deferred**: a third number, shown on the channel's Backfill card and in its summary line, and **never added to the work still to do**. That last part is the point — a limit that quietly shrank the backlog would let a capped corpus report itself as finished. Deferred is not a failure and not "no input"; it is work deliberately not attempted, and raising the limit is all it takes to ask for it. **This is a stopgap and it is meant to be removed:** windowing the engine's work is the real fix, and when that lands the default goes to 0. Note that a backfill already running when this shipped keeps the old behaviour until it is stopped and started again. - **A file called `audio.en-orig.vtt` is a subtitle, and the cleanup sweep used to think it was audio.** Every place in the app that asked "is this a media file?" answered by looking at the *start* of the name — anything beginning with `audio.` counted — and then subtracting a list of known exceptions. A list of exceptions is only ever as good as the last thing someone remembered to add to it, and it had fallen behind: a leaked subtitle track (`audio.en-orig.vtt`) matched nothing on the list, and a live-chat download fragment (`audio.live_chat.json.part-Frag114`) slipped past the `.part` rule because it doesn't *end* in `.part`. Since the sweep that deletes finished audio shares that same question, it could delete a subtitle believing it was audio — a small amount of real, unrecoverable data loss, in a sweep with no dry run and no trash. The same gap on the video-container side was worse: yt-dlp's own scratch file `source-media.temp.mp4` read as a finished video, and that is the one file the app *writes* from — it would extract audio from the corrupt scratch copy and then move it into the saved-video store as the permanent archive copy, orphaning the real container. **The rule is now the shape of the name rather than a list of exceptions**: a finished media file is `audio.<ext>` or `source-media.<ext>` with exactly one extension and nothing between, and `<ext>` has to be a format we actually recognize. That one rule rejects all four junk files, and it also rejects something an extension check alone would have let back in — the app's own half-written transcode output, `audio.tmp-12345.mp3`, which has a perfectly ordinary audio extension and whose deletion would race the transcode still writing it. Three knock-on fixes ride along: the "remove extra audio formats" sweep and the transcode-retry pass each carried their own private, staler copy of the old exception list (so the former could delete that subtitle too, and the latter could hand ffmpeg a `.vtt` file as "source audio"), and the audio-integrity checker could mistake a live-chat fragment for a resumable download. All four now ask the same question in the same place. **Everything moves in the safe direction:** strictly fewer files are ever deleted, videos whose only "audio" was one of these fakes now report honestly as having no audio instead of as a failed transcription, and such a video is no longer mistaken for one that has already been downloaded — so it gets fetched. See `common/lib/mediaFiles.ts` (new, with tests seeded from the real files this corpus holds) and `common/lib/videoStatus.ts`. - **Speaker-attribution turns that land outside their own chunk are now counted instead of quietly dropped.** The text-only attribution lane asks the model about one slice of a transcript at a time and discards any answer whose timestamp points somewhere outside that slice — correctly, since a mark in another slice's territory would fight that slice's own answer. But it discarded them silently, with no count anywhere, and a validation run measured the drop rate at **30% — 520 of 1,705 turns**. That is the difference between "this lane is working" and "this lane is throwing away one answer in three", and nothing on disk or in the log could tell you which. The discard is now recorded on the record itself (one entry per chunk, carrying the count, so a high rate doesn't bury every other warning) and summarized in the job log as a rate. Behaviour is otherwise unchanged and the lane remains switched off by default; this is instrumentation, not a change of policy. - **The backfill lane can now be paused from the dashboard, next to the work it pauses.** The hold itself is not new — the lane has always checked, every time it goes to pick up the next video, whether it is still switched on, and parked itself without ending if it wasn't. What was missing was anywhere sensible to flip it: the only switch lived on the Settings page, which is a strange place to look for a control over a job you are watching run. **Pause now sits beside Start Backfill Sweep, and it is a hold rather than a stop** — the job stays alive, keeps its place, and picks up within a few seconds when you resume, with nothing recomputed and nothing lost. It also survives a restart, because it is a remembered preference rather than something applied to a running pool. This matters most for speaker diarization, which is the one backfill that is genuinely heavy: it runs on the processor rather than the graphics card, pins several cores, and takes hours per long video — so on a desktop you are actually sitting at, "not right now" needs to be one click, for reasons the scheduler cannot possibly know about. **Pause and Stop are deliberately different.** Pause keeps the sweep armed and holds it; Stop ends the sweep, letting the video in flight finish first, and you re-arm it later. The button and the Settings checkbox are the same switch under the hood, so they can never disagree about whether the lane is running. diff --git a/editor/app/channels/[slug]/backfillActions.ts b/editor/app/channels/[slug]/backfillActions.ts @@ -52,6 +52,9 @@ export async function backfillChannelAction( onLog( `Backfill ${slug}: ${batch.succeeded} done, ${batch.fresh} already current, ` + `${batch.failed} failed; ${batch.missingInput} still need their media re-acquired` + + (batch.deferred > 0 + ? `; ${batch.deferred} deferred (over the diarization length limit)` + : "") + (batch.reacquired > 0 ? `; ${batch.reacquired} re-acquired, ${batch.reacquireCleaned} cleaned up` : "") + diff --git a/editor/app/channels/[slug]/components/stages/BackfillStage.tsx b/editor/app/channels/[slug]/components/stages/BackfillStage.tsx @@ -27,6 +27,10 @@ export type BackfillKindView = { // Videos whose input is gone. A COUNT only — the id list is corpus-sized and // deliberately not stored in the snapshot (see BackfillSnapshotEntry). missingInput: number; + // Videos this kind refuses to attempt under the current configuration (the + // diarization duration cap). A THIRD number, never added to the other two: + // summing it would let a capped corpus report as finished. + deferred: number; stale: number; }; @@ -55,6 +59,7 @@ export function BackfillStage({ const reachable = kinds.reduce((n, k) => n + k.reachableIds.length, 0); const missingInput = kinds.reduce((n, k) => n + k.missingInput, 0); + const deferred = kinds.reduce((n, k) => n + k.deferred, 0); const allReachableIds = [ ...new Set(kinds.flatMap((k) => k.reachableIds)), ].sort(); @@ -83,6 +88,18 @@ export function BackfillStage({ : " — re-download is off, so this run skips them."} </p> )} + {deferred > 0 && ( + <p + aria-label="backfill deferred" + className="mt-1 text-sm text-muted-foreground" + > + {deferred.toLocaleString()}{" "} + {deferred === 1 ? "video is" : "videos are"} too long to diarize + under the current limit and {deferred === 1 ? "is" : "are"} being + skipped — raise or clear <em>Max audio hours</em> in Settings to + include {deferred === 1 ? "it" : "them"}. + </p> + )} </div> {kinds.length > 1 && @@ -96,6 +113,7 @@ export function BackfillStage({ {k.reachableIds.length} reachable {k.stale > 0 && ` (${k.stale} stale)`} ·{" "} {k.missingInput.toLocaleString()} needing media + {k.deferred > 0 && ` · ${k.deferred.toLocaleString()} deferred`} </p> ))} diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts @@ -390,6 +390,13 @@ export function computeStageStatuses( (n, e) => n + e.missingInput, 0, ); + // Same treatment as missingInput: reported in the summary line, never folded + // into `pending`. `?? 0` because snapshots written before the cap existed have + // no such field. + const backfillDeferred = backfillEntries.reduce( + (n, e) => n + (e.deferred ?? 0), + 0, + ); const backfillParts: string[] = []; if (backfillPending > 0) { backfillParts.push( @@ -405,6 +412,11 @@ export function computeStageStatuses( `${backfillMissingInput.toLocaleString()} needing media re-acquired`, ); } + if (backfillDeferred > 0) { + backfillParts.push( + `${backfillDeferred.toLocaleString()} deferred (too long to diarize)`, + ); + } const backfill: StageStatus = { id: "backfill", title: "Backfill", diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx @@ -425,6 +425,10 @@ export default async function ChannelDetailPage({ label: kind.label, reachableIds: entry?.ids ?? [], missingInput: entry?.missingInput ?? 0, + // ?? 0 is load-bearing, not defensive: snapshots written before the + // cap existed have no `deferred` field, and .toLocaleString() on + // undefined throws in the render path. + deferred: entry?.deferred ?? 0, stale: entry?.stale ?? 0, }; })} diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts @@ -344,6 +344,7 @@ export async function saveSettingsAction( threshold: num("diarizationThreshold", dDiar.threshold), threads: num("diarizationThreads", dDiar.threads), concurrency: num("diarizationConcurrency", dDiar.concurrency), + maxAudioHours: num("diarizationMaxAudioHours", dDiar.maxAudioHours), python: String(formData.get("diarizationPython") ?? "").trim() || dDiar.python, diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -790,6 +790,14 @@ export function SettingsForm({ initial, apps, digestApps }: Props) { hint="Threads per diarize run." /> <Field + label="Max audio hours" + name="diarizationMaxAudioHours" + defaultValue={String(initial.diarization.maxAudioHours)} + type="number" + step="0.5" + hint="Videos longer than this are reported as deferred instead of being diarized, and are never counted as work still to do. This is a stopgap for an out-of-memory crash on very long recordings: the engine's memory use grows with the SQUARE of the number of speaker turns, so a dense six-hour stream can exhaust 16 GB after 40 minutes of work and produce nothing. 0 turns the limit off." + /> + <Field label="Backfill concurrency" name="diarizationConcurrency" defaultValue={String(initial.diarization.concurrency)}