commit e9ee8bd7c214999483e735383e7d882ed94f086c
parent bbcce85dd22484fdf1f7cdc262a4851c55e238ba
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 10 Aug 2026 09:43:10 -0400
Give the undigestable transcripts a name, a reason and a button
The digest lane's `deferred` bucket was one classification covering two very
different states, and the code said the wrong one about both.
Measured over all 79,219 video dirs: 1,942 videos have a raw transcript and NO
transcript.cues.json at all, against 47 with one that is genuinely superseded.
The comment and the card copy both described only the superseded case — 2.4% of
what the bucket actually holds — and both claimed it clears itself. It does not.
transcribeOne is the only automatic caller of normalizeTranscript, so a channel
with `handling: "youtube"` fetches subtitles with --skip-download and never
writes the sidecar; 1,683 of the 1,942 are piratesoftware alone, which is why it
showed ~6% digest coverage. It went unnoticed because buildIndex treats
cues.json as a cache and silently re-parses the raw VTT, so the published site
is correct and only the digest lane, which has no such fallback, can see it.
isCuesJsonFresh now returns a `reason` alongside `fresh` (missing / stale /
no-raw / no-meta). Every caller destructured `{ fresh }` and is unaffected, and
no branch's boolean moved. The classification stays `deferred` for both reasons:
one normalize pass fixes both, so the distinction is about copy, not dispatch.
normalizeAllTranscripts takes `channelSlugs`, the same shape diarizeAll already
has (its header says it was modelled on this controller), plus a
normalizeChannelTranscripts wrapper. `normalize-transcripts` becomes a
registered job kind instead of an unregistered string that showed as a raw
machine kind in /jobs. The Digest card renders the button next to the count it
clears, on the channel queue rather than either digest lane — queueing the
unblocker behind the lane it unblocks would be self-defeating on a sweep that
runs for weeks.
BackfillStage stopped hardcoding "too long to diarize under the current limit"
for every kind's deferred videos at once; the wording comes from the kind's new
`deferredHint` and gets one line per kind.
Also settled the open question about omnimirror's 712: they have audio.mp3 and
no transcript of any kind, so they are `blocked` on transcription and correctly
outside this. The plan's per-channel list had conflated untranscribed dirs with
unnormalized ones.
Verified: 674 common tests (664 + 10 new, incl. a fixture guard for the vacuous
pass the first draft had), tsc clean in common and editor, editor build to a
throwaway distDir so the live server's .next was untouched, and 31 e2e in
digest.spec + backfill.spec including a new end-to-end case that a VTT with no
cues.json defers, names itself correctly, and becomes reachable digest work
after the button runs.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
13 files changed, 675 insertions(+), 34 deletions(-)
diff --git a/common/controller/normalizeAll.test.ts b/common/controller/normalizeAll.test.ts
@@ -0,0 +1,288 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import path from "node:path";
+import { mkdtemp, mkdir, rm, writeFile, utimes } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import {
+ isCuesJsonFresh,
+ CUES_FILE_VERSION,
+} from "./normalizeTranscript";
+import { normalizeChannelTranscripts } from "./normalizeAll";
+import {
+ CUES_JSON_FILENAME,
+ META_FILENAME,
+ VTT_FILENAME,
+ WHISPER_FILENAME,
+} from "../lib/videoStatus";
+import type { Paths } from "../lib/paths";
+
+// These cover the two things the digest lane's `deferred` bucket turned out to
+// be, which the code had conflated into one:
+//
+// missing — no transcript.cues.json at all. NOTHING WRITES ONE AUTOMATICALLY
+// for a channel whose subtitles are downloaded rather than
+// transcribed, so this is permanent until the normalize pass runs.
+// 1,942 videos on the live corpus.
+// stale — a cues.json superseded by a rewrite of its inputs. 47 videos.
+//
+// The distinction is about COPY and about whether an operator has anything to
+// do, not about dispatch — both still classify `deferred` — so it lives on
+// isCuesJsonFresh's reason rather than in a new BackfillState.
+
+const CUES_BODY = JSON.stringify({
+ version: CUES_FILE_VERSION,
+ id: "vid1",
+ title: "A video",
+ cues: [{ start: 0, end: 5, text: "hello" }],
+});
+
+// mtimes are set EXPLICITLY, never left to write order: freshness here is pure
+// mtime math, and two files written in the same millisecond compare equal on
+// filesystems with coarse timestamps.
+async function stampAscending(
+ dir: string,
+ order: Array<[file: string, offsetSec: number]>,
+): Promise<void> {
+ const base = Date.now() / 1000 - 1000;
+ for (const [file, offset] of order) {
+ await utimes(path.join(dir, file), base + offset, base + offset);
+ }
+}
+
+async function videoFixture(files: {
+ meta?: boolean;
+ vtt?: boolean;
+ whisper?: boolean;
+ cues?: "fresh" | "stale" | false;
+}): Promise<{ dir: string; cleanup: () => Promise<void> }> {
+ const dir = await mkdtemp(path.join(tmpdir(), "normalize-cues-"));
+ const stamps: Array<[string, number]> = [];
+ if (files.meta !== false) {
+ await writeFile(
+ path.join(dir, META_FILENAME),
+ JSON.stringify({ id: "vid1", title: "A video", duration: 60 }),
+ );
+ stamps.push([META_FILENAME, 10]);
+ }
+ if (files.vtt) {
+ await writeFile(
+ path.join(dir, VTT_FILENAME),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nhello\n",
+ );
+ stamps.push([VTT_FILENAME, 10]);
+ }
+ if (files.whisper) {
+ await writeFile(
+ path.join(dir, WHISPER_FILENAME),
+ JSON.stringify({
+ transcription: [
+ { offsets: { from: 0, to: 5000 }, text: "hello" },
+ ],
+ }),
+ );
+ stamps.push([WHISPER_FILENAME, 10]);
+ }
+ if (files.cues) {
+ await writeFile(path.join(dir, CUES_JSON_FILENAME), CUES_BODY);
+ // stale = OLDER than its inputs, which is the whole definition.
+ stamps.push([CUES_JSON_FILENAME, files.cues === "stale" ? 5 : 20]);
+ }
+ await stampAscending(dir, stamps);
+ return { dir, cleanup: () => rm(dir, { recursive: true, force: true }) };
+}
+
+test("isCuesJsonFresh reports `missing` when there is no cues.json at all", async () => {
+ const { dir, cleanup } = await videoFixture({ vtt: true, cues: false });
+ try {
+ const r = await isCuesJsonFresh(dir);
+ assert.equal(r.fresh, false);
+ // The 97.6% case. Naming it is the point: it is the one an operator can fix
+ // and the one nothing fixes on its own.
+ assert.equal(r.reason, "missing");
+ } finally {
+ await cleanup();
+ }
+});
+
+test("isCuesJsonFresh reports `stale` when cues.json predates its raw transcript", async () => {
+ const { dir, cleanup } = await videoFixture({ vtt: true, cues: "stale" });
+ try {
+ const r = await isCuesJsonFresh(dir);
+ assert.equal(r.fresh, false);
+ assert.equal(r.reason, "stale");
+ } finally {
+ await cleanup();
+ }
+});
+
+test("isCuesJsonFresh reports `no-raw` when nothing was ever transcribed", async () => {
+ // Neither reason an operator can act on: normalizeTranscript would return
+ // `skipped`, and the digest lane counts these as `blocked` on transcription
+ // rather than deferred.
+ const { dir, cleanup } = await videoFixture({ cues: false });
+ try {
+ const r = await isCuesJsonFresh(dir);
+ assert.equal(r.fresh, false);
+ assert.equal(r.reason, "no-raw");
+ } finally {
+ await cleanup();
+ }
+});
+
+test("isCuesJsonFresh reports `no-meta` before it looks at the transcript", async () => {
+ const { dir, cleanup } = await videoFixture({
+ meta: false,
+ vtt: true,
+ cues: "fresh",
+ });
+ try {
+ const r = await isCuesJsonFresh(dir);
+ assert.equal(r.fresh, false);
+ assert.equal(r.reason, "no-meta");
+ } finally {
+ await cleanup();
+ }
+});
+
+test("isCuesJsonFresh still says fresh — and says so with a reason", async () => {
+ const { dir, cleanup } = await videoFixture({ vtt: true, cues: "fresh" });
+ try {
+ const r = await isCuesJsonFresh(dir);
+ assert.equal(r.fresh, true);
+ assert.equal(r.reason, "fresh");
+ } finally {
+ await cleanup();
+ }
+});
+
+// ---------------------------------------------------------------------------
+// The channel-scoped runner
+// ---------------------------------------------------------------------------
+
+async function corpusFixture(): Promise<{
+ paths: Paths;
+ cleanup: () => Promise<void>;
+}> {
+ const root = await mkdtemp(path.join(tmpdir(), "normalize-channel-"));
+ const channelsDir = path.join(root, "channels");
+
+ const video = async (
+ slug: string,
+ id: string,
+ files: { vtt?: boolean; cues?: "fresh" | false },
+ ): Promise<void> => {
+ const dir = path.join(channelsDir, slug, "data", id);
+ await mkdir(dir, { recursive: true });
+ const stamps: Array<[string, number]> = [];
+ await writeFile(
+ path.join(dir, META_FILENAME),
+ JSON.stringify({ id, title: `Video ${id}`, duration: 60 }),
+ );
+ stamps.push([META_FILENAME, 10]);
+ if (files.vtt) {
+ await writeFile(
+ path.join(dir, VTT_FILENAME),
+ "WEBVTT\n\n00:00:00.000 --> 00:00:05.000\nhello\n",
+ );
+ stamps.push([VTT_FILENAME, 10]);
+ }
+ if (files.cues) {
+ await writeFile(path.join(dir, CUES_JSON_FILENAME), CUES_BODY);
+ stamps.push([CUES_JSON_FILENAME, 20]);
+ }
+ await stampAscending(dir, stamps);
+ };
+
+ for (const slug of ["target", "other"]) {
+ await mkdir(path.join(channelsDir, slug), { recursive: true });
+ await writeFile(
+ path.join(channelsDir, slug, "config.json"),
+ JSON.stringify({
+ // REQUIRED — parseChannelConfig returns null without it, and a channel
+ // with no parseable config is silently dropped from the walk. A fixture
+ // missing this makes every assertion below pass vacuously, which is
+ // exactly how the first draft of this file went green.
+ handling: "youtube",
+ name: slug,
+ url: `https://example.com/${slug}`,
+ }),
+ );
+ }
+ // One of each outcome, which is what makes the three counters meaningful.
+ await video("target", "needs-normalizing", { vtt: true, cues: false });
+ await video("target", "already-current", { vtt: true, cues: "fresh" });
+ await video("target", "never-transcribed", {});
+ // The channel that must NOT be touched — the entire reason for the scope.
+ await video("other", "also-needs-normalizing", { vtt: true, cues: false });
+
+ return {
+ paths: { channelsDir } as Paths,
+ cleanup: () => rm(root, { recursive: true, force: true }),
+ };
+}
+
+test("normalizeChannelTranscripts reports wrote/fresh/skipped separately", async () => {
+ const { paths, cleanup } = await corpusFixture();
+ try {
+ const result = await normalizeChannelTranscripts({
+ paths,
+ channelSlug: "target",
+ onLog: () => {},
+ });
+ assert.equal(result.wrote, 1, "the video with a VTT and no cues.json");
+ assert.equal(result.fresh, 1, "the one already current — no rework");
+ // `skipped` is the honest floor, not a failure: normalizing cannot help a
+ // video that was never transcribed. Those are `blocked` on the digest card,
+ // a different number from `deferred`.
+ assert.equal(result.skipped, 1, "the never-transcribed one");
+ assert.equal(result.failed, 0);
+ } finally {
+ await cleanup();
+ }
+});
+
+test("normalizeChannelTranscripts leaves every other channel alone", async () => {
+ const { paths, cleanup } = await corpusFixture();
+ try {
+ const result = await normalizeChannelTranscripts({
+ paths,
+ channelSlug: "target",
+ onLog: () => {},
+ });
+ // GUARD AGAINST A VACUOUS PASS. An unparseable fixture config drops the
+ // channel from the walk entirely, and then "the other channel is untouched"
+ // is true because NOTHING ran. Assert the run did its work first.
+ assert.equal(result.wrote, 1, "the run must actually have done something");
+ // Scope is the whole point: the corpus-wide button walks ~79,000 dirs, and
+ // the digest card's fix must cost one channel.
+ const other = await isCuesJsonFresh(
+ path.join(paths.channelsDir, "other", "data", "also-needs-normalizing"),
+ );
+ assert.equal(other.reason, "missing");
+ } finally {
+ await cleanup();
+ }
+});
+
+test("normalizing turns a `missing` video into a fresh one", async () => {
+ const { paths, cleanup } = await corpusFixture();
+ try {
+ const dir = path.join(
+ paths.channelsDir,
+ "target",
+ "data",
+ "needs-normalizing",
+ );
+ assert.equal((await isCuesJsonFresh(dir)).reason, "missing");
+ await normalizeChannelTranscripts({
+ paths,
+ channelSlug: "target",
+ onLog: () => {},
+ });
+ // The end-to-end claim this whole change rests on: a deferred video becomes
+ // reachable digest work.
+ assert.equal((await isCuesJsonFresh(dir)).fresh, true);
+ } finally {
+ await cleanup();
+ }
+});
diff --git a/common/controller/normalizeAll.ts b/common/controller/normalizeAll.ts
@@ -12,6 +12,16 @@ import type { Paths } from "../lib/paths";
export type NormalizeAllOptions = {
paths: Paths;
+ // Restrict to these channel slugs. Empty/omitted = the whole corpus, which is
+ // what the /build button has always done. Channel scoping exists because the
+ // videos that need this are CONCENTRATED — 1,683 of the 1,942 unnormalized
+ // videos on this corpus are one channel — and clearing one channel should not
+ // mean walking all ~79,000 dirs from a page nobody opens during digest work.
+ //
+ // Same shape as diarizeAll's `channelSlugs`, deliberately: that controller
+ // says in its own header that it was modelled on this one, so this is the
+ // symmetry being completed rather than a new pattern.
+ channelSlugs?: string[];
onLog?: (msg: string) => void;
signal?: AbortSignal;
concurrency?: number;
@@ -29,7 +39,10 @@ export async function normalizeAllTranscripts(
): Promise<NormalizeAllResult> {
const log = opts.onLog ?? ((m: string) => console.log(m));
const limit = pLimit(opts.concurrency ?? 8);
- const channels = await listChannelStatsFromDisk(opts.paths);
+ const wanted = new Set(opts.channelSlugs ?? []);
+ const channels = (await listChannelStatsFromDisk(opts.paths)).filter(
+ (ch) => wanted.size === 0 || wanted.has(ch.slug),
+ );
const result: NormalizeAllResult = {
wrote: 0,
fresh: 0,
@@ -78,3 +91,18 @@ export async function normalizeAllTranscripts(
);
return result;
}
+
+// One channel. The digest lane's `deferred` bucket is exactly this run's work
+// list, so the button that clears it lives on the digest stage card — the count
+// and the fix in the same place.
+//
+// `skipped` here is the honest floor, not a failure: it counts videos with no
+// raw transcript or no metadata, which normalizing cannot help. On this corpus
+// that is 1,626 videos, and they are `blocked` on transcription rather than
+// deferred — a different number on the same card.
+export async function normalizeChannelTranscripts(
+ opts: Omit<NormalizeAllOptions, "channelSlugs"> & { channelSlug: string },
+): Promise<NormalizeAllResult> {
+ const { channelSlug, ...rest } = opts;
+ return normalizeAllTranscripts({ ...rest, channelSlugs: [channelSlug] });
+}
diff --git a/common/controller/normalizeTranscript.ts b/common/controller/normalizeTranscript.ts
@@ -187,12 +187,36 @@ export async function readTranscriptCoverage(
};
}
+// WHY a cues.json is not usable. `fresh` is the answer every caller had before
+// and still gets; `reason` is the answer the DIGEST CARD needs, because the two
+// not-fresh cases have opposite meanings for an operator:
+//
+// missing — there is no transcript.cues.json at all. NOTHING PRODUCES ONE
+// AUTOMATICALLY for these: transcribeOne is the only automatic
+// caller of normalizeTranscript, and a `handling: "youtube"` channel
+// downloads subtitles with --skip-download and never runs it. So
+// this state is permanent until someone runs the normalize pass.
+// Measured on this corpus: 1,942 videos, 1,683 of them piratesoftware.
+// stale — a cues.json exists but the metadata or the raw transcript has been
+// rewritten under it. This IS the "superseded" case the digest lane
+// was written for. Measured: 47 videos, all shondo-vods.
+//
+// Both are fixed by the same normalize pass, which is why they share the
+// `deferred` classification — but only `missing` was ever the whole story, and
+// the copy that claimed the stale cause for both was wrong for 97.6% of them.
+export type CuesFreshReason =
+ | "fresh"
+ | "missing"
+ | "stale"
+ | "no-raw"
+ | "no-meta";
+
// Helper: given a video dir, decide whether transcript.cues.json (if present)
// is at least as new as metadata.info.json and the raw transcript file. Used
// by buildIndex to know whether it can trust cues.json without re-parsing.
export async function isCuesJsonFresh(
videoDir: string,
-): Promise<{ fresh: boolean; cuesPath: string }> {
+): Promise<{ fresh: boolean; reason: CuesFreshReason; cuesPath: string }> {
const cuesPath = path.join(videoDir, CUES_JSON_FILENAME);
const metaPath = path.join(videoDir, META_FILENAME);
// Resolve the actual primary VTT (may be a regional/auto English track like
@@ -206,11 +230,18 @@ export async function isCuesJsonFresh(
mtimeMs(vttPath),
mtimeMs(whisperPath),
]);
- if (cuesMs === null || metaMs === null) return { fresh: false, cuesPath };
- if (cuesMs < metaMs) return { fresh: false, cuesPath };
// Prefer whisper if present (matches pickIndexTranscript priority).
const rawMs = whisperMs ?? vttMs;
- if (rawMs === null) return { fresh: false, cuesPath };
- if (cuesMs < rawMs) return { fresh: false, cuesPath };
- return { fresh: true, cuesPath };
+ // Ordered to match normalizeTranscript's OWN precedence (metadata, then raw,
+ // then the sidecar) so the reason names the thing a normalize run would
+ // actually report — `no-meta` and `no-raw` are its two `skipped` outcomes, and
+ // neither is fixable by running it. Every branch below returned fresh:false
+ // before this change too, so no caller's behaviour moves.
+ if (metaMs === null) return { fresh: false, reason: "no-meta", cuesPath };
+ if (rawMs === null) return { fresh: false, reason: "no-raw", cuesPath };
+ if (cuesMs === null) return { fresh: false, reason: "missing", cuesPath };
+ if (cuesMs < metaMs || cuesMs < rawMs) {
+ return { fresh: false, reason: "stale", cuesPath };
+ }
+ return { fresh: true, reason: "fresh", cuesPath };
}
diff --git a/common/jobs/jobKinds.ts b/common/jobs/jobKinds.ts
@@ -130,6 +130,23 @@ const JOB_KINDS: Record<string, JobKindMeta> = {
bookmarkable: false,
queueKeyStrategy: "parallel",
},
+ // Write the compact transcript.cues.json sidecar next to every raw transcript
+ // that lacks a current one — corpus-wide from /build, or one channel from the
+ // digest stage card. It was an UNREGISTERED kind string until now: passed to
+ // runManagedFunction with no entry here, so /jobs showed the raw machine kind.
+ //
+ // Not drainable: the walk honours the cancel signal (which is what stops it)
+ // but has no drain-aware inner loop, and the unit of work is a single file
+ // write, so "let the in-flight one finish" is already how it behaves. Not
+ // bookmarkable either — that would need a JobSpec and a jobReplayRegistry
+ // handler, and the button is one click from the card that reports the count.
+ "normalize-transcripts": {
+ kind: "normalize-transcripts",
+ label: "Normalize transcripts",
+ drainable: false,
+ bookmarkable: false,
+ queueKeyStrategy: "custom",
+ },
"redownload-incomplete-bucket": {
kind: "redownload-incomplete-bucket",
label: "Re-download truncated transcripts",
diff --git a/common/lib/backfillKinds.test.ts b/common/lib/backfillKinds.test.ts
@@ -976,6 +976,37 @@ test("digest: a stale cues.json defers rather than digesting superseded text", a
assert.equal(await classifyDigest({ staleCues: true }), "deferred");
});
+test("digest: a transcript with NO cues.json defers too — the case that is 97.6% of them", async () => {
+ // The population this branch actually catches. Measured over 79,219 video
+ // dirs: 1,942 have a raw transcript and no cues.json at all, against 47 with
+ // a superseded one — and 1,683 of the 1,942 are a single `handling: "youtube"`
+ // channel, which downloads subtitles with --skip-download and so never runs
+ // transcribeOne, the only automatic caller of normalizeTranscript.
+ //
+ // Same classification as the stale case on purpose: one normalize pass fixes
+ // both, so the distinction is about COPY (see CuesFreshReason), not dispatch.
+ assert.equal(
+ await classifyDigest({ transcript: true, cues: false }),
+ "deferred",
+ );
+});
+
+test("a kind that can defer says WHY, and digest's reason is not 'it clears itself'", () => {
+ // BackfillStage used to hardcode one sentence about the diarization duration
+ // cap for every kind's deferred videos at once. Correct only while diarization
+ // was the sole kind that could defer.
+ assert.match(
+ getBackfillKind("diarization")!.deferredHint!,
+ /Max audio hours/,
+ );
+ // The claim this whole change exists to retract: the old copy said the
+ // normalize pass clears these on its own. Nothing runs it on its own, so the
+ // hint has to name the action.
+ const digestHint = digestKind.deferredHint!;
+ assert.match(digestHint, /Normalize/);
+ assert.doesNotMatch(digestHint, /clears? (itself|them|it)/i);
+});
+
test("digest: a digest shared from a duplicate cluster counts as done", async () => {
// Worth ~11% of the sweep. If this entry disagreed with the snapshot's
// noDigest bucket here, every mirror would be regenerated.
diff --git a/common/lib/backfillKinds.ts b/common/lib/backfillKinds.ts
@@ -245,6 +245,17 @@ export type BackfillKind = {
label: string;
// One line of UI copy: what this backfill is, in the operator's terms.
hint: string;
+ // What a `deferred` video of THIS kind is waiting for, and what an operator
+ // can do about it. Belongs to the kind, not to the card: BackfillStage used to
+ // hardcode "too long to diarize under the current limit", which was correct
+ // only while diarization was the sole kind that could defer. Digest defers for
+ // an unrelated reason (no current normalized transcript), so a card summing
+ // several kinds' `deferred` into one hardcoded sentence now states a cause
+ // that is false for most of what it counts.
+ //
+ // Written as a sentence FRAGMENT completing "N videos are …", so the card
+ // keeps ownership of the count and its pluralization.
+ deferredHint?: string;
tier: BackfillCostTier;
// Where this operation's work runs. See BackfillLane — the queue key is what
// keeps CPU and GPU operations overlapping instead of taking turns.
@@ -296,6 +307,9 @@ const diarization: BackfillKind = {
id: "diarization",
label: "Speaker diarization",
hint: "Speaker turns captured from the audio, written to diarization.json beside the transcript.",
+ // The wording BackfillStage used to hardcode for every kind at once.
+ deferredHint:
+ "too long to diarize under the current limit — raise or clear Max audio hours in Settings to include them",
tier: "lane",
// CPU, on the shared backfill queue. Serialized against the other backfill
// kinds on purpose — two channels' worth of diarization at once just thrashes
@@ -638,6 +652,11 @@ const digest: BackfillKind = {
id: "digest",
label: "Digest",
hint: "Chapters and tags generated from the transcript by a local or metered model. Needs a transcript first.",
+ // See the `deferred` branch in state() below for the measurement behind this
+ // wording. It says "run the normalize pass" and NOT "it clears itself",
+ // because nothing automatic ever will.
+ deferredHint:
+ "waiting on a normalized transcript (transcript.cues.json) that nothing produces automatically — run Normalize transcripts on the channel to make them digestable",
tier: "lane",
// The transcript, declared. Nothing in the repo enforced this before.
dependsOn: ["transcription"],
@@ -675,12 +694,32 @@ const digest: BackfillKind = {
if (!isVideoTranscribed(files)) return "blocked";
const { fresh: cuesFresh } = await isCuesJsonFresh(videoDir);
if (!cuesFresh) {
- // The raw transcript changed under cues.json, so the normalize pass owes
- // this video a rewrite. Digesting now would describe superseded text and
- // then look fresh forever. It resolves itself, which is why digestVideo
- // reports it as `skipped` rather than a failure — and `deferred` is the
- // classification with the matching meaning: not attempted, not broken,
- // not counted as reachable work.
+ // NO CURRENT NORMALIZED TRANSCRIPT. Digesting now would either fail for
+ // want of one or describe superseded text and then look fresh forever, so
+ // the video is held back — `deferred` is the classification with the
+ // matching meaning: not attempted, not broken, not counted as reachable
+ // work. digestVideo reports the same condition as `skipped` rather than a
+ // failure, for the same reason.
+ //
+ // THIS DOES NOT RESOLVE ITSELF, and an earlier version of this comment
+ // said it did. Measured over the whole corpus (79,219 video dirs):
+ //
+ // 1,942 have NO cues.json at all ← reason "missing"
+ // 47 have one that is superseded ← reason "stale"
+ //
+ // so the superseded case this branch was written for is 2.4% of what it
+ // actually catches. The missing case is permanent: transcribeOne is the
+ // ONLY automatic caller of normalizeTranscript, and a channel with
+ // `handling: "youtube"` fetches subtitles with --skip-download and so
+ // never runs it — 1,683 of the 1,942 are piratesoftware alone. It went
+ // unnoticed because buildIndex treats cues.json as a CACHE and silently
+ // re-parses the raw VTT when it is absent, so the published site is
+ // correct and only this lane, which has no such fallback, can see it.
+ //
+ // The fix is the normalize pass, run deliberately: normalizeChannel-
+ // Transcripts (controller/normalizeAll.ts), wired to a button on the
+ // digest stage card next to this count. Both reasons are fixed by it,
+ // which is why they share one classification — see CuesFreshReason.
return "deferred";
}
const record = await loadDigest(videoDir);
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,7 +1,9 @@
# Changelog
## [Unreleased]
-- **There is now one definition of "digested", and every screen reads it.** The digest layer kept its own private tally of what still needed doing, separate from the one the digest runner actually uses — and the two could disagree by an entire channel. They are now the same number. Two consequences you will see immediately. **Channels whose reports were months old start reporting digest work at all**: eleven of them predated the old tally entirely and had been quietly reading as "nothing to do" (they account for about 1,500 videos, most of them in one channel). And **videos whose transcript is being rewritten are no longer offered as work**: they were being handed to the digest runner, which looked at them and immediately put them back. On the current corpus that is 1,933 videos, and 1,683 of them are in a single channel — so nearly all of that channel's apparent digest backlog was work that could never have started. The overall count goes *down* slightly as a result, from 75,613 to 75,199, which is the counter becoming honest rather than anything being skipped.
+- **The digest backlog that could never start now has a button that clears it, and honest copy about why.** Nearly 2,000 videos across the corpus have a transcript but no compact transcript file beside it, and the digest generator refuses those — so they sat in a "waiting" count on the Digest card under text that said the problem would resolve on its own. It does not. Nothing writes that file automatically for a channel whose captions are *downloaded* rather than transcribed here, which is why one channel alone accounts for 1,683 of them and was showing about 6% digest coverage. The card now says what is actually wrong, says that nothing will fix it unattended, and offers **Normalize transcripts** right there to fix it for that channel — instead of the corpus-wide button on the Build page that walks all 79,000 video folders. The button appears only when there is something for it to do.
+- **Skipped-video explanations on the Backfill card now match the reason each one was skipped.** The card had one hardcoded sentence about the speaker-capture length limit and showed it for every kind of skipped video, including ones skipped for completely unrelated reasons. Each kind of work now supplies its own explanation and gets its own line.
+- **There is now one definition of "digested", and every screen reads it.** The digest layer kept its own private tally of what still needed doing, separate from the one the digest runner actually uses — and the two could disagree by an entire channel. They are now the same number. Two consequences you will see immediately. **Channels whose reports were months old start reporting digest work at all**: eleven of them predated the old tally entirely and had been quietly reading as "nothing to do" (they account for about 1,500 videos, most of them in one channel). And **videos with no usable transcript file are no longer offered as work**: they were being handed to the digest runner, which looked at them and immediately put them back. On the current corpus that is 1,989 videos, and 1,683 of them are in a single channel — so nearly all of that channel's apparent digest backlog was work that could never have started. The overall count goes *down* slightly as a result, from 75,613 to 75,199, which is the counter becoming honest rather than anything being skipped.
- **A video with no transcript is now visible instead of invisible.** It used to be absent from every digest count, which meant a channel of untranscribed videos read as fully digested. The Digest card now says how many are waiting on transcription and, when that is all that is left, says so instead of "All digested". These are counted separately and never added to the work total — there is nothing you can do about them from that screen, and they clear themselves as transcription catches up.
- **Part-finished digests are counted apart from unstarted ones.** A digest is generated in sections, and a re-run only regenerates the sections that are actually out of date. So a video that already has its chapters and only needs its topic tags is a fraction of the work of one with neither — but both used to be reported with the same word. They are now separate, which matters most the moment topic tags are switched on: every already-digested video in the archive becomes part-finished at once, and without the distinction that would have looked identical to a change that invalidated the whole archive.
- **The digest coverage percentage on the dashboard was dividing by the wrong number.** It counted every video folder, including the ones with no transcript and the ones marked as impossible to transcribe — so the bar could never have reached 100% however long a sweep ran. It now divides by the videos that can actually carry a digest.
diff --git a/editor/app/channels/[slug]/components/stages/BackfillStage.tsx b/editor/app/channels/[slug]/components/stages/BackfillStage.tsx
@@ -27,10 +27,17 @@ export type BackfillKindView = {
// Videos whose input is gone. A COUNT only — the id list is corpus-sized and
// deliberately not stored in the snapshot (see BackfillSnapshotEntry).
missingInput: number;
- // Videos this kind refuses to attempt under the current configuration (the
- // diarization duration cap). A THIRD number, never added to the other two:
- // summing it would let a capped corpus report as finished.
+ // Videos this kind refuses to attempt under the current configuration. A
+ // THIRD number, never added to the other two: summing it would let a capped
+ // corpus report as finished.
deferred: number;
+ // WHY this kind defers, as a sentence fragment completing "N videos are …",
+ // resolved from the registry on the server. It used to be one hardcoded
+ // sentence about the diarization duration cap, which was only ever right
+ // because diarization was the sole kind that could defer — the moment a
+ // second kind defers for another reason, a card summing them states a cause
+ // that is false for most of what it counts.
+ deferredHint?: string;
// Videos waiting on another kind's output. A FOURTH number, never added to
// the others either — but unlike the three above it needs nothing from the
// operator, because the prerequisite lane brings it down on its own.
@@ -66,8 +73,11 @@ export function BackfillStage({
const reachable = kinds.reduce((n, k) => n + k.reachableIds.length, 0);
const missingInput = kinds.reduce((n, k) => n + k.missingInput, 0);
- const deferred = kinds.reduce((n, k) => n + k.deferred, 0);
const blocked = kinds.reduce((n, k) => n + k.blocked, 0);
+ // ONE LINE PER KIND, not one summed line: each kind defers for its own reason
+ // and the fix differs, so a single count under a single sentence would attach
+ // one kind's remedy to another kind's videos.
+ const deferredKinds = kinds.filter((k) => k.deferred > 0);
// What the blocked videos are waiting for, named. Only kinds that actually
// have blocked videos contribute, so the sentence never lists a prerequisite
// that is not holding anything up.
@@ -104,18 +114,20 @@ export function BackfillStage({
: " — re-download is off, so this run skips them."}
</p>
)}
- {deferred > 0 && (
+ {deferredKinds.map((k) => (
<p
+ key={k.id}
aria-label="backfill deferred"
+ data-kind={k.id}
className="mt-1 text-sm text-muted-foreground"
>
- {deferred.toLocaleString()}{" "}
- {deferred === 1 ? "video is" : "videos are"} too long to diarize
- under the current limit and {deferred === 1 ? "is" : "are"} being
- skipped — raise or clear <em>Max audio hours</em> in Settings to
- include {deferred === 1 ? "it" : "them"}.
+ {k.deferred.toLocaleString()}{" "}
+ {k.deferred === 1 ? "video is" : "videos are"}{" "}
+ {k.deferredHint ??
+ `being skipped by ${k.label} under the current configuration`}
+ .
</p>
- )}
+ ))}
{blocked > 0 && (
<p
aria-label="backfill blocked"
diff --git a/editor/app/channels/[slug]/components/stages/DigestStage.tsx b/editor/app/channels/[slug]/components/stages/DigestStage.tsx
@@ -18,6 +18,7 @@ import {
digestChannelAction,
type DigestLaneChoice,
} from "../../digestActions";
+import { normalizeChannelAction } from "../../normalizeActions";
import { VideoIdList } from "../VideoIdList";
type Props = {
@@ -32,10 +33,15 @@ type Props = {
// operation declared its dependency — absent from every bucket, so a channel
// of untranscribed videos read as fully digested.
blocked: number;
- // Videos whose cues.json is stale, so the normalize pass owes them a rewrite.
- // Digesting one now would describe superseded text and then look fresh
- // forever, so the runner skips them — this is the count of work that would
+ // Videos with a transcript but no CURRENT normalized transcript
+ // (transcript.cues.json) — either none at all, or one superseded by a rewrite
+ // of the raw file. The digest lane refuses them, so this is work that would
// have been offered and immediately declined.
+ //
+ // NOT self-clearing, which is why this card carries a button. Measured over
+ // the corpus: 1,942 have no cues.json and 47 have a stale one, and nothing
+ // automatic writes the missing ones — see the digest entry in
+ // common/lib/backfillKinds.ts.
deferred: number;
// Part of the reachable count above, not an addition to it: videos that have
// SOME configured section at the current identity and not the rest. Broken
@@ -130,11 +136,13 @@ export function DigestStage({
className="mt-1 text-sm text-muted-foreground"
>
{deferred.toLocaleString()} more{" "}
- {deferred === 1 ? "video is" : "videos are"} waiting on a transcript
- rewrite — {deferred === 1 ? "its" : "their"} <code>cues.json</code>{" "}
- is older than the transcript it came from, so digesting now would
- describe superseded text. Nothing to do here; the normalize pass
- clears {deferred === 1 ? "it" : "them"}.
+ {deferred === 1 ? "video has" : "videos have"} a transcript but no
+ current <code>transcript.cues.json</code>, so{" "}
+ {deferred === 1 ? "it is" : "they are"} held back from the digest
+ lane. Nothing produces one automatically — a channel whose subtitles
+ are downloaded rather than transcribed never runs the normalizer —
+ so this does not clear itself. Run <em>Normalize transcripts</em>{" "}
+ below to make {deferred === 1 ? "it" : "them"} digestable.
</p>
)}
{blocked > 0 && (
@@ -217,6 +225,26 @@ export function DigestStage({
</>
}
/>
+
+ {/*
+ The fix for the `deferred` count, on the card that reports it. Rendered
+ only when there is something to fix: this walks every video dir in the
+ channel, and offering it on a channel with nothing deferred would invite
+ a no-op pass over ~11,000 dirs on the largest one.
+
+ It runs on the CHANNEL queue rather than either digest lane — see
+ normalizeActions.ts. Queueing the unblocker behind the lane it unblocks
+ would be self-defeating on a sweep that runs for weeks.
+ */}
+ {deferred > 0 && (
+ <StreamActionLog
+ trigger={() => normalizeChannelAction(slug)}
+ cancelAction={cancelJobAction}
+ buttonLabel="Normalize transcripts"
+ runningLabel="Normalizing…"
+ label="Normalize transcripts"
+ />
+ )}
</div>
);
}
diff --git a/editor/app/channels/[slug]/normalizeActions.ts b/editor/app/channels/[slug]/normalizeActions.ts
@@ -0,0 +1,62 @@
+"use server";
+
+// Write the compact transcript.cues.json sidecar for one channel.
+//
+// This exists because the digest lane's `deferred` bucket had a count and no
+// affordance. A video with a raw transcript and no cues.json is permanently
+// undigestable — transcribeOne is the only automatic caller of
+// normalizeTranscript, and a `handling: "youtube"` channel fetches subtitles
+// with --skip-download and never reaches it — and the only fix in the repo was
+// a corpus-wide button on /build. So clearing one channel meant walking all
+// ~79,000 video dirs from a page nobody opens during digest work.
+//
+// ON THE CHANNEL QUEUE, not the digest or backfill one: this is exactly the
+// "channel-local bookkeeping" queueKeys.ts describes channelQueueKey for, and
+// putting it anywhere else would make the fix for a stalled digest lane queue
+// up BEHIND the digest lane it is meant to unblock.
+
+import { revalidatePath } from "next/cache";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import {
+ channelQueueKey,
+ resolveQueueKey,
+} from "yt-dlp-transcript-common/lib/queueKeys";
+import {
+ runManagedFunction,
+ type StreamActionResult,
+} from "yt-dlp-transcript-common/jobs/streamCommand";
+import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
+import { normalizeChannelTranscripts } from "yt-dlp-transcript-common/controller/normalizeAll";
+
+export async function normalizeChannelAction(
+ slug: string,
+ queueKey?: string,
+): Promise<StreamActionResult> {
+ const paths = getPaths();
+ return runManagedFunction({
+ kind: "normalize-transcripts",
+ queueKey: resolveQueueKey(channelQueueKey(slug), queueKey),
+ paths,
+ channelSlug: slug,
+ fn: async (onLog, signal) => {
+ const result = await normalizeChannelTranscripts({
+ paths,
+ channelSlug: slug,
+ onLog,
+ signal,
+ });
+ // `skipped` is reported rather than swallowed because it is the honest
+ // floor of this operation: videos with no raw transcript at all, which
+ // normalizing cannot help and which the card counts separately as
+ // `blocked` on transcription.
+ onLog(
+ `Normalize ${slug}: ${result.wrote} written, ${result.fresh} already current, ` +
+ `${result.skipped} skipped (no transcript or no metadata), ${result.failed} failed.`,
+ );
+ // The digest lane's classification is derived from disk, so the count this
+ // run just moved is only visible once the snapshot is rebuilt.
+ requestChannelSnapshot(paths, slug);
+ revalidatePath(`/channels/${slug}`);
+ },
+ });
+}
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -446,6 +446,9 @@ export default async function ChannelDetailPage({
// cap existed have no `deferred` field, and .toLocaleString() on
// undefined throws in the render path.
deferred: entry?.deferred ?? 0,
+ // From the registry, not the card: the card sums several kinds and
+ // cannot know why any one of them deferred.
+ deferredHint: kind.deferredHint,
// Same reason: every snapshot on disk predates this field.
blocked: entry?.blocked ?? 0,
// Resolved here, on the server, because BACKFILL_KIND_BY_ID reads
diff --git a/editor/e2e/digest.spec.ts b/editor/e2e/digest.spec.ts
@@ -743,3 +743,95 @@ test("tags are not generated unless the setting asks for them", async ({
expect(record.sections.chapters?.items.length).toBeGreaterThan(0);
expect(record.sections.tags).toBeUndefined();
});
+
+// ---------------------------------------------------------------------------
+// Deferred: a transcript with no normalized sidecar, and the button that fixes it
+//
+// The state this covers is 1,942 videos on the live corpus and was permanently
+// invisible to the digest lane: transcribeOne is the ONLY automatic caller of
+// normalizeTranscript, so a channel whose subtitles are DOWNLOADED (handling
+// "youtube", which passes --skip-download) never writes transcript.cues.json at
+// all. buildIndex re-parses the raw VTT when the sidecar is absent, so the
+// published site is correct and nothing else in the system notices — but the
+// digest lane has no such fallback, and these videos are excluded from a sweep
+// that costs GPU-weeks.
+//
+// The card used to report them under copy that said the normalize pass would
+// clear them on its own. Nothing runs it on its own. So this test asserts all
+// three halves: the count, the honest copy, and that the button turns a deferred
+// video into reachable digest work.
+// ---------------------------------------------------------------------------
+
+const UNNORMALIZED = "digestvid0002";
+
+// generateReport short-circuits on an existing snapshot, and the classification
+// here is derived from disk — so a re-read after a mutation must drop the
+// snapshot first or it reports the pre-normalize world forever.
+async function regenerateReport(
+ page: import("@playwright/test").Page,
+ slug: string,
+): Promise<void> {
+ await rm(resolvePath(`test-transcripts/channels/${slug}/snapshot.json`), {
+ force: true,
+ });
+ await generateReport(page, slug);
+}
+
+test("a transcript with no cues.json is deferred, and Normalize transcripts makes it digestable", async ({
+ page,
+}) => {
+ await resetData(null);
+ await writeSettings(digestSettings());
+ await writeChannelConfig(CHANNEL);
+ // Exactly the live shape: a real VTT, real metadata, no sidecar.
+ await writeDigestVideo({
+ channelSlug: CHANNEL,
+ videoId: UNNORMALIZED,
+ skipCuesJson: true,
+ });
+
+ const cuesRel = join(
+ "test-transcripts",
+ "channels",
+ CHANNEL,
+ "data",
+ UNNORMALIZED,
+ "transcript.cues.json",
+ );
+ expect(await pathExists(cuesRel)).toBe(false);
+
+ await generateReport(page, CHANNEL);
+ await page.goto(`/channels/${CHANNEL}`);
+
+ // Counted as deferred — NOT as reachable work, and NOT as blocked (the video
+ // is transcribed; nothing is waiting on the transcription lane).
+ const deferredLine = page.getByLabel("digest deferred");
+ await expect(deferredLine).toContainText("1");
+ await expect(deferredLine).toContainText("transcript.cues.json");
+ // The retracted claim. The old copy said the normalize pass clears these; it
+ // does, but only when somebody runs it, and nothing does.
+ await expect(deferredLine).toContainText("does not clear itself");
+ await expect(
+ page.getByRole("heading", { name: "Generate digests (0)" }),
+ ).toBeVisible();
+
+ // The affordance the card had no version of before this change.
+ await page.getByRole("button", { name: "Normalize transcripts" }).click();
+ await expect(page.getByLabel("Normalize transcripts output")).toContainText(
+ "1 written",
+ { timeout: 60_000 },
+ );
+ expect(await pathExists(cuesRel)).toBe(true);
+
+ // And the point of all of it: the video is now work the digest lane will take.
+ await regenerateReport(page, CHANNEL);
+ await page.goto(`/channels/${CHANNEL}`);
+ await expect(
+ page.getByRole("heading", { name: "Generate digests (1)" }),
+ ).toBeVisible();
+ await expect(page.getByLabel("digest deferred")).toHaveCount(0);
+ // The button retires with the count it existed to clear.
+ await expect(
+ page.getByRole("button", { name: "Normalize transcripts" }),
+ ).toHaveCount(0);
+});
diff --git a/editor/e2e/helpers.ts b/editor/e2e/helpers.ts
@@ -214,6 +214,13 @@ export async function writeDigestVideo(opts: {
// and every shared chapter would then land at the wrong moment while the
// artifact looked perfectly healthy.
startOffsetSeconds?: number;
+ // Write the VTT and metadata but NOT transcript.cues.json — the real state of
+ // 1,942 videos on the live corpus, and one nothing produces automatically: a
+ // `handling: "youtube"` channel fetches subtitles with --skip-download and so
+ // never reaches transcribeOne, the only automatic caller of the normalizer.
+ // Such a video is transcribed, published correctly (buildIndex re-parses the
+ // raw VTT when the sidecar is absent) and permanently undigestable.
+ skipCuesJson?: boolean;
}) {
const {
channelSlug,
@@ -222,6 +229,7 @@ export async function writeDigestVideo(opts: {
durationSeconds = 600,
channelName = channelSlug,
startOffsetSeconds = 0,
+ skipCuesJson = false,
} = opts;
const dir = join(testTranscriptsDir, "channels", channelSlug, "data", videoId);
await mkdir(dir, { recursive: true });
@@ -303,7 +311,7 @@ export async function writeDigestVideo(opts: {
const cuesPath = join(dir, "transcript.cues.json");
await writeFile(metaPath, JSON.stringify(meta, null, 2));
await writeFile(vttPath, vtt);
- await writeFile(cuesPath, JSON.stringify(detail));
+ if (!skipCuesJson) await writeFile(cuesPath, JSON.stringify(detail));
const older = new Date(Date.now() - 60_000);
await utimes(metaPath, older, older);