commit b2ad3270da6d3b219fc7466b059617dbf73e4576
parent bd125ab0aea18df0e79644a8ea558f5d5a44a7e8
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 29 Jul 2026 11:40:15 -0400
The worst digest failures left no trace anywhere — persist them, then queue them
A correctness bug first. `digestVideo` has two total-failure paths, and both
`continue`d without calling `writeDigestSection`. The reason was sound: an empty
section carrying current provenance would read as FRESH and the video would
never be retried. The consequence was not — the failures that matter MOST
persisted **zero** warnings, surviving only in a job log that rotates at 500
records / 30 days. A warnings-driven review queue would have been systematically
blind to exactly the videos it exists to catch.
`DigestRecord.failures` is a sibling of `sections`, not a member of it, which is
what keeps the retry intact: `isSectionFresh` reads
`sections[section].provenance` and nothing else, so nothing recorded here can
make a failed video look done.
It distinguishes the two failures, because they look identical from outside and
need different fixes:
no-output — every chunk failed or came back empty. The model never proposed
anything: an engine, prompt or context problem.
all-rejected — the model proposed items and every one failed a guard.
The validation run's worst video was the second kind: 13 chapters clamped away
as out-of-range. "Never proposed" and "proposed and thrown away" are different
problems and only a recorded artifact tells them apart.
Writing tests for this found a second bug: `loadDigest` rebuilds the record
field by field rather than spreading, so `failures` round-tripped to nothing —
silently, because the write succeeded and only the read omitted it. Exactly the
shape of the `pageHashes: [""]` failure from Phase 2, and again caught only by
asserting on what came back rather than on what went in. Noted in the reader.
Then the queue. `channelSnapshot` already loads the digest sidecar inside its
per-video scan, so `buckets.digestWarnings` costs no extra I/O; it lists videos
with warnings on a written section OR a recorded total failure — the second
kind being the one that was previously invisible. From there a `digest_warnings`
video filter, an /actionable section linking into it, and the coverage counters
that were missing: `noDigest` on the dashboard channels table and the needs-work
rows, muted rather than coloured because during the backfill it is nearly every
video and a red badge on all 63 channels is not information.
Deliberately deferred: approve/dismiss state. There is no review-state field on
the sidecar and it needs a new field or a sibling file; per-video edit and
regenerate-on-the-other-lane already exist in DigestPanel.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
15 files changed, 466 insertions(+), 3 deletions(-)
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -162,6 +162,19 @@ export type ChannelSnapshot = {
// transcription problem, not a digest one. Optional: older snapshots lack
// it; readers must default to [].
noDigest: string[];
+ // Videos whose digest pass recorded something a human should look at:
+ // either warnings alongside a section that WAS written, or a total failure
+ // that wrote no section at all (DigestRecord.failures).
+ //
+ // Both kinds matter and the second is the one that used to be invisible —
+ // a total failure deliberately writes no section so the video retries, and
+ // before `failures` existed its warnings survived only in a rotating job
+ // log. A review queue keyed on written warnings alone would have been blind
+ // to precisely the worst outputs.
+ //
+ // Costs no extra I/O: the sidecar is already loaded here for noDigest.
+ // Optional: older snapshots lack it; readers must default to [].
+ digestWarnings?: string[];
};
undownloadedIds: string[];
excludedFromDownload?: ExcludedFromDownload;
@@ -465,6 +478,7 @@ export async function generateChannelSnapshot(
const downloadedAutoSubsOnly: string[] = [];
const supersededAutoSubs: string[] = [];
const noDigest: string[] = [];
+ const digestWarnings: string[] = [];
const digestEngines: Record<string, number> = {};
// The SAME freshness target countMissingDigests and runDigestBatch use, so
// the Digest stage's count and the batch runner's progress target cannot
@@ -654,6 +668,15 @@ export async function generateChannelSnapshot(
isSectionFresh(digest, section, digestTarget.target),
);
if (!fresh) noDigest.push(id);
+ // Reviewable regardless of freshness: a video that failed outright is
+ // ALSO in noDigest (it has no section), and a video whose section landed
+ // with warnings is fresh and would otherwise never be surfaced again.
+ if (
+ (digest?.warnings?.length ?? 0) > 0 ||
+ (digest?.failures?.length ?? 0) > 0
+ ) {
+ digestWarnings.push(id);
+ }
}
if (files.isUntranscribable) {
untranscribable.push(id);
@@ -762,6 +785,7 @@ export async function generateChannelSnapshot(
supersededAutoSubs: supersededAutoSubs.sort(),
needsCookies: needsCookies.sort(),
noDigest: noDigest.sort(),
+ digestWarnings: digestWarnings.sort(),
},
digestEngines,
undownloadedIds,
diff --git a/common/controller/digestVideo.ts b/common/controller/digestVideo.ts
@@ -42,7 +42,11 @@ import {
type DigestSectionKind,
type DigestWarning,
} from "../lib/digest";
-import { loadDigest, writeDigestSection } from "../lib/digest-server";
+import {
+ loadDigest,
+ writeDigestFailure,
+ writeDigestSection,
+} from "../lib/digest-server";
import { readDigestContext, type DigestContext } from "../lib/digestContext-server";
import { parseChapters, parseTags, type DigestChunkOutput } from "../lib/digestParse";
import { chunkCuesForContext } from "../lib/transcriptWindow";
@@ -248,9 +252,23 @@ export async function digestVideo(
if (outputs.length === 0) {
// Nothing usable. Do NOT write a section — an empty section with current
// provenance would read as "fresh" and the video would never be retried.
+ // But DO persist the warnings, in the sibling `failures` field that
+ // freshness never reads: this is the worst outcome the generator has, and
+ // it used to leave no trace anywhere except a job log that rotates.
log(
- `${opts.channelSlug}/${opts.videoId}: ${section} produced no usable output; leaving the sidecar untouched so it retries.`,
+ `${opts.channelSlug}/${opts.videoId}: ${section} produced no usable output (${runWarnings.length} warning(s)); leaving the sidecar untouched so it retries.`,
);
+ record = await writeDigestFailure(videoDir, {
+ section,
+ at: new Date().toISOString(),
+ appId: app.id,
+ model: reportedModel,
+ promptVersion: PROMPT_VERSION,
+ reason: "no-output",
+ chunks: chunks.length,
+ chunksOk: outputs.length,
+ warnings: runWarnings,
+ });
warningCount += runWarnings.length;
continue;
}
@@ -264,9 +282,26 @@ export async function digestVideo(
const warnings = [...runWarnings, ...parsed.warnings];
if (items.length === 0) {
+ // The model DID propose content and every item failed a guard — a
+ // different failure from "produced nothing", and one the warnings can
+ // actually explain (the validation run's worst video had 13 chapters
+ // clamped away as out-of-range). Recording which of the two happened is
+ // the difference between a review queue that can act and one that can only
+ // report a blank.
log(
`${opts.channelSlug}/${opts.videoId}: every ${section} entry was rejected (${warnings.length} warning(s)); leaving the sidecar untouched so it retries.`,
);
+ record = await writeDigestFailure(videoDir, {
+ section,
+ at: new Date().toISOString(),
+ appId: app.id,
+ model: reportedModel,
+ promptVersion: PROMPT_VERSION,
+ reason: "all-rejected",
+ chunks: chunks.length,
+ chunksOk: outputs.length,
+ warnings,
+ });
warningCount += warnings.length;
continue;
}
diff --git a/common/lib/digest-server.test.ts b/common/lib/digest-server.test.ts
@@ -0,0 +1,221 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { mkdtemp, rm } from "node:fs/promises";
+import { tmpdir } from "node:os";
+import path from "node:path";
+import {
+ DIGEST_SCHEMA_VERSION,
+ isSectionFresh,
+ type DigestFreshnessTarget,
+ type DigestProvenance,
+} from "./digest";
+import {
+ loadDigest,
+ writeDigestFailure,
+ writeDigestSection,
+} from "./digest-server";
+
+// Run with:
+// pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/digest-server.test.ts
+
+async function withVideoDir(fn: (dir: string) => Promise<void>): Promise<void> {
+ const dir = await mkdtemp(path.join(tmpdir(), "ttb-digest-server-"));
+ try {
+ await fn(dir);
+ } finally {
+ await rm(dir, { recursive: true, force: true });
+ }
+}
+
+const PROVENANCE: DigestProvenance = {
+ appId: "ollama-direct",
+ model: "qwen2.5:7b",
+ modelRequested: "qwen2.5:7b",
+ lane: "local-gpu",
+ generatedAt: "2026-07-29T00:00:00.000Z",
+ promptVersion: 2,
+ contextHash: "ctx",
+ chunks: 3,
+ chunksOk: 3,
+};
+
+const TARGET: DigestFreshnessTarget = {
+ appId: "ollama-direct",
+ model: "qwen2.5:7b",
+ promptVersion: 2,
+ contextHash: "ctx",
+};
+
+// The whole point of `failures`: the two total-failure paths in digestVideo
+// deliberately write no section so the video RETRIES, which used to mean the
+// warnings from the worst outputs survived nowhere but a rotating job log.
+test("a recorded failure persists its warnings without making the video look fresh", async () => {
+ await withVideoDir(async (dir) => {
+ await writeDigestFailure(dir, {
+ section: "chapters",
+ at: "2026-07-29T00:00:00.000Z",
+ appId: "ollama-direct",
+ model: "qwen2.5:7b",
+ promptVersion: 2,
+ reason: "all-rejected",
+ chunks: 3,
+ chunksOk: 3,
+ warnings: [
+ { code: "out-of-range", section: "chapters", detail: "13 clamped" },
+ ],
+ });
+
+ const record = await loadDigest(dir);
+ assert.equal(record?.failures?.length, 1);
+ assert.equal(record?.failures?.[0].reason, "all-rejected");
+ assert.equal(record?.failures?.[0].warnings.length, 1);
+ // The retry behaviour is the constraint this design exists to preserve.
+ assert.equal(record?.sections.chapters, undefined);
+ assert.equal(
+ isSectionFresh(record, "chapters", TARGET),
+ false,
+ "a failure must never read as fresh, or the video is never retried",
+ );
+ });
+});
+
+// "the model proposed nothing" and "the model proposed and every item was
+// rejected by a guard" look identical from outside and need different fixes.
+test("the two failure reasons are recorded distinctly", async () => {
+ await withVideoDir(async (dir) => {
+ await writeDigestFailure(dir, {
+ section: "chapters",
+ at: "2026-07-29T00:00:00.000Z",
+ appId: "a",
+ model: "m",
+ promptVersion: 2,
+ reason: "no-output",
+ chunks: 5,
+ chunksOk: 0,
+ warnings: [],
+ });
+ assert.equal((await loadDigest(dir))?.failures?.[0].reason, "no-output");
+ assert.equal((await loadDigest(dir))?.failures?.[0].chunksOk, 0);
+ });
+});
+
+// A sweep retries. Appending would grow the sidecar without bound on a video
+// that fails every single pass over 81 days.
+test("only the latest failure per section is kept", async () => {
+ await withVideoDir(async (dir) => {
+ for (const at of ["2026-07-01T00:00:00.000Z", "2026-07-29T00:00:00.000Z"]) {
+ await writeDigestFailure(dir, {
+ section: "chapters",
+ at,
+ appId: "a",
+ model: "m",
+ promptVersion: 2,
+ reason: "no-output",
+ chunks: 1,
+ chunksOk: 0,
+ warnings: [],
+ });
+ }
+ const record = await loadDigest(dir);
+ assert.equal(record?.failures?.length, 1);
+ assert.equal(record?.failures?.[0].at, "2026-07-29T00:00:00.000Z");
+ });
+});
+
+test("a failure in one section does not disturb another section's", async () => {
+ await withVideoDir(async (dir) => {
+ for (const section of ["chapters", "tags"] as const) {
+ await writeDigestFailure(dir, {
+ section,
+ at: "2026-07-29T00:00:00.000Z",
+ appId: "a",
+ model: "m",
+ promptVersion: 2,
+ reason: "no-output",
+ chunks: 1,
+ chunksOk: 0,
+ warnings: [],
+ });
+ }
+ const record = await loadDigest(dir);
+ assert.equal(record?.failures?.length, 2);
+ assert.deepEqual(
+ record?.failures?.map((f) => f.section).sort(),
+ ["chapters", "tags"],
+ );
+ });
+});
+
+test("a section that later succeeds clears its own failure, not the other's", async () => {
+ await withVideoDir(async (dir) => {
+ for (const section of ["chapters", "tags"] as const) {
+ await writeDigestFailure(dir, {
+ section,
+ at: "2026-07-29T00:00:00.000Z",
+ appId: "a",
+ model: "m",
+ promptVersion: 2,
+ reason: "all-rejected",
+ chunks: 1,
+ chunksOk: 1,
+ warnings: [],
+ });
+ }
+ await writeDigestSection(dir, {
+ section: "chapters",
+ items: [
+ { id: "c0", start: 0, clock: "0:00", title: "Intro", decidedBy: "ai" },
+ ],
+ provenance: PROVENANCE,
+ warnings: [],
+ });
+
+ const record = await loadDigest(dir);
+ assert.deepEqual(
+ record?.failures?.map((f) => f.section),
+ ["tags"],
+ "the succeeding section's failure is history; the other's is not ours to clear",
+ );
+ assert.equal(isSectionFresh(record, "chapters", TARGET), true);
+ assert.equal(record?.digestSchemaVersion, DIGEST_SCHEMA_VERSION);
+ });
+});
+
+// A mirror carrying a borrowed digest whose own regeneration fails must not
+// lose its attribution — the viewer labels borrowed content as borrowed.
+test("recording a failure preserves derivedFrom", async () => {
+ await withVideoDir(async (dir) => {
+ const { writeSharedDigest } = await import("./digest-server");
+ await writeSharedDigest(
+ dir,
+ {
+ digestSchemaVersion: DIGEST_SCHEMA_VERSION,
+ promptVersion: 2,
+ contextHash: "ctx",
+ warnings: [],
+ sections: {
+ chapters: { provenance: PROVENANCE, items: [] },
+ },
+ },
+ {
+ slug: "chan/canon",
+ clusterId: "c1",
+ sharedAt: "2026-07-29T00:00:00.000Z",
+ offsetSeconds: 0.2,
+ },
+ );
+ await writeDigestFailure(dir, {
+ section: "tags",
+ at: "2026-07-29T00:00:00.000Z",
+ appId: "a",
+ model: "m",
+ promptVersion: 2,
+ reason: "no-output",
+ chunks: 1,
+ chunksOk: 0,
+ warnings: [],
+ });
+ const record = await loadDigest(dir);
+ assert.equal(record?.derivedFrom?.slug, "chan/canon");
+ });
+});
diff --git a/common/lib/digest-server.ts b/common/lib/digest-server.ts
@@ -23,6 +23,7 @@ import {
type DigestOverrides,
type DigestProvenance,
type DigestRecord,
+ type DigestSectionFailure,
type DigestSectionKind,
type DigestTag,
type DigestWarning,
@@ -63,6 +64,12 @@ export async function loadDigest(
? (parsed.warnings as DigestWarning[])
: [],
sections: parsed.sections,
+ // This reader rebuilds the record field by field rather than spreading,
+ // so EVERY new field has to be added here or it round-trips to nothing —
+ // silently, since the write succeeds and the read just omits it.
+ ...(Array.isArray(parsed.failures)
+ ? { failures: parsed.failures as DigestSectionFailure[] }
+ : {}),
...(Array.isArray(parsed.history)
? { history: parsed.history as DigestHistoryEntry[] }
: {}),
@@ -210,6 +217,12 @@ export async function writeDigestSection(
(w) => w.section !== input.section,
);
+ // This section just succeeded, so its recorded total failure is history.
+ // Another section's failure is not ours to clear.
+ const keptFailures = (existing?.failures ?? []).filter(
+ (f) => f.section !== input.section,
+ );
+
const entry: DigestHistoryEntry = {
section: input.section,
generatedAt: input.provenance.generatedAt,
@@ -230,6 +243,7 @@ export async function writeDigestSection(
contextHash: input.provenance.contextHash,
warnings: [...keptWarnings, ...input.warnings],
sections,
+ ...(keptFailures.length > 0 ? { failures: keptFailures } : {}),
history,
};
// A freshly generated section makes the record this video's own again: it is
@@ -238,6 +252,44 @@ export async function writeDigestSection(
return record;
}
+// Record a generation pass that produced nothing usable.
+//
+// Writes NO section, on purpose — see DigestSectionFailure. The record it
+// leaves is what makes the failure reviewable at all: without it the only trace
+// is a job log that rotates, and the videos the model does worst on are exactly
+// the ones a review queue most needs to surface.
+//
+// At most one failure per section is kept: a sweep retries, and appending would
+// grow the sidecar without bound on a video that fails every pass.
+export async function writeDigestFailure(
+ videoDir: string,
+ failure: DigestSectionFailure,
+): Promise<DigestRecord> {
+ const existing = await loadDigest(videoDir);
+ const failures = [
+ ...(existing?.failures ?? []).filter((f) => f.section !== failure.section),
+ failure,
+ ];
+ const record: DigestRecord = {
+ digestSchemaVersion: DIGEST_SCHEMA_VERSION,
+ // A failed pass must NOT claim the record's top-level identity — those
+ // mirror the last pass that actually wrote a section, and overwriting them
+ // here would make a corpus survey read a failure as a generation.
+ promptVersion: existing?.promptVersion ?? failure.promptVersion,
+ contextHash: existing?.contextHash ?? "",
+ warnings: existing?.warnings ?? [],
+ sections: existing?.sections ?? {},
+ failures,
+ ...(existing?.history ? { history: existing.history } : {}),
+ // Preserved: a mirror whose own regeneration failed is still carrying the
+ // canonical member's digest, and dropping this would silently un-attribute
+ // borrowed content the viewer labels as borrowed.
+ ...(existing?.derivedFrom ? { derivedFrom: existing.derivedFrom } : {}),
+ };
+ await writeDigest(videoDir, record);
+ return record;
+}
+
// Copy a canonical member's digest onto an aligned duplicate, stamped with the
// provenance of where it came from and the measured timing offset that made
// sharing safe. The receiving video's own overrides are untouched, so a human
diff --git a/common/lib/digest.ts b/common/lib/digest.ts
@@ -301,6 +301,39 @@ export type DigestHistoryEntry = {
warningCount: number;
};
+// A generation pass that produced NOTHING usable, recorded so the failure
+// survives.
+//
+// The two total-failure paths in digestVideo deliberately do not write a
+// section: an empty section carrying current provenance would read as FRESH and
+// the video would never be retried. The consequence was that the WORST failures
+// persisted zero evidence — they existed only in a job log that rotates at 500
+// records / 30 days — so a warnings-driven review queue would have been
+// systematically blind to exactly the videos it exists to catch.
+//
+// This is a sibling of `sections`, not a member of it, which is what keeps the
+// retry behaviour intact: isSectionFresh reads `sections[section].provenance`
+// and nothing else, so nothing here can make a failed video look done.
+export type DigestSectionFailure = {
+ section: DigestSectionKind;
+ at: string; // ISO
+ appId: string;
+ model: string;
+ promptVersion: number;
+ // WHY nothing survived, and the distinction is the whole point:
+ // "no-output" — every chunk failed or came back empty. The model never
+ // proposed anything: an engine, prompt or context problem.
+ // "all-rejected" — the model proposed items and every one failed a guard.
+ // The content exists and the guard threw it away.
+ // The validation run's worst video was the second kind — 13 chapters clamped
+ // away as out-of-range — and it looked identical to the first from outside.
+ // They need different fixes, and only a recorded artifact tells them apart.
+ reason: "no-output" | "all-rejected";
+ chunks: number;
+ chunksOk: number;
+ warnings: DigestWarning[];
+};
+
export type DigestRecord = {
digestSchemaVersion: number;
// The most recent generation pass's prompt/context identity. Per-section
@@ -311,6 +344,11 @@ export type DigestRecord = {
contextHash: string;
warnings: DigestWarning[];
sections: DigestSections;
+ // Latest total failure per section, if any. At most one entry per section —
+ // an 81-day sweep retries, and an append-only list would grow without bound
+ // on a video that fails every time. Cleared for a section that later
+ // succeeds. See DigestSectionFailure.
+ failures?: DigestSectionFailure[];
history?: DigestHistoryEntry[];
// Set when this digest was SHARED from another video (a duplicate cluster's
// canonical member) rather than generated for this one. Keeps the sharing
diff --git a/editor/app/actionable/lib/loadActionable.ts b/editor/app/actionable/lib/loadActionable.ts
@@ -31,6 +31,7 @@ export type ActionableSummary = {
cleanTranscribedAudio: ActionableRow[];
cleanExtraFormats: ActionableRow[];
staleOrMissing: ActionableRow[];
+ digestWarnings: ActionableRow[];
duplicates: DuplicateReport | null;
// The human decisions kept alongside the report — a cluster's canonical
// choice, "not a duplicate", and the `confirmed` flag that is the only thing
@@ -98,6 +99,18 @@ export function actionableCleanExtraFormatsCount(row: ActionableRow): number {
return row.snapshot?.buckets.multipleAudioFormats?.length ?? 0;
}
+// The digest layer's two work lists. `noDigest` is "has no digest at the
+// current identity" — the backfill's denominator. `digestWarnings` is "the
+// model produced something a human should look at", which includes the total
+// failures that write no section and so are invisible to any count of files.
+export function actionableNoDigestCount(row: ActionableRow): number {
+ return row.snapshot?.buckets.noDigest?.length ?? 0;
+}
+
+export function actionableDigestWarningsCount(row: ActionableRow): number {
+ return row.snapshot?.buckets.digestWarnings?.length ?? 0;
+}
+
// Estimated bytes each cleanup would reclaim (default 0 for snapshots written
// before cleanupBytes existed).
export function actionableCleanTranscribedBytes(row: ActionableRow): number {
@@ -162,6 +175,13 @@ export async function loadActionableSummary(
actionableCleanExtraFormatsCount(a),
);
+ const digestWarnings = rows
+ .filter((r) => actionableDigestWarningsCount(r) > 0)
+ .sort(
+ (a, b) =>
+ actionableDigestWarningsCount(b) - actionableDigestWarningsCount(a),
+ );
+
const staleOrMissing = rows
.filter(isStaleOrMissing)
.sort((a, b) => a.channel.slug.localeCompare(b.channel.slug));
@@ -175,6 +195,7 @@ export async function loadActionableSummary(
cleanTranscribedAudio,
cleanExtraFormats,
staleOrMissing,
+ digestWarnings,
duplicates,
duplicateOverrides,
};
diff --git a/editor/app/actionable/page.tsx b/editor/app/actionable/page.tsx
@@ -7,6 +7,7 @@ import {
actionableCleanExtraFormatsCount,
actionableCleanTranscribedBytes,
actionableCleanTranscribedCount,
+ actionableDigestWarningsCount,
actionableIncompleteTranscriptCount,
actionableShortAudioCount,
actionableUndownloadedCount,
@@ -57,7 +58,8 @@ export default async function ActionablePage() {
summary.shortAudio.length === 0 &&
summary.cleanTranscribedAudio.length === 0 &&
summary.cleanExtraFormats.length === 0 &&
- summary.staleOrMissing.length === 0;
+ summary.staleOrMissing.length === 0 &&
+ summary.digestWarnings.length === 0;
const sections: { config: SectionConfig; rows: ActionableRow[] }[] = [
{
@@ -130,6 +132,27 @@ export default async function ActionablePage() {
},
{
config: {
+ id: "digest-warnings",
+ title: "Channels with digest passes that need a look",
+ description:
+ "Videos where the AI digest pass recorded warnings, or produced nothing usable at all. The second kind is the one worth opening: a total failure deliberately writes no section so the video retries, and “the model proposed nothing” and “the model proposed chapters and every one was rejected by a guard” look identical from outside but need different fixes.",
+ countLabel: "with warnings",
+ emptyLabel: "None recorded.",
+ getCount: actionableDigestWarningsCount,
+ primaryAction: (r) => (
+ <Link
+ href={`/channels/${r.channel.slug}?filter=digest_warnings`}
+ aria-label={`review digest warnings for ${r.channel.slug}`}
+ className="inline-flex items-center px-2.5 py-1 rounded-md border border-border text-xs font-medium hover:bg-muted whitespace-nowrap"
+ >
+ Review
+ </Link>
+ ),
+ },
+ rows: summary.digestWarnings,
+ },
+ {
+ config: {
id: "short-audio",
title: "Channels with truncated downloads (short audio)",
description:
diff --git a/editor/app/api/widget/actionable/route.ts b/editor/app/api/widget/actionable/route.ts
@@ -1,6 +1,7 @@
import { NextResponse } from "next/server";
import { getPaths } from "yt-dlp-transcript-common/lib/paths";
import {
+ actionableNoDigestCount,
actionableUndownloadedCount,
actionableUntranscribedCount,
loadActionableSummary,
@@ -12,6 +13,12 @@ export type WidgetActionableChannel = {
slug: string;
undownloaded: number;
untranscribed: number;
+ // Videos with no digest at the CURRENT identity — the backfill's per-channel
+ // work list. Reported but NOT used to decide whether a channel "needs work":
+ // during the backfill that is 99.87% of the corpus, so counting it would put
+ // every channel in the list forever and drown the two buckets a human can
+ // actually act on today.
+ noDigest: number;
};
export type WidgetActionablePayload = {
@@ -29,6 +36,7 @@ export async function GET() {
slug: row.channel.slug,
undownloaded: actionableUndownloadedCount(row),
untranscribed: actionableUntranscribedCount(row),
+ noDigest: actionableNoDigestCount(row),
}))
.filter((c) => c.undownloaded > 0 || c.untranscribed > 0)
.sort(
diff --git a/editor/app/channels/[slug]/components/VideoListPane.tsx b/editor/app/channels/[slug]/components/VideoListPane.tsx
@@ -68,6 +68,7 @@ const FILTER_OPTIONS: { value: VideoFilter; label: string }[] = [
{ value: "downloaded_no_transcript", label: "No transcript" },
{ value: "partial", label: "Partial" },
{ value: "incomplete_transcript", label: "Incomplete transcript" },
+ { value: "digest_warnings", label: "Digest warnings" },
{ value: "short_audio", label: "Short audio" },
{ value: "untranscribable", label: "Untranscribable" },
{ value: "running", label: "Running" },
diff --git a/editor/app/channels/[slug]/lib/videoRows.ts b/editor/app/channels/[slug]/lib/videoRows.ts
@@ -39,6 +39,11 @@ export type VideoRow = {
// duration — the audio download truncated silently. Independent flag (not a
// `status`) so it composes with `transcribed`. See transcriptCoverage.
incompleteTranscript: boolean;
+ // The digest pass recorded warnings, or failed outright and wrote no section
+ // at all. Independent flag (it composes with `transcribed`, and a total
+ // failure is also in noDigest) so the review queue can ask "what did the model
+ // do badly here?" separately from "what has no digest yet?".
+ digestWarnings: boolean;
// Download completed but the audio was far shorter than the video — the source
// served a truncated stream (download-outcome "failed-short-audio"). The stub
// is kept on disk; re-download with a different format to fix. Independent flag
@@ -65,6 +70,7 @@ export type VideoFilter =
| "incomplete_transcript"
| "short_audio"
| "untranscribable"
+ | "digest_warnings"
| "running";
const VIDEO_FILTERS: readonly VideoFilter[] = [
@@ -77,6 +83,7 @@ const VIDEO_FILTERS: readonly VideoFilter[] = [
"incomplete_transcript",
"short_audio",
"untranscribable",
+ "digest_warnings",
"running",
];
@@ -111,6 +118,8 @@ function matchesFilter(r: VideoRow, filter: VideoFilter): boolean {
return r.shortAudio;
case "untranscribable":
return r.untranscribable;
+ case "digest_warnings":
+ return r.digestWarnings;
case "running":
return r.running;
}
diff --git a/editor/app/channels/[slug]/lib/videoRowsServer.ts b/editor/app/channels/[slug]/lib/videoRowsServer.ts
@@ -42,6 +42,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] {
const partial = new Set(buckets.partialDownloads);
const incompleteTranscript = new Set(buckets.incompleteTranscript);
const shortAudioSet = new Set(buckets.shortAudio ?? []);
+ const digestWarningsSet = new Set(buckets.digestWarnings ?? []);
const corruptSourceSet = new Set(buckets.corruptSource);
const corruptFullSourceSet = new Set(buckets.corruptFullSource);
const failedTranscription = new Set(input.failedTranscriptionIds);
@@ -113,6 +114,7 @@ export function computeVideoRows(input: ComputeRowsInput): VideoRow[] {
excluded,
incompleteTranscript: incompleteTranscript.has(id),
shortAudio: shortAudioSet.has(id),
+ digestWarnings: digestWarningsSet.has(id),
running: runningIds.has(id),
status,
});
diff --git a/editor/app/components/dashboard/ChannelsTable.tsx b/editor/app/components/dashboard/ChannelsTable.tsx
@@ -32,6 +32,12 @@ export function ChannelsTable({
<th className="text-left font-medium px-3 py-2">Slug</th>
<th className="text-left font-medium px-3 py-2">Handling</th>
<th className="text-right font-medium px-3 py-2">Videos</th>
+ <th
+ className="text-right font-medium px-3 py-2 whitespace-nowrap"
+ title="Videos with no AI digest at the current prompt/model identity"
+ >
+ No digest
+ </th>
<th className="text-left font-medium px-3 py-2 whitespace-nowrap">
Last sync
</th>
@@ -57,6 +63,11 @@ export function ChannelsTable({
<td className="px-3 py-2 text-right tabular-nums">
{c.videoCount}
</td>
+ {/* Muted: during the backfill this is nearly every video, so
+ it is a coverage readout rather than a call to action. */}
+ <td className="px-3 py-2 text-right tabular-nums text-xs text-muted-foreground">
+ {c.noDigest}
+ </td>
<td className="px-3 py-2 text-xs text-muted-foreground whitespace-nowrap tabular-nums">
{c.lastSyncedAt == null
? "never"
diff --git a/editor/app/components/dashboard/NeedsWorkPanel.tsx b/editor/app/components/dashboard/NeedsWorkPanel.tsx
@@ -68,6 +68,17 @@ export function NeedsWorkPanel({
✎ {c.untranscribed}
</span>
)}
+ {/* Muted, not coloured: during the backfill this is nearly
+ every video in the channel, so it is context for the row
+ rather than a call to action like the two above. */}
+ {c.noDigest > 0 && (
+ <span
+ title={`${c.noDigest} video(s) with no digest at the current identity`}
+ className="rounded px-1.5 py-0.5 text-[10px] tabular-nums bg-muted text-muted-foreground"
+ >
+ ◆ {c.noDigest}
+ </span>
+ )}
</span>
</div>
<div className="flex flex-wrap items-center gap-1.5">
diff --git a/editor/app/components/dashboard/types.ts b/editor/app/components/dashboard/types.ts
@@ -11,4 +11,8 @@ export type DashboardChannel = {
hasUrl: boolean;
undownloaded: number;
untranscribed: number;
+ // Videos with no digest at the CURRENT identity. The backfill's per-channel
+ // denominator, and the only per-channel number that makes corpus coverage
+ // legible while a multi-week sweep is running.
+ noDigest: number;
};
diff --git a/editor/app/page.tsx b/editor/app/page.tsx
@@ -9,6 +9,7 @@ import {
siteChannelSlugs,
} from "yt-dlp-transcript-common/lib/site";
import {
+ actionableNoDigestCount,
actionableUndownloadedCount,
actionableUntranscribedCount,
loadActionableSummary,
@@ -75,6 +76,7 @@ export default async function Dashboard({
hasUrl: Boolean(r.channel.config.url),
undownloaded: actionableUndownloadedCount(r),
untranscribed: actionableUntranscribedCount(r),
+ noDigest: actionableNoDigestCount(r),
}));
// "Needs work" payload, same shape the widget endpoint the cockpit polls
@@ -86,6 +88,7 @@ export default async function Dashboard({
slug: c.slug,
undownloaded: c.undownloaded,
untranscribed: c.untranscribed,
+ noDigest: c.noDigest,
}))
.sort(
(a, b) =>