commit a3170a6cd448647a36480422882e91736ccfdb5a
parent 96b068b283ac6a8c744ff0374e4420a231a792bd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 27 Jul 2026 01:06:57 -0400
Fix the pilot's "un-created job": a pre-hydration lost click
STATE.md carried this as the one thing in the digest work flagged as
actually broken and "still unexplained": clicking "Digest channel" a second
time produced no job record and an empty log, with no dedupe guard in
runManagedFunction to explain it. It is on the critical path, because a
~100-video validation run is driven by exactly that button.
Reproduced against the REAL corpus on a production build, and the cause is
not in the job system at all: the click fires NO POST. The server-rendered
button has no handler until React hydrates, so a click before that is
silently discarded — no request, no job, no log, no error. Instrumented,
the first click produced 0 POSTs and a click five seconds later started the
job normally. runManagedFunction has no dedupe guard because there is
nothing to dedupe: the action was never invoked.
The window is long precisely where it hurts. The 39-video channel page is
heavy enough that hydration lags a visibly-rendered button, and STATE.md
already records that the same page takes 30 s - 9.9 min under a dev server,
which is where the pilot hit it.
StreamActionLog now disables its button until mounted. That makes the
swallowed click impossible rather than merely unlikely, turns "nothing
happened" into a button that is visibly not ready, and fixes the same
latent race for every other action built on this component (sync, download,
transcribe, digest). It also makes e2e clicks WAIT for enablement instead
of losing them — the same pre-hydration lost click that made the charts
specs flaky and needed a retry helper.
Verified after the fix: one click, one POST, job runs.
Also adds bin/digest-validate.ts, which scores digests that already exist
on disk using the bake-off's own metric definitions — the bake-off scores a
hand-picked sample it generated itself, and a validation run needs the same
numbers over what a real sweep wrote. Videos whose digest was shared from a
duplicate's canonical member are excluded from the quality metrics: they
measure the canonical's generation, not their own.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
2 files changed, 232 insertions(+), 1 deletion(-)
diff --git a/common/bin/digest-validate.ts b/common/bin/digest-validate.ts
@@ -0,0 +1,213 @@
+#!/usr/bin/env tsx
+// Score digests that ALREADY EXIST on disk, using the same metric definitions as
+// the bake-off harness (common/bin/digest-bakeoff.ts).
+//
+// The bake-off generates into a scratch directory and scores what it generated;
+// this reads what a real sweep wrote. That is the difference between "how does
+// this configuration behave on a hand-picked sample" and "how did it behave on
+// the corpus", and the second is what a validation run is for.
+//
+// Usage: pnpm exec tsx bin/digest-validate.ts <channel> [<channel> ...]
+
+import path from "node:path";
+import { readdir, readFile } from "node:fs/promises";
+import { getPaths } from "../lib/paths";
+import { loadDigest } from "../lib/digest-server";
+import { toHms } from "../lib/digestPrompt";
+import type { DigestRecord, DigestWarningCode } from "../lib/digest";
+
+// Copied verbatim from digest-bakeoff.ts so the two are comparable. A model that
+// segments correctly but names every section "Discussion" has produced a table
+// of contents nobody can navigate.
+const GENERIC_TITLE_RE =
+ /^(the\s+)?(intro(duction)?|outro|conclusion|discussion|continued|continuation|overview|summary|recap|closing( remarks)?|opening( remarks)?|final thoughts|misc(ellaneous)?|other|general|topics?|segment|section|chapter|part)\b/i;
+const GENERIC_TITLE_TAIL_RE =
+ /\b(part|section|segment|chapter)\s+(\d+|one|two|three|four|five|six|seven|eight|nine|ten)$/i;
+
+function isGenericTitle(title: string): boolean {
+ const t = title.trim();
+ return GENERIC_TITLE_RE.test(t) || GENERIC_TITLE_TAIL_RE.test(t);
+}
+
+function normalizeTitle(title: string): string {
+ return title.trim().toLowerCase().replace(/\s+/g, " ");
+}
+
+type VideoScore = {
+ slug: string;
+ durationSeconds: number;
+ chunks: number;
+ chunksOk: number;
+ kept: number;
+ maxGapSeconds: number;
+ generic: number;
+ duplicate: number;
+ warnings: number;
+ derived: boolean;
+};
+
+async function readDuration(videoDir: string): Promise<number> {
+ try {
+ const raw = await readFile(
+ path.join(videoDir, "transcript.cues.json"),
+ "utf8",
+ );
+ const j = JSON.parse(raw);
+ if (typeof j?.duration === "number" && j.duration > 0) return j.duration;
+ const cues = Array.isArray(j?.cues) ? j.cues : [];
+ return cues.length > 0 ? Number(cues[cues.length - 1]?.end ?? 0) : 0;
+ } catch {
+ return 0;
+ }
+}
+
+function scoreVideo(
+ slug: string,
+ record: DigestRecord,
+ durationSeconds: number,
+): VideoScore {
+ const section = record.sections.chapters;
+ const items = section?.items ?? [];
+ const p = section?.provenance;
+
+ // MAX COVERAGE GAP, including the leading and trailing gaps — a video whose
+ // chapters all sit in the last 20 minutes has a coverage failure that
+ // consecutive-gap-only scoring hides.
+ const starts = items.map((c) => c.start).sort((a, b) => a - b);
+ const bounds = [0, ...starts, durationSeconds];
+ let maxGap = 0;
+ for (let i = 1; i < bounds.length; i++) {
+ maxGap = Math.max(maxGap, bounds[i] - bounds[i - 1]);
+ }
+
+ let generic = 0;
+ let duplicate = 0;
+ const seen = new Set<string>();
+ for (const c of items) {
+ if (isGenericTitle(c.title)) generic++;
+ const n = normalizeTitle(c.title);
+ if (seen.has(n)) duplicate++;
+ seen.add(n);
+ }
+
+ return {
+ slug,
+ durationSeconds,
+ chunks: p?.chunks ?? 0,
+ chunksOk: p?.chunksOk ?? 0,
+ kept: items.length,
+ maxGapSeconds: durationSeconds > 0 ? maxGap : 0,
+ generic,
+ duplicate,
+ warnings: record.warnings.length,
+ derived: record.derivedFrom != null,
+ };
+}
+
+async function main(): Promise<void> {
+ const channels = process.argv.slice(2);
+ if (channels.length === 0) {
+ console.error("usage: digest-validate.ts <channel> [<channel> ...]");
+ process.exit(1);
+ }
+ const paths = getPaths();
+ const videos: VideoScore[] = [];
+ const byCode = new Map<DigestWarningCode, number>();
+ const provenanceSeen = new Map<string, number>();
+
+ for (const channelSlug of channels) {
+ const dataDir = path.join(paths.channelsDir, channelSlug, "data");
+ const ids = await readdir(dataDir).catch(() => [] as string[]);
+ for (const id of ids) {
+ const videoDir = path.join(dataDir, id);
+ const record = await loadDigest(videoDir);
+ if (!record) continue;
+ const duration = await readDuration(videoDir);
+ videos.push(scoreVideo(`${channelSlug}/${id}`, record, duration));
+ for (const w of record.warnings) {
+ byCode.set(w.code, (byCode.get(w.code) ?? 0) + 1);
+ }
+ const p = record.sections.chapters?.provenance;
+ if (p) {
+ const key = `${p.appId}/${p.modelRequested ?? p.model} v${p.promptVersion}${
+ p.promptVariant ? `+${p.promptVariant}` : ""
+ }`;
+ provenanceSeen.set(key, (provenanceSeen.get(key) ?? 0) + 1);
+ }
+ }
+ }
+
+ if (videos.length === 0) {
+ console.log("No digests found.");
+ return;
+ }
+
+ // Videos whose digest was SHARED from a cluster canonical are excluded from
+ // the quality metrics: they measure the canonical's generation, not this
+ // video's, and counting them twice would flatter whichever number they help.
+ const native = videos.filter((v) => !v.derived);
+ const sum = (f: (v: VideoScore) => number): number =>
+ native.reduce((a, v) => a + f(v), 0);
+ const chunks = sum((v) => v.chunks);
+ const chunksOk = sum((v) => v.chunksOk);
+ const zeroYield = chunks - chunksOk;
+ const kept = sum((v) => v.kept);
+ const audioHours = sum((v) => v.durationSeconds) / 3600;
+ const warningsTotal = sum((v) => v.warnings);
+ const pct = (n: number): string => `${(n * 100).toFixed(1)}%`;
+
+ console.log(`Digests scored: ${videos.length} (${native.length} natively generated, ${videos.length - native.length} shared from a duplicate)`);
+ console.log(`Audio hours (native): ${audioHours.toFixed(1)}`);
+ console.log("");
+ console.log(`Zero-yield chunks: ${zeroYield}/${chunks} (${chunks > 0 ? pct(zeroYield / chunks) : "n/a"})`);
+ console.log(`Chapters/hour: ${audioHours > 0 ? (kept / audioHours).toFixed(2) : "n/a"} (${kept} kept)`);
+ console.log(`Generic titles: ${sum((v) => v.generic)}/${kept} (${kept > 0 ? pct(sum((v) => v.generic) / kept) : "n/a"})`);
+ console.log(`Duplicate titles: ${sum((v) => v.duplicate)}/${kept} (${kept > 0 ? pct(sum((v) => v.duplicate) / kept) : "n/a"})`);
+ console.log(`Warnings recorded: ${warningsTotal}`);
+ const rejected = warningsTotal;
+ console.log(`Rejection rate: ${kept + rejected > 0 ? pct(rejected / (kept + rejected)) : "n/a"} (rejected / (kept + rejected))`);
+
+ const gaps = native
+ .filter((v) => v.durationSeconds > 0)
+ .sort((a, b) => b.maxGapSeconds - a.maxGapSeconds);
+ console.log(
+ `Worst coverage gap: ${gaps[0] ? `${toHms(gaps[0].maxGapSeconds)} (${gaps[0].slug}, ${toHms(gaps[0].durationSeconds)} long)` : "n/a"}`,
+ );
+ const medianGap = gaps.length > 0 ? gaps[Math.floor(gaps.length / 2)].maxGapSeconds : 0;
+ console.log(`Median coverage gap: ${toHms(medianGap)}`);
+
+ console.log("");
+ console.log("Rejections by guard:");
+ if (byCode.size === 0) console.log(" (none)");
+ for (const [code, n] of [...byCode.entries()].sort((a, b) => b[1] - a[1])) {
+ console.log(` ${code.padEnd(20)} ${n}`);
+ }
+
+ console.log("");
+ console.log("Provenance seen (this is what freshness compares):");
+ for (const [key, n] of [...provenanceSeen.entries()].sort((a, b) => b[1] - a[1])) {
+ console.log(` ${key} ×${n}`);
+ }
+
+ console.log("");
+ console.log("Worst 8 by coverage gap:");
+ for (const v of gaps.slice(0, 8)) {
+ console.log(
+ ` ${toHms(v.maxGapSeconds).padStart(9)} of ${toHms(v.durationSeconds).padStart(9)} ${String(v.kept).padStart(3)} ch ${v.slug}`,
+ );
+ }
+
+ const zeroChapter = native.filter((v) => v.kept === 0);
+ if (zeroChapter.length > 0) {
+ console.log("");
+ console.log(`Videos with NO chapters at all: ${zeroChapter.length}`);
+ for (const v of zeroChapter.slice(0, 10)) {
+ console.log(` ${v.slug} (${toHms(v.durationSeconds)})`);
+ }
+ }
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/common/components/StreamActionLog.tsx b/common/components/StreamActionLog.tsx
@@ -45,6 +45,24 @@ export function StreamActionLog({
}: Props) {
const accessibleName = label ?? buttonLabel;
const router = useRouter();
+ // A click before hydration is SILENTLY DISCARDED — the server-rendered button
+ // has no handler yet, so nothing fires: no request, no job, no log, no error.
+ // On a heavy page that is a long window (the 39-video channel page takes
+ // seconds to hydrate even in production, minutes under a dev server), and it
+ // is the whole of STATE.md's "the pilot's un-created job": clicking "Digest
+ // channel" produced no job record and an empty log, with no dedupe guard in
+ // runManagedFunction to explain it — because the action was never invoked at
+ // all. Reproduced against the real corpus: the first click fires no POST, a
+ // second click a few seconds later starts the job normally.
+ //
+ // Disabling until mounted makes the swallowed click impossible rather than
+ // merely unlikely, and it turns the failure from "nothing happened" into a
+ // button that is visibly not ready yet. It also fixes the same latent race for
+ // every other action built on this component (sync, download, transcribe),
+ // and makes e2e clicks WAIT for enablement instead of losing the click — see
+ // the pre-hydration lost-click that made the charts specs flaky.
+ const [hydrated, setHydrated] = useState(false);
+ useEffect(() => setHydrated(true), []);
const [running, setRunning] = useState(false);
const [log, setLog] = useState("");
const [error, setError] = useState<string | null>(null);
@@ -163,7 +181,7 @@ export function StreamActionLog({
type="button"
size="sm"
onClick={handleClick}
- disabled={running || disabled}
+ disabled={!hydrated || running || disabled}
>
{running ? runningLabel : buttonLabel}
</Button>