#!/usr/bin/env tsx // Score digests that ALREADY EXIST on disk, using the same metric definitions as // the bake-off harness (plans/tools/digest-bakeoff.ts). // // The bake-off scores IN MEMORY and writes no digests at all; this reads what a // real sweep wrote to disk. That is the difference between "how does this // configuration behave on a hand-picked sample" and "how did it behave on the // corpus", and the second is what a validation run is for. // // (An earlier version of this comment said the bake-off "generates into a scratch // directory and scores what it generated". It does not, and never did.) // // Usage: pnpm exec tsx bin/digest-validate.ts [ ...] import path from "node:path"; import { readdir, readFile } from "node:fs/promises"; import { getPaths } from "../lib/paths"; import { loadDigest } from "../lib/digest-server"; import { toHms } from "../lib/digestPrompt"; import type { DigestRecord, DigestWarningCode } from "../lib/digest"; // Copied verbatim from digest-bakeoff.ts so the two are comparable. A model that // segments correctly but names every section "Discussion" has produced a table // of contents nobody can navigate. const GENERIC_TITLE_RE = /^(the\s+)?(intro(duction)?|outro|conclusion|discussion|continued|continuation|overview|summary|recap|closing( remarks)?|opening( remarks)?|final thoughts|misc(ellaneous)?|other|general|topics?|segment|section|chapter|part)\b/i; const GENERIC_TITLE_TAIL_RE = /\b(part|section|segment|chapter)\s+(\d+|one|two|three|four|five|six|seven|eight|nine|ten)$/i; function isGenericTitle(title: string): boolean { const t = title.trim(); return GENERIC_TITLE_RE.test(t) || GENERIC_TITLE_TAIL_RE.test(t); } function normalizeTitle(title: string): string { return title.trim().toLowerCase().replace(/\s+/g, " "); } // The tag equivalent of isGenericTitle, and deliberately a SEPARATE regex rather // than a reuse of it: a generic chapter title and a useless tag are different // failures. "Introduction" is a bad chapter name; the tags that make a tag corpus // worthless are the ones that describe the MEDIUM or the FORMAT rather than the // subject ("video", "podcast", "commentary", "discussion"), because they match // every video and so partition nothing. // // Anchored whole-string, unlike the chapter version's prefix match: a tag is a // short noun phrase, so "video production" is a real topic and must not be // condemned by the leading word — the first live run produced exactly that tag. const GENERIC_TAG_RE = /^(the\s+)?(video|videos|vid|clip|clips|stream|streams|livestream|vod|podcast|episode|show|channel|content|media|footage|recording|broadcast|commentary|discussion|conversation|talk|talking|chat|chatting|interview|review|reaction|reacting|opinion|opinions|thoughts|rant|update|updates|news|misc|miscellaneous|other|general|various|topics?|entertainment|social media|youtube|twitch)$/i; function isGenericTag(tag: string): boolean { return GENERIC_TAG_RE.test(tag.trim()); } function normalizeTag(tag: string): string { return tag.trim().toLowerCase().replace(/\s+/g, " "); } type VideoScore = { slug: string; durationSeconds: number; chunks: number; chunksOk: number; kept: number; maxGapSeconds: number; generic: number; duplicate: number; warnings: number; derived: boolean; // Tags. Kept on the same row as the chapter metrics so one pass over the corpus // scores both, and so `derived` excludes shared digests from both alike. // // `hasTags` distinguishes "this video has no tags section" from "it has one that // came back empty" — before tags were ever generated the whole corpus was the // first case, and a metric that could not tell them apart reported a 0% failure // rate on a capability that had never run once. hasTags: boolean; tagChunks: number; tagChunksOk: number; tags: number; tagGeneric: number; tagDuplicate: number; tagWarnings: number; }; async function readDuration(videoDir: string): Promise { try { const raw = await readFile( path.join(videoDir, "transcript.cues.json"), "utf8", ); const j = JSON.parse(raw); if (typeof j?.duration === "number" && j.duration > 0) return j.duration; const cues = Array.isArray(j?.cues) ? j.cues : []; return cues.length > 0 ? Number(cues[cues.length - 1]?.end ?? 0) : 0; } catch { return 0; } } function scoreVideo( slug: string, record: DigestRecord, durationSeconds: number, ): VideoScore { const section = record.sections.chapters; const items = section?.items ?? []; const p = section?.provenance; // MAX COVERAGE GAP, including the leading and trailing gaps — a video whose // chapters all sit in the last 20 minutes has a coverage failure that // consecutive-gap-only scoring hides. const starts = items.map((c) => c.start).sort((a, b) => a - b); const bounds = [0, ...starts, durationSeconds]; let maxGap = 0; for (let i = 1; i < bounds.length; i++) { maxGap = Math.max(maxGap, bounds[i] - bounds[i - 1]); } let generic = 0; let duplicate = 0; const seen = new Set(); for (const c of items) { if (isGenericTitle(c.title)) generic++; const n = normalizeTitle(c.title); if (seen.has(n)) duplicate++; seen.add(n); } // Tags, same accumulate-then-sum shape. parseTags already de-dups across the // chunk overlap, so a duplicate surviving to disk is a PARSER failure rather // than the expected overlap noise — which is why it is counted at all. const tagSection = record.sections.tags; const tagItems = tagSection?.items ?? []; const tp = tagSection?.provenance; let tagGeneric = 0; let tagDuplicate = 0; const tagsSeen = new Set(); for (const t of tagItems) { if (isGenericTag(t.tag)) tagGeneric++; const n = normalizeTag(t.tag); if (tagsSeen.has(n)) tagDuplicate++; tagsSeen.add(n); } return { slug, durationSeconds, chunks: p?.chunks ?? 0, chunksOk: p?.chunksOk ?? 0, kept: items.length, maxGapSeconds: durationSeconds > 0 ? maxGap : 0, generic, duplicate, warnings: record.warnings.length, derived: record.derivedFrom != null, hasTags: tagSection != null, tagChunks: tp?.chunks ?? 0, tagChunksOk: tp?.chunksOk ?? 0, tags: tagItems.length, tagGeneric, tagDuplicate, tagWarnings: record.warnings.filter((w) => w.section === "tags").length, }; } async function main(): Promise { const channels = process.argv.slice(2); if (channels.length === 0) { console.error("usage: digest-validate.ts [ ...]"); process.exit(1); } const paths = getPaths(); const videos: VideoScore[] = []; const byCode = new Map(); const provenanceSeen = new Map(); // Tag vocabulary, keyed the same way VideoScore.slug is, so the cross-video // reuse metric can be computed after the shared-digest filter has been applied. const tagsBySlug = new Map(); for (const channelSlug of channels) { const dataDir = path.join(paths.channelsDir, channelSlug, "data"); const ids = await readdir(dataDir).catch(() => [] as string[]); for (const id of ids) { const videoDir = path.join(dataDir, id); const record = await loadDigest(videoDir); if (!record) continue; const duration = await readDuration(videoDir); const slug = `${channelSlug}/${id}`; videos.push(scoreVideo(slug, record, duration)); tagsBySlug.set( slug, (record.sections.tags?.items ?? []).map((t) => normalizeTag(t.tag)), ); for (const w of record.warnings) { byCode.set(w.code, (byCode.get(w.code) ?? 0) + 1); } const p = record.sections.chapters?.provenance; if (p) { const key = `${p.appId}/${p.modelRequested ?? p.model} v${p.promptVersion}${ p.promptVariant ? `+${p.promptVariant}` : "" }`; provenanceSeen.set(key, (provenanceSeen.get(key) ?? 0) + 1); } } } if (videos.length === 0) { console.log("No digests found."); return; } // Videos whose digest was SHARED from a cluster canonical are excluded from // the quality metrics: they measure the canonical's generation, not this // video's, and counting them twice would flatter whichever number they help. const native = videos.filter((v) => !v.derived); const sum = (f: (v: VideoScore) => number): number => native.reduce((a, v) => a + f(v), 0); const chunks = sum((v) => v.chunks); const chunksOk = sum((v) => v.chunksOk); const zeroYield = chunks - chunksOk; const kept = sum((v) => v.kept); const audioHours = sum((v) => v.durationSeconds) / 3600; const warningsTotal = sum((v) => v.warnings); const pct = (n: number): string => `${(n * 100).toFixed(1)}%`; console.log(`Digests scored: ${videos.length} (${native.length} natively generated, ${videos.length - native.length} shared from a duplicate)`); console.log(`Audio hours (native): ${audioHours.toFixed(1)}`); console.log(""); console.log(`Zero-yield chunks: ${zeroYield}/${chunks} (${chunks > 0 ? pct(zeroYield / chunks) : "n/a"})`); console.log(`Chapters/hour: ${audioHours > 0 ? (kept / audioHours).toFixed(2) : "n/a"} (${kept} kept)`); console.log(`Generic titles: ${sum((v) => v.generic)}/${kept} (${kept > 0 ? pct(sum((v) => v.generic) / kept) : "n/a"})`); console.log(`Duplicate titles: ${sum((v) => v.duplicate)}/${kept} (${kept > 0 ? pct(sum((v) => v.duplicate) / kept) : "n/a"})`); console.log(`Warnings recorded: ${warningsTotal}`); const rejected = warningsTotal; console.log(`Rejection rate: ${kept + rejected > 0 ? pct(rejected / (kept + rejected)) : "n/a"} (rejected / (kept + rejected))`); const gaps = native .filter((v) => v.durationSeconds > 0) .sort((a, b) => b.maxGapSeconds - a.maxGapSeconds); console.log( `Worst coverage gap: ${gaps[0] ? `${toHms(gaps[0].maxGapSeconds)} (${gaps[0].slug}, ${toHms(gaps[0].durationSeconds)} long)` : "n/a"}`, ); const medianGap = gaps.length > 0 ? gaps[Math.floor(gaps.length / 2)].maxGapSeconds : 0; console.log(`Median coverage gap: ${toHms(medianGap)}`); // ---- Tags ------------------------------------------------------------- // // Reported unconditionally, INCLUDING the "not generated" case. Tags were a // shipped capability for months with zero coverage — parseTags, a schema, a // digestVideo branch and a settings checkbox all existed and 0 of 109 sidecars // had ever contained a tags section. A metrics block that appears only when // there is something to measure is exactly how that stays invisible. console.log(""); const withTags = native.filter((v) => v.hasTags); if (withTags.length === 0) { console.log( `Tags: NOT GENERATED for any of these ${native.length} video(s). ` + `Add "tags" to settings.digest.sections to generate them ` + `(measured surcharge in the same pass: 5–15%).`, ); } else { const tagSum = (f: (v: VideoScore) => number): number => withTags.reduce((a, v) => a + f(v), 0); const tags = tagSum((v) => v.tags); const tagChunks = tagSum((v) => v.tagChunks); const tagChunksOk = tagSum((v) => v.tagChunksOk); const tagZeroYield = tagChunks - tagChunksOk; const emptySection = withTags.filter((v) => v.tags === 0).length; console.log( `Tags: ${withTags.length}/${native.length} native video(s) have a tags section.`, ); console.log( `Zero-yield tag chunks: ${tagZeroYield}/${tagChunks} (${tagChunks > 0 ? pct(tagZeroYield / tagChunks) : "n/a"})`, ); console.log( `Tags per video: ${(tags / withTags.length).toFixed(2)} mean (${tags} total, ` + `min ${Math.min(...withTags.map((v) => v.tags))}, max ${Math.max(...withTags.map((v) => v.tags))})`, ); console.log( `Empty tag sections: ${emptySection}/${withTags.length} (${pct(emptySection / withTags.length)})`, ); console.log( `Generic tags: ${tagSum((v) => v.tagGeneric)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagGeneric) / tags) : "n/a"})`, ); console.log( `Duplicate tags: ${tagSum((v) => v.tagDuplicate)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagDuplicate) / tags) : "n/a"}) — parseTags de-dups, so any of these is a parser bug`, ); console.log(`Tag warnings: ${tagSum((v) => v.tagWarnings)}`); // Vocabulary reuse is the whole point of tags: a tag used once is a label, a // tag used across videos is an index. A corpus where every tag is unique has // produced nothing searchable, and no per-video metric above can see that. const freq = new Map(); for (const v of withTags) { // One count per VIDEO, not per occurrence: a tag repeated inside one video // is already de-duplicated by parseTags, so counting occurrences would just // re-measure that. for (const t of new Set(tagsBySlug.get(v.slug) ?? [])) { freq.set(t, (freq.get(t) ?? 0) + 1); } } const reused = [...freq.values()].filter((n) => n > 1).length; console.log( `Distinct tags: ${freq.size} across ${withTags.length} video(s); ` + `${reused} (${freq.size > 0 ? pct(reused / freq.size) : "n/a"}) appear in more than one video`, ); const top = [...freq.entries()] .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])) .slice(0, 12); if (top.length > 0) { console.log("Most reused tags:"); for (const [tag, n] of top) console.log(` ${String(n).padStart(4)}× ${tag}`); } } console.log(""); console.log("Rejections by guard:"); if (byCode.size === 0) console.log(" (none)"); for (const [code, n] of [...byCode.entries()].sort((a, b) => b[1] - a[1])) { console.log(` ${code.padEnd(20)} ${n}`); } console.log(""); console.log("Provenance seen (this is what freshness compares):"); for (const [key, n] of [...provenanceSeen.entries()].sort((a, b) => b[1] - a[1])) { console.log(` ${key} ×${n}`); } console.log(""); console.log("Worst 8 by coverage gap:"); for (const v of gaps.slice(0, 8)) { console.log( ` ${toHms(v.maxGapSeconds).padStart(9)} of ${toHms(v.durationSeconds).padStart(9)} ${String(v.kept).padStart(3)} ch ${v.slug}`, ); } const zeroChapter = native.filter((v) => v.kept === 0); if (zeroChapter.length > 0) { console.log(""); console.log(`Videos with NO chapters at all: ${zeroChapter.length}`); for (const v of zeroChapter.slice(0, 10)) { console.log(` ${v.slug} (${toHms(v.durationSeconds)})`); } } } main().catch((err) => { console.error(err); process.exit(1); });