commit 9c60e59f976f6b3fd878349008f48427d1f804fa
parent 5d880ad784668dc187f9d376a07095c1b757ff6f
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Mon, 7 Sep 2026 15:58:24 -0400
repo: the dead weight goes, and the bake-off harnesses move beside their results
Deleted, each confirmed unreferenced by grep over the tree (node_modules and
worktrees excluded):
sites/ one file, sites/jeralyzer/site.json.example. The
real per-site config is transcripts/sites/<id>/
site.json; nothing reads this and nothing ever did.
(editor/e2e's "test-transcripts/sites/jeralyzer/
site.json" is the fixture corpus, not this.)
common/bin/migrate-to-sites.ts a one-shot CLI for the single-site -> sites/
migration, done on all six sites. The controller
(controller/migrateToSites.ts) STAYS: the Sites
page's Migrate button calls it.
create-archives.sh tail ~8 lines of commented-out transcript-archive
commands, dead since the corpus outgrew
Cloudflare's 25 MB per-asset limit.
export/.compose-cache/ 28 KB, gitignored, regenerated by compose.
export/.r2-staging/ (2.5 GB of staged zips) is
DELIBERATELY untouched — that is the operator's.
Moved to plans/tools/, beside the reports they produced in plans/bakeoff/ and
plans/attribution-pilot/: digest-bakeoff.ts (1425), attribution-bakeoff.ts
(1281), attribution-pilot.ts (1122). 3,828 lines, 58% of common/bin/, none of
it part of the product. Their `../lib/...` imports became `../../common/lib/...`.
One new file was needed: common/bin/_lmdb.ts, a three-line re-export. A bare
`import { open } from "lmdb"` resolves from the IMPORTING file upward, and
plans/ is not a workspace package — it has no node_modules and the root has
only tsx, so digest-bakeoff could no longer resolve it. It goes through
common/bin/ (beside _parseFlags.ts, which these three already import) rather
than common/lib/, because it is bin plumbing and not library surface.
`PARALLEL_TRANSCRIBE_LIMIT` removed from SETUP.md, README.md and
homepage/content/docs/install.md. Zero readers in the tree — settings.json's
`parallelTranscribeLimit` replaced it, as editor/CHANGELOG.md:244 records. The
CHANGELOG entries for it and for the migrate-to-sites CLI are left alone: they
are a dated record of what shipped, not instructions.
Path references updated in plans/{FACTS,STATE,digest-context-review}.md,
plans/bakeoff/*.md, and four code comments (common/lib/{digest,engineTiming,
attributionScore}.ts, common/bin/{diarize-backfill,digest-validate}.ts).
FACTS.md's `:70-75` line anchor still holds — only import lines changed, the
line count did not.
Verified: `tsc --noEmit -p plans/tools` (new minimal tsconfig extending
tsconfig.base.json) clean for all three. All three load and start under
`cd common && pnpm exec tsx ../plans/tools/<name>.ts`, which proves the
relative-import and lmdb resolution end to end; the runs were killed
immediately and their one stray artifact removed — nothing was written to
transcripts/. `tsc --noEmit` clean in all six packages; common 876/876.
Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>
Diffstat:
23 files changed, 3865 insertions(+), 3920 deletions(-)
diff --git a/README.md b/README.md
@@ -444,7 +444,6 @@ by environment variables:
| `YTDLP_BIN` | `yt-dlp` on PATH | The downloader. |
| `WHISPER_BIN` / `WHISPER_MODEL` | `whisper-cli` on PATH | Default transcription backend and its model. |
| `FFMPEG_BIN` / `FFPROBE_BIN` | on PATH | Transcode and duration checks. |
-| `PARALLEL_TRANSCRIBE_LIMIT` | `4` | Max simultaneous transcription jobs. |
Everything else lives in the editor's **Settings** page and is optional — a missing or
partial settings file falls back to defaults. Full list in
diff --git a/SETUP.md b/SETUP.md
@@ -273,7 +273,6 @@ any of them via environment variables before launching:
| `PARAKEET_CLI` / `PARAKEET_MODEL` / `PARAKEET_STITCH_BIN` | `parakeet-cli` / — / `scripts/parakeet-stitch.mjs` | parakeet.cpp CLI, model, and wrapper. |
| `FFMPEG_BIN` / `FFPROBE_BIN` | `ffmpeg` / `ffprobe` (PATH) | Audio transcode + duration checks. |
| `RSYNC_BIN` | `rsync` (PATH) | Saved-video backup. |
-| `PARALLEL_TRANSCRIBE_LIMIT` | `4` | Max parallel transcription jobs. |
| `WORKER_TOKEN` | — | Bearer token for the remote-worker transcription API (set on both ends when used). |
Feature-area docs cover their own env vars: [SCHEDULED_SYNC.md](SCHEDULED_SYNC.md)
diff --git a/common/bin/_lmdb.ts b/common/bin/_lmdb.ts
@@ -0,0 +1,12 @@
+// The LMDB binding, re-exported for scripts that live OUTSIDE this package.
+//
+// plans/tools/*.ts (the one-off bake-off harnesses, parked beside their results
+// in plans/bakeoff/) import common by relative path. That works for common's
+// own modules, but a bare `import { open } from "lmdb"` resolves from the
+// IMPORTING file's directory upward — and plans/ is not a workspace package, so
+// it has no node_modules and the root has only tsx. Going through this file
+// resolves `lmdb` from common/node_modules, where it is a declared dependency.
+//
+// Beside _parseFlags.ts rather than in lib/ on purpose: it is bin plumbing, not
+// library surface.
+export { open } from "lmdb";
diff --git a/common/bin/attribution-bakeoff.ts b/common/bin/attribution-bakeoff.ts
@@ -1,1281 +0,0 @@
-#!/usr/bin/env tsx
-// Score attribution PROMPT VARIANTS against a fixed pair of videos, without
-// writing a single sidecar.
-//
-// WHY THIS EXISTS SEPARATELY FROM attribution-pilot.ts, which measures the same
-// lane. The pilot calls attributeOneVideo because the SHIPPED PATH was what was
-// under test — the freshness skip, the downgrade guard, the provenance, the
-// sidecar a human then reads. This is the opposite situation: the thing under
-// test is a prompt and a schema that have won nothing yet, so it drives
-// attributionPrompt + attributionTurns + the digest app DIRECTLY and scores in
-// memory. That is digest-bakeoff.ts's defining property and it is inherited
-// deliberately: a losing variant leaves NOTHING in the corpus and no freshness
-// record has to be invalidated to undo a round.
-//
-// find transcripts/channels -name attribution.json | wc -l
-//
-// is 9 before and after, and this harness checks that itself — see the sidecar
-// census at the end of main(). writeAttribution is never imported.
-//
-// THE QUESTION ROUND 2 ASKS. The 2026-08-08 pilot rejected the text-only lane on
-// both pre-stated thresholds: 14.70 new labels per chunk against a <= 0.3 bar,
-// and 66.2 s/chunk engine against the digest lane's 11.2. The per-chunk data
-// says the cause is NOT the knownSpeakers roster the write-up blamed — on
-// destiny/5nmDzKB23OU 15 of 16 labels first appear in chunk 0, where no roster
-// exists, and on rekietalaw/EsZhaCfc8HQ the collapse is intermittent rather than
-// monotonic. Dozens of labels sit at exactly 60 characters, the schema's own
-// maxLength. The model was continuing the transcript into a free-string field.
-//
-// So the variable is the SCHEMA, and only the schema. Same engine (qwen2.5:7b
-// @ 8192), same chunker, same videos, same scorer, same metrics as the pilot —
-// lib/attributionScore.ts exists so that last one is true by construction rather
-// than by two files agreeing.
-//
-// UNIT DISCIPLINE, inherited and not negotiable. plans/STATE.md records a
-// RETRACTED throughput headline that came from pricing a sweep in
-// seconds-per-audio-hour when the unit of work is the CHUNK (density varies 4x
-// across this corpus). This reports seconds per CHUNK, prints chunks-per-
-// audio-hour beside any rate, and never prints a bare seconds-per-audio-hour.
-//
-// Examples:
-// tsx bin/attribution-bakeoff.ts --dry-run
-// tsx bin/attribution-bakeoff.ts --variant closed-cast --label round2-closed
-// tsx bin/attribution-bakeoff.ts --variant open --label round2-open-control
-// tsx bin/attribution-bakeoff.ts --variant closed-cast --model gemma2:9b \
-// --num-ctx 8192 --label round2-gemma
-
-import os from "node:os";
-import path from "node:path";
-import { mkdir, readdir, writeFile } from "node:fs/promises";
-import { getPaths, type Paths } from "../lib/paths";
-import { parseFlags } from "./_parseFlags";
-import {
- ATTRIBUTION_FILENAME,
- ATTRIBUTION_PROMPT_VERSION,
- type AttributionRecord,
- type AttributionWarning,
-} from "../lib/attribution";
-import type { AttributionSettings } from "../lib/settings";
-import { OLLAMA_DIGEST_APP_ID } from "../lib/digestApps";
-import type { DigestApp, DigestAppConfig } from "../lib/digestApps";
-import { resolveAttributionTarget } from "../controller/attributionTarget";
-import {
- ATTRIBUTION_CAST_SAMPLES,
- ATTRIBUTION_MAX_SPEAKERS,
- ATTRIBUTION_MAX_TURNS_PER_CHUNK,
- ATTRIBUTION_UNKNOWN_SPEAKER,
- CAST_SYSTEM_PROMPT,
- TURN_SYSTEM_PROMPT,
- buildCastPrompt,
- buildClosedTurnPrompt,
- buildTurnPrompt,
- castSchema,
- isUselessSpeakerLabel,
- selectTranscriptSamples,
- turnSchema,
- turnSchemaClosed,
-} from "../lib/attributionPrompt";
-import {
- assembleTurns,
- createSpeakerRoster,
- type SpeakerMark,
-} from "../lib/attributionTurns";
-import {
- chunkRangesFor,
- isTranscriptTextLabel,
- scoreRecord,
- transcriptHaystack,
- type VideoMetrics,
-} from "../lib/attributionScore";
-import { parseEngineCall, isUnparsedEngineLine, type CallTiming } from "../lib/engineTiming";
-import {
- DIGEST_OVERLAP_CUES,
- hmsToSeconds,
- maxCuesForContext,
- toHms,
-} from "../lib/digestPrompt";
-import { chunkCuesForContext } from "../lib/transcriptWindow";
-import { transcriptToMarkdown } from "../lib/transcriptToMarkdown";
-import { readDigestContext } from "../lib/digestContext-server";
-import {
- chunksPerAudioHour,
- readChunkCensus,
- sweepDays,
-} from "../controller/digestPlan";
-import {
- isCuesJsonFresh,
- readNormalizedTranscript,
-} from "../controller/normalizeTranscript";
-import type { Cue } from "../lib/vtt";
-
-// The digest lane's own cost on the SAME model and context — round 2's 190
-// engine-seconds over 17 chunks. The bar a second sweep has to justify itself
-// against, not a footnote.
-const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2;
-
-// The pilot's measured text-only numbers on these same videos, so a round-2
-// report states what it is beating without anyone opening the round-1 file.
-const PILOT_BASELINE = {
- newLabelsPerChunkMean: 14.7,
- secondsPerChunkEngine: 66.2,
- byVideo: {
- "destiny/5nmDzKB23OU": { chunks: 4, labels: 16, newLabelsPerChunk: 0.33 },
- "rekietalaw/EsZhaCfc8HQ": { chunks: 8, labels: 90, newLabelsPerChunk: 12.57 },
- } as Record<string, { chunks: number; labels: number; newLabelsPerChunk: number }>,
-};
-
-// THE PROBE PAIR, and why these two. Both have a measured round-1 baseline, and
-// between them they exhibit BOTH failure shapes: destiny collapses in chunk 0
-// where no roster exists, rekietalaw collapses intermittently (chunks 1, 3 and 6
-// bad; 0, 2, 4, 5, 7 clean). 12 chunk calls + 2 cast calls.
-const DEFAULT_VIDEOS = ["destiny/5nmDzKB23OU", "rekietalaw/EsZhaCfc8HQ"];
-
-// The thresholds, written BEFORE the run — see plans/STATE.md. A bar chosen
-// after seeing the numbers is not a bar.
-const THRESHOLDS = {
- newLabelsPerChunk: 0.3,
- secondsPerChunkEngine: 25,
- transcriptTextLabelRate: 0,
-};
-
-type Variant = "closed-cast" | "open";
-
-// ---------------------------------------------------------------------------
-// The sample
-// ---------------------------------------------------------------------------
-
-type ProbeVideo = {
- slug: string;
- channelSlug: string;
- videoDir: string;
- videoId: string;
- title: string;
- channel: string;
- durationSeconds: number;
- dir: string;
- cues: Cue[];
- chunks: Cue[][];
- chunkRanges: { start: number; end: number }[];
-};
-
-async function priceVideo(
- slug: string,
- paths: Paths,
- maxCues: number,
-): Promise<ProbeVideo | null> {
- const slash = slug.indexOf("/");
- if (slash <= 0) throw new Error(`--videos expects <channelSlug>/<videoDir>, got "${slug}"`);
- const channelSlug = slug.slice(0, slash);
- const videoDir = slug.slice(slash + 1);
- const dir = path.join(paths.channelsDir, channelSlug, "data", videoDir);
-
- const { fresh, cuesPath } = await isCuesJsonFresh(dir);
- if (!fresh) return null;
- const transcript = await readNormalizedTranscript(cuesPath);
- const cues = transcript?.cues ?? [];
- if (!transcript || cues.length === 0) return null;
-
- const chunks = chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES });
- const lastCue = cues[cues.length - 1];
- return {
- slug,
- channelSlug,
- videoDir,
- videoId: transcript.id || videoDir,
- title: transcript.title || videoDir,
- channel: transcript.channel || channelSlug,
- durationSeconds: Math.round(transcript.duration || lastCue.end || lastCue.start || 0),
- dir,
- cues,
- chunks,
- // Derived by the SHARED helper, so the scorer's chunk indexing cannot drift
- // from the chunker this run actually used.
- chunkRanges: chunkRangesFor(cues, maxCues),
- };
-}
-
-// Render one chunk exactly as attributeOne.ts does — ABSOLUTE timestamps,
-// because an attribution mark lands on one shared timeline that later chunks
-// keep adding to. Re-basing per chunk would make every chunk's marks collide.
-function renderChunk(video: ProbeVideo, cues: Cue[]): string {
- return transcriptToMarkdown(
- {
- id: video.videoId,
- title: video.title,
- channel: video.channel,
- duration: video.durationSeconds,
- cues,
- },
- {
- timestamps: true,
- includeDescription: false,
- includeTags: false,
- stampForCue: (_clock, seconds) => toHms(Math.max(0, seconds)),
- },
- );
-}
-
-// ---------------------------------------------------------------------------
-// The two variants
-// ---------------------------------------------------------------------------
-
-type CastMember = { name: string; confidence: number };
-
-type VariantRun = {
- record: AttributionRecord;
- warnings: AttributionWarning[];
- chunksOk: number;
- // closed-cast only
- cast?: CastMember[];
- castRejected?: string[];
- castCallSeconds?: number;
- unknownTurns?: number;
- offCastTurns?: number;
- // Turns whose timestamp fell outside the chunk's own range. Counted, because
- // they are DROPPED silently and a run that produced nothing but these would
- // otherwise be indistinguishable from a model that returned nothing at all.
- outOfRangeTurns: number;
- totalTurnsAccepted: number;
-};
-
-type RunCtx = {
- app: DigestApp;
- config: DigestAppConfig;
- contextNote?: string;
- log: (m: string) => void;
-};
-
-function emptyRecord(
- video: ProbeVideo,
- model: string,
- chunks: number,
- chunksOk: number,
- warnings: AttributionWarning[],
-): AttributionRecord {
- return {
- videoId: video.videoId,
- generatedAt: new Date().toISOString(),
- speakers: [],
- segments: [],
- ...(warnings.length ? { warnings } : {}),
- provenance: {
- method: "text-only",
- appId: "bakeoff",
- model,
- modelRequested: model,
- // NOT the shipped ATTRIBUTION_PROMPT_VERSION. This record never reaches
- // disk, and stamping it with the live version would be a lie waiting for
- // someone to copy it somewhere that matters.
- promptVersion: -1,
- generatedAt: new Date().toISOString(),
- chunks,
- chunksOk,
- durationMs: 0,
- },
- };
-}
-
-// PASS A — cast discovery. ONE call for the whole video.
-async function discoverCast(
- video: ProbeVideo,
- ctx: RunCtx,
-): Promise<{ cast: CastMember[]; rejected: string[]; seconds: number; model: string }> {
- const samples = selectTranscriptSamples(video.cues, ATTRIBUTION_CAST_SAMPLES);
- ctx.log(
- ` ${video.slug}: cast discovery over ${samples.length} evenly-spaced excerpt(s), 1 call.`,
- );
- const started = Date.now();
- const result = await ctx.app.run({
- config: ctx.config,
- system: CAST_SYSTEM_PROMPT,
- prompt: buildCastPrompt({
- title: video.title,
- channel: video.channel,
- samples,
- ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
- }),
- schema: castSchema(),
- onLog: ctx.log,
- });
- const seconds = (Date.now() - started) / 1000;
-
- const haystack = transcriptHaystack(video.cues);
- const cast: CastMember[] = [];
- const rejected: string[] = [];
- const seen = new Set<string>();
- const raw = (result.data as { cast?: unknown })?.cast;
- if (Array.isArray(raw)) {
- for (const item of raw) {
- const e = item as { name?: unknown; confidence?: unknown };
- if (typeof e?.name !== "string") continue;
- const name = e.name.trim();
- // THE SAME GUARDS THE RECORD GETS, applied to the cast — pass A uses a
- // free string, so it can fail the exact way pass B is being fixed for. A
- // transcript fragment admitted here would be minted into the enum and
- // become a legitimate label for the rest of the video.
- if (isUselessSpeakerLabel(name) || isTranscriptTextLabel(name, haystack)) {
- rejected.push(name);
- continue;
- }
- const key = name.toLowerCase();
- if (seen.has(key)) continue;
- seen.add(key);
- cast.push({
- name,
- confidence:
- typeof e.confidence === "number" && Number.isFinite(e.confidence)
- ? Math.min(1, Math.max(0, e.confidence))
- : 0,
- });
- if (cast.length >= ATTRIBUTION_MAX_SPEAKERS) break;
- }
- }
- return { cast, rejected, seconds, model: result.model };
-}
-
-// PASS B — turn assignment, per chunk, against a CLOSED label space.
-async function runClosedCast(video: ProbeVideo, ctx: RunCtx): Promise<VariantRun> {
- const warnings: AttributionWarning[] = [];
- const { cast, rejected, seconds: castCallSeconds, model: castModel } =
- await discoverCast(video, ctx);
-
- ctx.log(
- ` ${video.slug}: cast = ${
- cast.map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "(none)"
- }${rejected.length ? ` — rejected ${rejected.length}` : ""}`,
- );
-
- if (cast.length === 0) {
- // Pass A returning nothing usable is a RESULT, not a crash: the plan's third
- // decision row ("Pass A itself returns garbage") escalates to gemma2:9b.
- warnings.push({ code: "empty", detail: "cast discovery produced no usable name" });
- return {
- record: emptyRecord(video, castModel, video.chunks.length, 0, warnings),
- warnings,
- chunksOk: 0,
- cast,
- castRejected: rejected,
- castCallSeconds,
- outOfRangeTurns: 0,
- totalTurnsAccepted: 0,
- };
- }
-
- const names = cast.map((c) => c.name);
- const schema = turnSchemaClosed(names, ATTRIBUTION_MAX_TURNS_PER_CHUNK);
- // The roster is SEEDED with the whole cast and never grows. That is the point:
- // cross-chunk identity stops being something the model has to remember and
- // becomes a property of the grammar it decodes against.
- const roster = createSpeakerRoster(names);
- const marks: SpeakerMark[] = [];
- let model = castModel;
- let chunksOk = 0;
- let unknownTurns = 0;
- let offCastTurns = 0;
- let outOfRange = 0;
- let accepted = 0;
-
- for (let i = 0; i < video.chunks.length; i++) {
- const chunk = video.chunks[i];
- const startSeconds = Math.max(0, Math.floor(chunk[0].start));
- const endSeconds = Math.max(
- startSeconds,
- Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
- );
- try {
- const result = await ctx.app.run({
- config: ctx.config,
- system: TURN_SYSTEM_PROMPT,
- prompt: buildClosedTurnPrompt({
- title: video.title,
- channel: video.channel,
- startSeconds,
- endSeconds,
- transcript: renderChunk(video, chunk),
- cast: names,
- ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
- }),
- schema,
- onLog: ctx.log,
- });
- chunksOk++;
- model = result.model || model;
- const raw = (result.data as { turns?: unknown })?.turns;
- if (!Array.isArray(raw)) continue;
- for (const item of raw) {
- const e = item as { start?: unknown; speaker?: unknown };
- if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue;
- const at = hmsToSeconds(e.start);
- if (at === null) {
- warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) });
- continue;
- }
- if (at < startSeconds || at > endSeconds) {
- outOfRange++;
- continue;
- }
- if (e.speaker.trim() === ATTRIBUTION_UNKNOWN_SPEAKER) {
- // Counted, then dropped. An honest Unknown is a good answer and the
- // rate is worth reading, but it must not become a speaker.
- unknownTurns++;
- continue;
- }
- if (isUselessSpeakerLabel(e.speaker)) {
- unknownTurns++;
- continue;
- }
- // THE CHECK THAT MAKES THIS A MEASUREMENT. The enum pins `speaker`, and
- // the parser verifies it anyway — the belt-and-braces nameClusters
- // already applies to cluster indices. If a label outside the cast ever
- // lands here, the grammar did NOT hold and the whole hypothesis is
- // refuted; offCastTurns is that count, and it must be 0.
- const index = roster.lookup(e.speaker);
- if (index === undefined) {
- offCastTurns++;
- warnings.push({
- code: "unknown-cluster",
- chunk: i,
- detail: `off-cast label: ${e.speaker.slice(0, 60)}`,
- });
- continue;
- }
- marks.push({ at, speaker: index });
- accepted++;
- }
- } catch (err) {
- const message = (err as Error)?.message ?? String(err);
- warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) });
- ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`);
- }
- }
-
- const lastCue = video.cues[video.cues.length - 1];
- // Only cast members that actually SPOKE become speakers. A name pass A
- // proposed and pass B never used is not in the record — it would otherwise
- // score as a label at 0 seconds and inflate the label count for nothing.
- const used = [...new Set(marks.map((m) => m.speaker))].sort((a, b) => a - b);
- const remap = new Map(used.map((old, i) => [old, i]));
- const { speakers, segments } = assembleTurns({
- labels: used.map((i) => roster.labels[i]),
- marks: marks.map((m) => ({ at: m.at, speaker: remap.get(m.speaker)! })),
- transcriptEnd: lastCue.end || lastCue.start,
- });
-
- const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings);
- record.speakers = speakers;
- record.segments = segments;
- return {
- record,
- warnings,
- chunksOk,
- cast,
- castRejected: rejected,
- castCallSeconds,
- unknownTurns,
- offCastTurns,
- outOfRangeTurns: outOfRange,
- totalTurnsAccepted: accepted,
- };
-}
-
-// The OPEN control — the shipped prompt and schema, driven in memory. Reproduces
-// round 1 without writing a sidecar, so a round-2 report can state the delta
-// under identical box conditions instead of against a number measured last week.
-async function runOpen(video: ProbeVideo, ctx: RunCtx): Promise<VariantRun> {
- const warnings: AttributionWarning[] = [];
- const roster = createSpeakerRoster();
- const marks: SpeakerMark[] = [];
- let model = "";
- let chunksOk = 0;
- let accepted = 0;
- let outOfRange = 0;
-
- for (let i = 0; i < video.chunks.length; i++) {
- const chunk = video.chunks[i];
- const startSeconds = Math.max(0, Math.floor(chunk[0].start));
- const endSeconds = Math.max(
- startSeconds,
- Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
- );
- try {
- const result = await ctx.app.run({
- config: ctx.config,
- system: TURN_SYSTEM_PROMPT,
- prompt: buildTurnPrompt({
- title: video.title,
- channel: video.channel,
- startSeconds,
- endSeconds,
- transcript: renderChunk(video, chunk),
- ...(roster.labels.length > 0 ? { knownSpeakers: roster.labels.slice() } : {}),
- ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
- }),
- schema: turnSchema(),
- onLog: ctx.log,
- });
- chunksOk++;
- model = result.model || model;
- const raw = (result.data as { turns?: unknown })?.turns;
- if (!Array.isArray(raw)) continue;
- for (const item of raw) {
- const e = item as { start?: unknown; speaker?: unknown };
- if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue;
- const at = hmsToSeconds(e.start);
- if (at === null) {
- warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) });
- continue;
- }
- if (at < startSeconds || at > endSeconds) {
- outOfRange++;
- continue;
- }
- if (isUselessSpeakerLabel(e.speaker)) continue;
- marks.push({ at, speaker: roster.intern(e.speaker) });
- accepted++;
- }
- } catch (err) {
- const message = (err as Error)?.message ?? String(err);
- warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) });
- ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`);
- }
- }
-
- const lastCue = video.cues[video.cues.length - 1];
- const { speakers, segments } = assembleTurns({
- labels: roster.labels,
- marks,
- transcriptEnd: lastCue.end || lastCue.start,
- });
- const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings);
- record.speakers = speakers;
- record.segments = segments;
- return {
- record,
- warnings,
- chunksOk,
- outOfRangeTurns: outOfRange,
- totalTurnsAccepted: accepted,
- };
-}
-
-// ---------------------------------------------------------------------------
-// The run
-// ---------------------------------------------------------------------------
-
-type VideoResult = {
- slug: string;
- title: string;
- durationSeconds: number;
- chunks: number;
- chunksOk: number;
- wallSeconds: number;
- calls: CallTiming[];
- unparsedLogLines: number;
- warningsByCode: Record<string, number>;
- metrics: VideoMetrics | null;
- cast?: CastMember[];
- castRejected?: string[];
- castCallSeconds?: number;
- unknownTurns?: number;
- offCastTurns?: number;
- outOfRangeTurns: number;
- totalTurnsAccepted: number;
- speakers: { label: string; seconds: number }[];
-};
-
-function sum(v: number[]): number {
- return v.reduce((a, b) => a + b, 0);
-}
-
-function boxState(): Record<string, unknown> {
- return {
- loadavg: os.loadavg().map((n) => Number(n.toFixed(2))),
- freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)),
- totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)),
- };
-}
-
-// How many attribution.json files exist right now. THE SAFETY CLAIM of this
-// harness, checked rather than asserted in a comment.
-async function countSidecars(paths: Paths): Promise<number> {
- let total = 0;
- let channels: string[];
- try {
- channels = (await readdir(paths.channelsDir, { withFileTypes: true }))
- .filter((e) => e.isDirectory())
- .map((e) => e.name);
- } catch {
- return 0;
- }
- for (const channel of channels) {
- const dataDir = path.join(paths.channelsDir, channel, "data");
- let videos;
- try {
- videos = await readdir(dataDir, { withFileTypes: true });
- } catch {
- continue;
- }
- for (const v of videos) {
- if (!v.isDirectory()) continue;
- try {
- const files = await readdir(path.join(dataDir, v.name));
- if (files.includes(ATTRIBUTION_FILENAME)) total++;
- } catch {
- // A video dir that vanished mid-walk is not a sidecar.
- }
- }
- }
- return total;
-}
-
-async function runVideo(
- video: ProbeVideo,
- variant: Variant,
- ctx: Omit<RunCtx, "log">,
- log: (m: string) => void,
-): Promise<VideoResult> {
- const calls: CallTiming[] = [];
- let unparsed = 0;
- // TEES, and it has to. This one closure is both the engine-timing sink and the
- // harness's own progress channel — the cast list, the per-chunk failures. An
- // earlier version only parsed, which silently swallowed the cast line and made
- // a 0-label result impossible to diagnose from the console.
- const onLog = (line: string) => {
- const call = parseEngineCall(line);
- if (call) {
- calls.push(call);
- return;
- }
- if (isUnparsedEngineLine(line)) {
- unparsed++;
- return;
- }
- log(line);
- };
-
- const started = Date.now();
- const out =
- variant === "closed-cast"
- ? await runClosedCast(video, { ...ctx, log: onLog })
- : await runOpen(video, { ...ctx, log: onLog });
- const wallSeconds = (Date.now() - started) / 1000;
-
- const warningsByCode: Record<string, number> = {};
- for (const w of out.warnings) {
- warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1;
- }
-
- // Scored WITH the cues, so transcriptTextLabelRate is a number rather than
- // null — it is the metric this round exists to drive to zero.
- const metrics =
- out.record.speakers.length > 0
- ? scoreRecord(out.record, {
- chunkRanges: video.chunkRanges,
- durationSeconds: video.durationSeconds,
- cues: video.cues,
- })
- : null;
-
- const result: VideoResult = {
- slug: video.slug,
- title: video.title,
- durationSeconds: video.durationSeconds,
- chunks: video.chunks.length,
- chunksOk: out.chunksOk,
- wallSeconds,
- calls,
- unparsedLogLines: unparsed,
- warningsByCode,
- metrics,
- ...(out.cast ? { cast: out.cast } : {}),
- ...(out.castRejected ? { castRejected: out.castRejected } : {}),
- ...(out.castCallSeconds !== undefined ? { castCallSeconds: out.castCallSeconds } : {}),
- ...(out.unknownTurns !== undefined ? { unknownTurns: out.unknownTurns } : {}),
- ...(out.offCastTurns !== undefined ? { offCastTurns: out.offCastTurns } : {}),
- outOfRangeTurns: out.outOfRangeTurns,
- totalTurnsAccepted: out.totalTurnsAccepted,
- speakers: out.record.speakers.map((s) => ({ label: s.label, seconds: s.seconds ?? 0 })),
- };
-
- log(
- ` ${video.slug} ${video.chunks.length} chunk(s) → ${out.chunksOk} ok, ` +
- `${out.record.speakers.length} label(s), ${out.record.segments.length} segment(s)` +
- (metrics
- ? `, new/chunk ${fixed(metrics.newLabelsPerChunk)}, ` +
- `transcript-text ${metrics.transcriptTextLabelRate === null ? "n/a" : pct(metrics.transcriptTextLabelRate)}, ` +
- `top share ${pct(metrics.dominantSpeakerShare)}`
- : "") +
- (out.offCastTurns ? `, OFF-CAST ${out.offCastTurns}` : "") +
- (out.unknownTurns ? `, Unknown ${out.unknownTurns}` : "") +
- (out.outOfRangeTurns ? `, out-of-range ${out.outOfRangeTurns}` : "") +
- `, ${wallSeconds.toFixed(0)}s wall`,
- );
- return result;
-}
-
-// ---------------------------------------------------------------------------
-// Reporting
-// ---------------------------------------------------------------------------
-
-function pct(n: number): string {
- return `${(n * 100).toFixed(0)}%`;
-}
-function fixed(n: number | null | undefined, digits = 2): string {
- return n === null || n === undefined ? "n/a" : n.toFixed(digits);
-}
-
-type CostSummary = {
- chunks: number;
- chunksOk: number;
- calls: number;
- callsWithEngineTiming: number;
- audioSeconds: number;
- chunksPerAudioHour: number;
- // Chunk-call timings ONLY. The cast call is one per VIDEO and pricing it as a
- // chunk would flatter or punish the variant depending on video length; it is
- // reported separately, in its own unit.
- secondsPerChunkWall: number | null;
- secondsPerChunkEngine: number | null;
- secondsPerChunkEngineExLoad: number | null;
- meanLoadSeconds: number | null;
- decodeTokensPerSecond: number | null;
- outputTokens: number;
- castCallSecondsMean: number | null;
- unparsedLogLines: number;
-};
-
-function summarizeCost(results: VideoResult[], variant: Variant): CostSummary {
- const calls = results.flatMap((r) => r.calls);
- // The cast call is the FIRST call of each closed-cast video. Dropping it from
- // the per-chunk denominators keeps "s/chunk" meaning one chunk.
- const chunkCalls =
- variant === "closed-cast"
- ? results.flatMap((r) => r.calls.slice(1))
- : calls;
- const withEngine = chunkCalls.filter((c) => c.engineSeconds !== undefined);
- const withLoad = chunkCalls.filter((c) => c.loadSeconds !== undefined);
- const chunks = sum(results.map((r) => r.chunks));
- const audioSeconds = sum(results.map((r) => r.durationSeconds));
- const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0));
- const outTokens = sum(calls.map((c) => c.outputTokens ?? 0));
- const castSeconds = results
- .map((r) => r.castCallSeconds)
- .filter((s): s is number => s !== undefined);
- const chunkWall = sum(
- results.map((r) => r.wallSeconds - (r.castCallSeconds ?? 0)),
- );
- return {
- chunks,
- chunksOk: sum(results.map((r) => r.chunksOk)),
- calls: calls.length,
- callsWithEngineTiming: withEngine.length,
- audioSeconds,
- chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds),
- secondsPerChunkWall: chunks > 0 ? chunkWall / chunks : null,
- secondsPerChunkEngine: withEngine.length
- ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length
- : null,
- secondsPerChunkEngineExLoad: withEngine.length
- ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) /
- withEngine.length
- : null,
- meanLoadSeconds: withLoad.length
- ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length
- : null,
- decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null,
- outputTokens: outTokens,
- castCallSecondsMean: castSeconds.length ? sum(castSeconds) / castSeconds.length : null,
- unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)),
- };
-}
-
-// The verdict, computed against thresholds fixed before the run.
-function verdict(results: VideoResult[], cost: CostSummary) {
- const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null);
- const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null);
- const newPerChunk = withSeams.length
- ? sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length
- : null;
- const textRates = scored
- .map((m) => m.transcriptTextLabelRate)
- .filter((r): r is number => r !== null);
- const textRate = textRates.length ? Math.max(...textRates) : null;
- const engine = cost.secondsPerChunkEngine;
- const offCast = sum(results.map((r) => r.offCastTurns ?? 0));
-
- const identityOk = newPerChunk !== null && newPerChunk <= THRESHOLDS.newLabelsPerChunk;
- const textOk = textRate !== null && textRate <= THRESHOLDS.transcriptTextLabelRate;
- const costOk = engine !== null && engine < THRESHOLDS.secondsPerChunkEngine;
- return {
- newLabelsPerChunkMean: newPerChunk,
- worstTranscriptTextLabelRate: textRate,
- secondsPerChunkEngine: engine,
- offCastTurns: offCast,
- identityOk,
- textOk,
- costOk,
- // A single label at ~100% scores a perfect headline and is worthless.
- degenerateVideos: scored.filter((m) => m.labelsPerVideo === 1).length,
- passes: identityOk && textOk && costOk && offCast === 0,
- };
-}
-
-function reportMarkdown(args: {
- label: string;
- variant: Variant;
- model: string;
- numCtx?: number;
- maxCues: number;
- results: VideoResult[];
- cost: CostSummary;
- v: ReturnType<typeof verdict>;
- census: { chunks: number; videos: number; chunksPerAudioHour: number; statsSchemaStale: boolean } | null;
- boxStart: Record<string, unknown>;
- boxEnd: Record<string, unknown>;
- startedAt: string;
- sidecarsBefore: number;
- sidecarsAfter: number;
-}): string {
- const { results, cost, v } = args;
- const out: string[] = [];
- out.push(`# Attribution bake-off — ${args.label}`);
- out.push("");
- out.push(
- `Variant **${args.variant}** · ${results.length} video(s) · model \`${args.model}\` @ numCtx ` +
- `${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`,
- );
- out.push("");
- out.push(
- "**No sidecar was written.** `attribution.json` count " +
- `${args.sidecarsBefore} before, ${args.sidecarsAfter} after` +
- (args.sidecarsBefore === args.sidecarsAfter ? "." : " — **MISMATCH, investigate.**"),
- );
- out.push("");
-
- out.push("## Verdict against the thresholds fixed before the run");
- out.push("");
- out.push("| criterion | bar | measured | |");
- out.push("| --- | ---: | ---: | :-: |");
- out.push(
- `| new labels/chunk (mean) | ≤ ${THRESHOLDS.newLabelsPerChunk} | ${fixed(v.newLabelsPerChunkMean)} | ${v.identityOk ? "PASS" : "FAIL"} |`,
- );
- out.push(
- `| transcript-text labels (worst video) | ${THRESHOLDS.transcriptTextLabelRate} | ${
- v.worstTranscriptTextLabelRate === null ? "n/a" : pct(v.worstTranscriptTextLabelRate)
- } | ${v.textOk ? "PASS" : "FAIL"} |`,
- );
- out.push(
- `| s/chunk engine | < ${THRESHOLDS.secondsPerChunkEngine} | ${fixed(v.secondsPerChunkEngine, 1)} | ${v.costOk ? "PASS" : "FAIL"} |`,
- );
- if (args.variant === "closed-cast") {
- out.push(
- `| off-cast labels (grammar held) | 0 | ${v.offCastTurns} | ${v.offCastTurns === 0 ? "PASS" : "FAIL"} |`,
- );
- }
- out.push("");
- out.push(`**${v.passes ? "ALL THRESHOLDS MET" : "NOT MET"}.**`);
- out.push("");
- out.push(
- `Round 1 (open schema, same model/context/videos) measured ` +
- `**${PILOT_BASELINE.newLabelsPerChunkMean} new labels/chunk** and ` +
- `**${PILOT_BASELINE.secondsPerChunkEngine} s/chunk engine**. Digest lane baseline: ` +
- `${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE} s/chunk engine.`,
- );
- out.push("");
- out.push(
- `Single-label (degenerate-risk) videos: ${v.degenerateVideos} of ${results.length}. ` +
- "A perfect headline with one label at ~100% talk time is worthless — read the two together.",
- );
- out.push("");
-
- if (args.variant === "closed-cast") {
- out.push("## Pass A — cast discovery (1 call per video)");
- out.push("");
- out.push("| video | cast | rejected | call s |");
- out.push("| --- | --- | ---: | ---: |");
- for (const r of results) {
- out.push(
- `| ${r.slug} | ${
- (r.cast ?? []).map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "—"
- } | ${r.castRejected?.length ?? 0} | ${fixed(r.castCallSeconds, 1)} |`,
- );
- }
- out.push("");
- out.push(
- "**Pass A is independently valuable.** If pass B churns, this alone is the",
- 'cast-list product — ~1 call per video, order 10 days corpus-wide — shipping as',
- '"videos featuring X" without ever claiming who said which line.',
- );
- out.push("");
- }
-
- out.push("## Cross-chunk identity, and the label space");
- out.push("");
- out.push(
- "| video | chunks | labels | new/chunk | round-1 new/chunk | transcript-text | truncated | singleton | top share |",
- );
- out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
- for (const r of results) {
- const m = r.metrics;
- const base = PILOT_BASELINE.byVideo[r.slug];
- out.push(
- `| ${r.slug} | ${r.chunks} | ${m?.labelsPerVideo ?? "—"} | ${fixed(m?.newLabelsPerChunk)} | ${
- base ? base.newLabelsPerChunk.toFixed(2) : "—"
- } | ${
- m?.transcriptTextLabelRate === null || m === null ? "n/a" : pct(m.transcriptTextLabelRate)
- } | ${m ? pct(m.truncatedLabelRate) : "—"} | ${
- m === null || m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate)
- } | ${m ? pct(m.dominantSpeakerShare) : "—"} |`,
- );
- }
- out.push("");
- const anyText = results.some((r) => (r.metrics?.transcriptTextLabels.length ?? 0) > 0);
- if (anyText) {
- out.push("Labels that are transcript text — the failure this round exists to eliminate:");
- out.push("");
- for (const r of results) {
- for (const l of r.metrics?.transcriptTextLabels ?? []) {
- out.push(`- \`${r.slug}\`: ${JSON.stringify(l)}`);
- }
- }
- out.push("");
- }
-
- if (args.variant === "closed-cast") {
- out.push("## Pass B — how often the model declined");
- out.push("");
- out.push("| video | turns accepted | Unknown | off-cast | out-of-range |");
- out.push("| --- | ---: | ---: | ---: | ---: |");
- for (const r of results) {
- out.push(
- `| ${r.slug} | ${r.totalTurnsAccepted} | ${r.unknownTurns ?? 0} | ${r.offCastTurns ?? 0} | ${r.outOfRangeTurns} |`,
- );
- }
- out.push("");
- out.push(
- "`Unknown` is a GOOD answer, not a failure — the bias must run toward dropping on",
- "uncertainty. `off-cast` must be 0: anything else means the enum did not convert to",
- "a grammar and the whole hypothesis is refuted.",
- );
- out.push("");
- }
-
- out.push("## What the labels say");
- out.push("");
- for (const r of results) {
- const total = sum(r.speakers.map((s) => s.seconds)) || 1;
- out.push(
- `- \`${r.slug}\` — ${
- r.speakers
- .map((s) => `${s.label} ${pct(s.seconds / total)}`)
- .join(", ") || "(no speaker)"
- }`,
- );
- }
- out.push("");
-
- const warnings: Record<string, number> = {};
- for (const r of results) {
- for (const [k, n] of Object.entries(r.warningsByCode)) warnings[k] = (warnings[k] ?? 0) + n;
- }
- out.push(
- `Warnings by code: ${
- Object.keys(warnings).length
- ? Object.entries(warnings).map(([k, n]) => `${k} ${n}`).join(", ")
- : "none"
- }`,
- );
- out.push("");
-
- out.push("## Cost");
- out.push("");
- out.push(
- `${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s) total ` +
- `(${cost.callsWithEngineTiming} chunk call(s) with engine timing). Sample density ` +
- `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` +
- `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`,
- );
- out.push("");
- out.push("| quantity | value | round 1 |");
- out.push("| --- | ---: | ---: |");
- out.push(`| s/chunk, wall | ${fixed(cost.secondsPerChunkWall, 1)} | — |`);
- out.push(
- `| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} | ${PILOT_BASELINE.secondsPerChunkEngine} |`,
- );
- out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} | — |`);
- out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} | — |`);
- out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} | — |`);
- out.push(`| TOTAL output tokens | ${cost.outputTokens} | — |`);
- if (args.variant === "closed-cast") {
- out.push(`| cast call s/VIDEO (not per chunk) | ${fixed(cost.castCallSecondsMean, 1)} | — |`);
- }
- out.push(
- `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} | — |`,
- );
- out.push("");
- out.push(
- "Output tokens are the thing to watch: engine time tracked wall time within 7% in",
- "round 1, so 66.2 s/chunk was pure decode volume (31,266 output tokens on one",
- "13-chunk video). An enum token per turn instead of a 60-character string is the",
- "mechanism by which this variant is meant to be cheaper as well as better.",
- );
- out.push("");
-
- if (args.census) {
- const idle = cost.secondsPerChunkEngineExLoad;
- const contended = cost.secondsPerChunkWall;
- out.push("## Corpus projection — per-turn lane");
- out.push("");
- out.push(
- `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` +
- `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` +
- (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") +
- ".",
- );
- out.push("");
- out.push(
- `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` +
- `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended** ` +
- `— from n = ${results.length} video(s) / ${cost.chunks} chunk(s), one box, one hour.`,
- );
- out.push("");
- out.push(
- `Cast-only, for comparison: **1 call per video** over ${args.census.videos.toLocaleString()} ` +
- `videos at ${fixed(cost.castCallSecondsMean, 1)}s = ` +
- `${
- cost.castCallSecondsMean === null
- ? "n/a"
- : ((args.census.videos * cost.castCallSecondsMean) / 86400).toFixed(1)
- } days.`,
- );
- out.push("");
- out.push(
- "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that has",
- "completed 0.17% of its own ~194k calls, with nothing arbitrating between them.",
- );
- out.push("");
- }
-
- out.push("## Box");
- out.push("");
- out.push(`Start: \`${JSON.stringify(args.boxStart)}\``);
- out.push("");
- out.push(`End: \`${JSON.stringify(args.boxEnd)}\``);
- out.push("");
- if (cost.unparsedLogLines > 0) {
- out.push(
- `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`,
- );
- out.push("");
- }
-
- out.push("## Units");
- out.push("");
- out.push(
- "- The unit of per-turn work is the **chunk** (one model call). Seconds-per-audio-",
- " hour is not a unit here: chunk density varies 4× across this corpus, and pricing",
- " in it is the error behind a retracted throughput headline. This report does not",
- " print one.",
- "- The unit of cast discovery is **calls per video**, priced separately and never",
- " folded into s/chunk.",
- "- Wall and engine are different quantities and are never substituted for each",
- " other. Wall is the realistic ceiling on this shared box; engine excl. load is",
- " the floor.",
- `- Every rate carries its denominator. At n = ${results.length} videos a bare percentage`,
- " would be a lie of precision.",
- );
- out.push("");
- return out.join("\n");
-}
-
-// ---------------------------------------------------------------------------
-
-async function main(): Promise<void> {
- const flags = parseFlags(process.argv.slice(2));
- const paths = getPaths();
- const variant: Variant = flags.variant === "open" ? "open" : "closed-cast";
- const dryRun = flags["dry-run"] === "true";
- const slugs = (flags.videos ?? DEFAULT_VIDEOS.join(",")).split(",").map((s) => s.trim()).filter(Boolean);
- const label = flags.label ?? `${variant}-${new Date().toISOString().slice(0, 10)}`;
-
- // The lane is enabled IN MEMORY, exactly as the pilot does it. settings.json
- // is never written — a bake-off must not be the thing that arms a lane.
- const probeSettings: AttributionSettings = {
- enabled: true,
- appId: OLLAMA_DIGEST_APP_ID,
- model: flags.model ?? "",
- diarizedEnabled: false,
- textOnlyEnabled: true,
- promptVersion: ATTRIBUTION_PROMPT_VERSION,
- };
- const { app, config: baseConfig, modelRequested } = resolveAttributionTarget(
- "text-only",
- probeSettings,
- );
- // Overrides land on a COPY of the app config. resolveAttributionTarget reads
- // the real settings for the ollama URL and timeout, which we want, but the
- // model and context are this run's variables.
- const config: DigestAppConfig = {
- ...baseConfig,
- model: modelRequested,
- ...(flags["num-ctx"] ? { numCtx: Number(flags["num-ctx"]) } : {}),
- };
- const maxCues = maxCuesForContext(config.numCtx);
-
- const videos: ProbeVideo[] = [];
- const skipped: string[] = [];
- for (const slug of slugs) {
- const v = await priceVideo(slug, paths, maxCues);
- if (v) videos.push(v);
- else skipped.push(slug);
- }
-
- const totalChunks = sum(videos.map((v) => v.chunks.length));
- const totalAudio = sum(videos.map((v) => v.durationSeconds));
- const census = await readChunkCensus(paths, maxCues);
- const castCalls = variant === "closed-cast" ? videos.length : 0;
-
- console.log(
- `Attribution bake-off — variant ${variant}, app ${app.id}, model ${modelRequested}, ` +
- `numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`,
- );
- console.log(`Settings override (in memory only): ${JSON.stringify(probeSettings)}`);
- console.log("NO SIDECAR IS WRITTEN. writeAttribution is not imported by this harness.");
- console.log("");
- for (const v of videos) {
- console.log(
- ` ${v.slug} ${toHms(v.durationSeconds)} ${v.cues.length} cues → ${v.chunks.length} chunk(s)`,
- );
- }
- if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`);
- console.log("");
- console.log(
- `${videos.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` +
- `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour.`,
- );
- console.log(
- `Calls this run: ${castCalls} cast + ${totalChunks} chunk = ${castCalls + totalChunks}.`,
- );
- if (census) {
- console.log(
- `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s)` +
- (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""),
- );
- }
- console.log(`Box: ${JSON.stringify(boxState())}`);
- console.log("");
-
- if (videos.length === 0) {
- console.error("No video had a usable transcript — nothing to measure.");
- process.exitCode = 1;
- return;
- }
-
- if (dryRun) {
- console.log("--dry-run: priced only, no model call and no write.");
- return;
- }
-
- const sidecarsBefore = await countSidecars(paths);
- console.log(`attribution.json on disk before: ${sidecarsBefore}`);
- console.log("");
-
- const boxStart = boxState();
- const startedAt = new Date().toISOString();
- const results: VideoResult[] = [];
- for (const video of videos) {
- const context = await readDigestContext(paths, video.channelSlug);
- results.push(
- await runVideo(
- video,
- variant,
- { app, config, ...(context.note ? { contextNote: context.note } : {}) },
- (m) => console.log(m),
- ),
- );
- }
- const boxEnd = boxState();
-
- const sidecarsAfter = await countSidecars(paths);
- const cost = summarizeCost(results, variant);
- const v = verdict(results, cost);
-
- const report = {
- version: 1 as const,
- label,
- variant,
- startedAt,
- finishedAt: new Date().toISOString(),
- engine: {
- appId: app.id,
- modelRequested,
- numCtx: config.numCtx,
- maxCues,
- overlapCues: DIGEST_OVERLAP_CUES,
- castSamples: ATTRIBUTION_CAST_SAMPLES,
- maxTurnsPerChunk: ATTRIBUTION_MAX_TURNS_PER_CHUNK,
- },
- thresholds: THRESHOLDS,
- verdict: v,
- skipped,
- sidecars: { before: sidecarsBefore, after: sidecarsAfter },
- box: { start: boxStart, end: boxEnd },
- census,
- cost,
- baseline: {
- pilot: PILOT_BASELINE,
- digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE,
- },
- videos: results,
- };
-
- const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot");
- await mkdir(outDir, { recursive: true });
- const markdown = reportMarkdown({
- label,
- variant,
- model: modelRequested,
- numCtx: config.numCtx,
- maxCues,
- results,
- cost,
- v,
- census,
- boxStart,
- boxEnd,
- startedAt,
- sidecarsBefore,
- sidecarsAfter,
- });
- const jsonPath = path.join(outDir, `${label}.json`);
- const mdPath = path.join(outDir, `${label}.md`);
- await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n");
- await writeFile(mdPath, markdown);
-
- console.log("");
- console.log(markdown);
- console.log(`Wrote ${jsonPath}`);
- console.log(`Wrote ${mdPath}`);
-
- // THE SAFETY CLAIM, enforced. A bake-off that left a sidecar behind would have
- // silently made a losing variant look fresh to the whole system.
- if (sidecarsAfter !== sidecarsBefore) {
- console.error(
- `attribution.json count changed: ${sidecarsBefore} -> ${sidecarsAfter}. ` +
- "This harness must never write one.",
- );
- process.exitCode = 1;
- } else {
- console.log(`attribution.json on disk after: ${sidecarsAfter} (unchanged).`);
- }
-
- if (sum(results.map((r) => r.chunksOk)) === 0) {
- console.error("No chunk succeeded — nothing was measured.");
- process.exitCode = 1;
- }
-}
-
-main().catch((err) => {
- console.error(err);
- process.exit(1);
-});
diff --git a/common/bin/attribution-pilot.ts b/common/bin/attribution-pilot.ts
@@ -1,1122 +0,0 @@
-#!/usr/bin/env tsx
-// Measure the attribution lanes over a small sample BEFORE deciding whether the
-// text-only one is worth 25-55 GPU-days over the whole corpus.
-//
-// WHY THIS EXISTS. Both lanes shipped with nothing armed, deliberately: the
-// text-only lane costs roughly one model call per transcript CHUNK, which
-// corpus-wide is ~194,000 calls — the same order as the digest sweep, which has
-// itself completed 0.17%. That spend is decided on measurements, the way the
-// digest layer was (pilot -> bake-off -> measured defaults -> sweep), and this
-// is the measurement.
-//
-// WHY IT CALLS attributeOneVideo RATHER THAN BYPASSING IT, unlike
-// digest-bakeoff.ts which drives the prompt modules directly. The bake-off was
-// comparing ENGINES and had to leave nothing behind; this is measuring the
-// SHIPPED PATH — the freshness skip, the downgrade guard, the provenance it
-// records and the sidecar it writes are all part of what is under test, and the
-// record is the artifact a human then reads to judge the labels. So it writes
-// real attribution.json files into the real corpus. That is bounded (one small
-// sidecar per video, nothing consumes it yet) and reversible with one command:
-//
-// find transcripts/channels -name attribution.json -delete
-//
-// NO EDITOR, NO SETTINGS WRITE. attributeOneVideo takes an explicit `settings`
-// override and never calls writeSettings, so the pilot enables the lane
-// IN MEMORY and the operator's settings.json is untouched. A pilot must not be
-// the thing that arms a lane.
-//
-// UNIT DISCIPLINE, and it is not negotiable. plans/STATE.md records a RETRACTED
-// throughput headline that came from pricing a sweep in seconds-per-audio-hour
-// when the unit of work is the chunk (density varies 4x across this corpus). So
-// this harness reports seconds per CHUNK for the text-only lane and calls per
-// VIDEO for the diarized one, prints chunks-per-audio-hour beside any rate, and
-// REFUSES to print a bare seconds-per-audio-hour figure. sweepDaysFromAudio is
-// deliberately not imported.
-//
-// Examples:
-// tsx bin/attribution-pilot.ts --sample bakeoff --dry-run
-// tsx bin/attribution-pilot.ts --sample bakeoff --method text-only --label round1
-// tsx bin/attribution-pilot.ts --sample channel:quarteringvlogs --label autocap
-// tsx bin/attribution-pilot.ts --sample video:ObviousRises-rumble/v6z1o2g \
-// --method diarized --label diarized-smoke
-
-import os from "node:os";
-import path from "node:path";
-import { mkdir, readFile, readdir, writeFile } from "node:fs/promises";
-import { getPaths, type Paths } from "../lib/paths";
-import { parseFlags } from "./_parseFlags";
-import {
- ATTRIBUTION_PROMPT_VERSION,
- attributionSpeechSeconds,
- type AttributionRecord,
- type AttributionMethod,
-} from "../lib/attribution";
-import { loadAttribution } from "../lib/attribution-server";
-import { loadDiarization } from "../lib/diarization-server";
-import { selectClusterSamples } from "../lib/attributionPrompt";
-import type { AttributionSettings } from "../lib/settings";
-import { OLLAMA_DIGEST_APP_ID } from "../lib/digestApps";
-import {
- attributeOneVideo,
- type AttributeOneOutcome,
-} from "../controller/attributeOne";
-import { resolveAttributionTarget } from "../controller/attributionTarget";
-import {
- DIGEST_OVERLAP_CUES,
- maxCuesForContext,
- toHms,
-} from "../lib/digestPrompt";
-import { chunkCuesForContext } from "../lib/transcriptWindow";
-import {
- chunksPerAudioHour,
- readChunkCensus,
- sweepDays,
- MEASURED_SECONDS_PER_CHUNK,
-} from "../controller/digestPlan";
-import {
- isCuesJsonFresh,
- readNormalizedTranscript,
-} from "../controller/normalizeTranscript";
-import type { Cue } from "../lib/vtt";
-
-// The digest lane's own cost on the SAME model, context and eight videos —
-// round 2's 190 engine-seconds over 17 chunks. Printed beside the pilot's
-// number because "is attribution more expensive per chunk than a digest" is one
-// of the two questions this run answers.
-const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2;
-
-// ---------------------------------------------------------------------------
-// The sample
-// ---------------------------------------------------------------------------
-
-type PilotVideo = {
- slug: string;
- channelSlug: string;
- // The DIRECTORY name, which is not always the video id: the bake-off sample's
- // the-quartering-rumble entry is videoDir v6xl0vu against videoId v6ve4n0 (a
- // Rumble canonical-id reconciliation). Resolving the path off videoId returns
- // no-transcript and silently drops a stratum from the sample.
- videoDir: string;
- videoId: string;
- title: string;
- bucket?: string;
- durationSeconds: number;
- dir: string;
-};
-
-// A video that made it far enough to be priced: cues on disk, chunk count known.
-type PricedVideo = PilotVideo & {
- cues: Cue[];
- chunks: number;
- chunkRanges: { start: number; end: number }[];
-};
-
-async function resolveSample(
- spec: string,
- paths: Paths,
-): Promise<PilotVideo[]> {
- if (spec === "bakeoff") {
- const file = path.join(paths.monorepoRoot, "plans", "bakeoff", "sample.json");
- const raw = JSON.parse(await readFile(file, "utf8")) as {
- videos: {
- slug: string;
- channelSlug: string;
- videoId: string;
- videoDir: string;
- title: string;
- bucket: string;
- durationSeconds: number;
- }[];
- };
- return raw.videos.map((v) => ({
- slug: v.slug,
- channelSlug: v.channelSlug,
- videoDir: v.videoDir,
- videoId: v.videoId,
- title: v.title,
- bucket: v.bucket,
- durationSeconds: v.durationSeconds,
- dir: path.join(paths.channelsDir, v.channelSlug, "data", v.videoDir),
- }));
- }
- if (spec.startsWith("channel:")) {
- const channelSlug = spec.slice("channel:".length);
- const dataDir = path.join(paths.channelsDir, channelSlug, "data");
- const entries = await readdir(dataDir, { withFileTypes: true });
- return entries
- .filter((e) => e.isDirectory())
- .map((e) => ({
- slug: `${channelSlug}/${e.name}`,
- channelSlug,
- videoDir: e.name,
- videoId: e.name,
- title: "",
- durationSeconds: 0,
- dir: path.join(dataDir, e.name),
- }))
- .sort((a, b) => a.videoDir.localeCompare(b.videoDir));
- }
- if (spec.startsWith("video:")) {
- const rest = spec.slice("video:".length);
- const slash = rest.indexOf("/");
- if (slash <= 0) throw new Error(`--sample video:<slug>/<dir> expected, got "${spec}"`);
- const channelSlug = rest.slice(0, slash);
- const videoDir = rest.slice(slash + 1);
- return [
- {
- slug: rest,
- channelSlug,
- videoDir,
- videoId: videoDir,
- title: "",
- durationSeconds: 0,
- dir: path.join(paths.channelsDir, channelSlug, "data", videoDir),
- },
- ];
- }
- throw new Error(`unknown --sample "${spec}" (bakeoff | channel:<slug> | video:<slug>/<dir>)`);
-}
-
-// Chunk boundaries EXACTLY as findTurns computes them — same chunker, same
-// overlap, same floor/ceil. Every cross-chunk metric rests on this agreeing with
-// the runner, which is why it is derived from the resolved config's numCtx
-// rather than from a literal 600, and why the scorer asserts its count against
-// provenance.chunks and reports null (never 0) on a mismatch.
-function chunkRangesFor(cues: Cue[], maxCues: number): { start: number; end: number }[] {
- return chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES }).map(
- (chunk) => {
- const start = Math.max(0, Math.floor(chunk[0].start));
- const last = chunk[chunk.length - 1];
- return { start, end: Math.max(start, Math.ceil(last.end || last.start)) };
- },
- );
-}
-
-async function price(video: PilotVideo, maxCues: number): Promise<PricedVideo | null> {
- const { fresh, cuesPath } = await isCuesJsonFresh(video.dir);
- if (!fresh) return null;
- const transcript = await readNormalizedTranscript(cuesPath);
- const cues = transcript?.cues ?? [];
- if (!transcript || cues.length === 0) return null;
- const ranges = chunkRangesFor(cues, maxCues);
- const lastCue = cues[cues.length - 1];
- return {
- ...video,
- videoId: video.videoId || transcript.id,
- title: video.title || transcript.title || video.videoId,
- durationSeconds:
- video.durationSeconds > 0
- ? video.durationSeconds
- : Math.round(transcript.duration || lastCue.end || lastCue.start || 0),
- cues,
- chunks: ranges.length,
- chunkRanges: ranges,
- };
-}
-
-// ---------------------------------------------------------------------------
-// Per-call timing, parsed out of the log stream digestApps already emits
-// ---------------------------------------------------------------------------
-
-// `ollama qwen2.5:7b: 4102 in / 218 out tokens in 5.0s wall (load 0.4s, prefill
-// 0.1s, decode 3.0s, engine 3.5s) (num_ctx 8192)`
-//
-// No new instrumentation: attributeOneVideo forwards the caller's onLog straight
-// into the app's run(), so one closure sees every call of the video it wraps.
-const CALL_RE =
- /^ollama (\S+): (\d+|\?) in \/ (\d+|\?) out tokens in ([\d.]+)s wall(?: \(([^)]*)\))?/;
-const PHASE_RE = /^(load|prefill|decode|engine) ([\d.]+)s$/;
-
-type CallTiming = {
- wallSeconds: number;
- loadSeconds?: number;
- prefillSeconds?: number;
- decodeSeconds?: number;
- engineSeconds?: number;
- inputTokens?: number;
- outputTokens?: number;
-};
-
-function parseCall(line: string): CallTiming | null {
- const m = CALL_RE.exec(line);
- if (!m) return null;
- const out: CallTiming = { wallSeconds: Number(m[4]) };
- if (m[2] !== "?") out.inputTokens = Number(m[2]);
- if (m[3] !== "?") out.outputTokens = Number(m[3]);
- // nsToMs OMITS a phase rather than zeroing it, and the whole parenthetical is
- // dropped when the engine reports nothing — so an absent phase stays absent
- // here too, and every engine-derived figure is averaged over its own
- // denominator rather than over the call count.
- for (const part of (m[5] ?? "").split(", ")) {
- const p = PHASE_RE.exec(part.trim());
- if (!p) continue;
- const seconds = Number(p[2]);
- if (p[1] === "load") out.loadSeconds = seconds;
- else if (p[1] === "prefill") out.prefillSeconds = seconds;
- else if (p[1] === "decode") out.decodeSeconds = seconds;
- else out.engineSeconds = seconds;
- }
- return out;
-}
-
-// ---------------------------------------------------------------------------
-// Metrics
-// ---------------------------------------------------------------------------
-
-// Roles a label can carry that name nobody. isUselessSpeakerLabel already blocks
-// "Speaker 1" before the write, so the record cannot contain one; this measures
-// the WIDER set the runner permits. Anchored whole-string, not a prefix match:
-// "Host" is generic, "Host Jane Doe" is not.
-const GENERIC_LABEL_RE =
- /^(the\s+)?(co-?host|host|guest|caller|narrator|interviewer|moderator|panelist|announcer|commentator|reporter|audience|crowd|man|woman|male|female|person|voice|speaker)(\s*\d+)?$/i;
-
-type VideoMetrics = {
- labels: string[];
- labelsPerVideo: number;
- // Null rather than 0 whenever the property could not be measured — a single
- // chunk has no seam to cross, and a chunk-count mismatch means the
- // reconstruction disagrees with the runner. Reporting 0 for "not measured"
- // is how a harness reports a perfect score for something that never ran.
- newLabelsPerChunk: number | null;
- singletonLabelRate: number | null;
- dominantSpeakerShare: number;
- talkShares: number[];
- nearDuplicateLabelRate: number;
- nearDuplicatePairs: string[];
- genericLabelRate: number;
- genericSecondsShare: number;
- nonAsciiLabelRate: number;
- coverageRate: number;
- maxGapSeconds: number;
- medianGapSeconds: number;
- markFreeChunks: number | null;
- segments: number;
- chunkReconstructionOk: boolean;
- labelFirstChunk: (number | null)[];
- labelChunkCount: (number | null)[];
-};
-
-function normalizeForCompare(label: string): string {
- return label
- .trim()
- .replace(/\s+/g, " ")
- .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "")
- .toLowerCase();
-}
-
-// Labels the runner's deliberately conservative normalizeLabel keeps apart but
-// that are obviously one person ("Jane" / "Jane Doe"). The runner is right to
-// refuse the merge — a merged speaker silently attributes words to the wrong
-// person — but a SCORER may flag, and this rate is the direct read on whether
-// that conservatism is set right.
-function nearDuplicate(a: string, b: string): boolean {
- const ta = new Set(normalizeForCompare(a).split(" ").filter(Boolean));
- const tb = new Set(normalizeForCompare(b).split(" ").filter(Boolean));
- if (ta.size === 0 || tb.size === 0) return false;
- const [small, big] = ta.size <= tb.size ? [ta, tb] : [tb, ta];
- let shared = 0;
- for (const t of small) if (big.has(t)) shared++;
- if (shared === small.size) return true; // proper token-subset
- for (const t of small) if (t.length >= 4 && big.has(t)) return true;
- return false;
-}
-
-function median(values: number[]): number {
- if (values.length === 0) return 0;
- const s = [...values].sort((a, b) => a - b);
- const mid = Math.floor(s.length / 2);
- return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
-}
-
-function scoreRecord(record: AttributionRecord, video: PricedVideo): VideoMetrics {
- const ranges = video.chunkRanges;
- // The runner records what it actually chunked; if our reconstruction differs,
- // every chunk-indexed number below is meaningless and is reported as null.
- const recordedChunks = record.provenance.chunks;
- const reconstructionOk =
- recordedChunks === undefined || recordedChunks === ranges.length;
- const multiChunk = reconstructionOk && ranges.length > 1;
-
- const chunkOf = (t: number): number => {
- // The 40-cue overlap means a seam second sits in two chunks; first match is
- // the earlier one, which is the chunk that ESTABLISHED the label.
- for (let i = 0; i < ranges.length; i++) {
- if (t >= ranges[i].start && t <= ranges[i].end) return i;
- }
- return t < ranges[0].start ? 0 : ranges.length - 1;
- };
-
- const chunksByLabel = new Map<number, Set<number>>();
- for (const s of record.segments) {
- const set = chunksByLabel.get(s.speaker) ?? new Set<number>();
- set.add(chunkOf(s.start));
- chunksByLabel.set(s.speaker, set);
- }
-
- const labels = record.speakers.map((s) => s.label);
- const seconds = record.speakers.map((s) => s.seconds ?? 0);
- const totalSeconds = seconds.reduce((a, b) => a + b, 0);
- const labelFirstChunk = record.speakers.map((s) => {
- const set = chunksByLabel.get(s.index);
- if (!reconstructionOk || !set || set.size === 0) return null;
- return Math.min(...set);
- });
- const labelChunkCount = record.speakers.map((s) => {
- const set = chunksByLabel.get(s.index);
- if (!reconstructionOk || !set) return null;
- return set.size;
- });
-
- // THE HEADLINE. Labels first appearing after chunk 0, per later chunk. Near 0
- // means the cast stabilised; >= 1 means the model reinvented it every chunk
- // and the lane is not delivering the one property it exists for.
- const newLabelsPerChunk = multiChunk
- ? labelFirstChunk.filter((c) => c !== null && c > 0).length / (ranges.length - 1)
- : null;
- const singletonLabelRate =
- multiChunk && labels.length > 0
- ? labelChunkCount.filter((c) => c === 1).length / labels.length
- : null;
-
- // THE DEGENERATE-CASE GUARD. One label at ~100% of talk time scores a perfect
- // 0 on the headline while being worthless, and the headline alone cannot tell
- // those apart.
- const dominantSpeakerShare =
- totalSeconds > 0 ? Math.max(...seconds) / totalSeconds : 0;
-
- const dupPairs: string[] = [];
- const involved = new Set<number>();
- for (let i = 0; i < labels.length; i++) {
- for (let j = i + 1; j < labels.length; j++) {
- if (!nearDuplicate(labels[i], labels[j])) continue;
- dupPairs.push(`${labels[i]} ~ ${labels[j]}`);
- involved.add(i);
- involved.add(j);
- }
- }
-
- const genericIdx = labels
- .map((l, i) => (GENERIC_LABEL_RE.test(l.trim()) ? i : -1))
- .filter((i) => i >= 0);
- const genericSeconds = genericIdx.reduce((a, i) => a + seconds[i], 0);
-
- // COVERAGE, and what it is worth. Text-only marks are TILED — each runs to the
- // next change, the last to the end — so this is ~1.0 by construction. It is a
- // structural check (anything else is a bug), not a quality signal. The honest
- // coverage metric for this lane is markFreeChunks.
- const duration = video.durationSeconds || 0;
- const covered = attributionSpeechSeconds(record);
- const sorted = [...record.segments].sort((a, b) => a.start - b.start);
- const gaps: number[] = [];
- let cursor = 0;
- for (const s of sorted) {
- if (s.start > cursor) gaps.push(s.start - cursor);
- cursor = Math.max(cursor, s.end);
- }
- if (duration > cursor) gaps.push(duration - cursor);
-
- const markFreeChunks = reconstructionOk
- ? ranges.filter(
- (r) => !record.segments.some((s) => s.start >= r.start && s.start <= r.end),
- ).length
- : null;
-
- return {
- labels,
- labelsPerVideo: labels.length,
- newLabelsPerChunk,
- singletonLabelRate,
- dominantSpeakerShare,
- talkShares: totalSeconds > 0 ? seconds.map((s) => s / totalSeconds) : seconds.map(() => 0),
- nearDuplicateLabelRate: labels.length > 0 ? involved.size / labels.length : 0,
- nearDuplicatePairs: dupPairs,
- genericLabelRate: labels.length > 0 ? genericIdx.length / labels.length : 0,
- genericSecondsShare: totalSeconds > 0 ? genericSeconds / totalSeconds : 0,
- nonAsciiLabelRate:
- labels.length > 0
- ? labels.filter((l) => /[^\x20-\x7E]/.test(l)).length / labels.length
- : 0,
- coverageRate: duration > 0 ? covered / duration : 0,
- maxGapSeconds: gaps.length ? Math.max(...gaps) : 0,
- medianGapSeconds: median(gaps),
- markFreeChunks,
- segments: record.segments.length,
- chunkReconstructionOk: reconstructionOk,
- labelFirstChunk,
- labelChunkCount,
- };
-}
-
-// ---------------------------------------------------------------------------
-// The run
-// ---------------------------------------------------------------------------
-
-type VideoResult = {
- slug: string;
- bucket?: string;
- title: string;
- durationSeconds: number;
- expectedChunks: number;
- outcome: AttributeOneOutcome;
- // Wall time around attributeOneVideo — the realistic ceiling on this shared
- // box, and NOT the same quantity as the engine's own number.
- videoWallSeconds: number;
- calls: CallTiming[];
- unparsedLogLines: number;
- recordedChunks?: number;
- recordedChunksOk?: number;
- warningsByCode: Record<string, number>;
- metrics: VideoMetrics | null;
- // Diarized lane only.
- clustersOffered?: number;
- clustersNamed?: number;
- confidences?: number[];
-};
-
-function boxState(): Record<string, unknown> {
- const load = os.loadavg();
- return {
- loadavg: load.map((n) => Number(n.toFixed(2))),
- freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)),
- totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)),
- };
-}
-
-function sum(values: number[]): number {
- return values.reduce((a, b) => a + b, 0);
-}
-
-async function runVideo(
- video: PricedVideo,
- method: AttributionMethod,
- settings: AttributionSettings,
- paths: Paths,
- force: boolean,
- log: (m: string) => void,
-): Promise<VideoResult> {
- const calls: CallTiming[] = [];
- let unparsed = 0;
- const onLog = (line: string) => {
- const call = parseCall(line);
- if (call) calls.push(call);
- // Only the ollama call lines are parseable; the runner's own progress lines
- // are not, and are not counted as drift.
- else if (line.startsWith("ollama ")) unparsed++;
- };
-
- const started = Date.now();
- const outcome = await attributeOneVideo({
- paths,
- videoDir: video.dir,
- videoId: video.videoId,
- channelSlug: video.channelSlug,
- method,
- settings,
- force,
- onLog,
- });
- const videoWallSeconds = (Date.now() - started) / 1000;
-
- const record = await loadAttribution(video.dir);
- const warningsByCode: Record<string, number> = {};
- for (const w of record?.warnings ?? []) {
- warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1;
- }
-
- const result: VideoResult = {
- slug: video.slug,
- ...(video.bucket ? { bucket: video.bucket } : {}),
- title: video.title,
- durationSeconds: video.durationSeconds,
- expectedChunks: video.chunks,
- outcome,
- videoWallSeconds,
- calls,
- unparsedLogLines: unparsed,
- ...(record?.provenance.chunks !== undefined
- ? { recordedChunks: record.provenance.chunks, recordedChunksOk: record.provenance.chunksOk }
- : {}),
- warningsByCode,
- metrics: record && outcome === "attributed" ? scoreRecord(record, video) : null,
- };
-
- if (method === "diarized") {
- const diarization = await loadDiarization(video.dir);
- if (diarization) {
- result.clustersOffered = selectClusterSamples(diarization.turns, video.cues).length;
- }
- result.clustersNamed = record?.speakers.length ?? 0;
- result.confidences = (record?.speakers ?? [])
- .map((s) => s.confidence)
- .filter((c): c is number => typeof c === "number");
- }
-
- const m = result.metrics;
- log(
- ` ${video.slug}${video.bucket ? ` [${video.bucket}]` : ""} ${video.chunks} chunk(s) → ` +
- `${outcome}` +
- (m
- ? `, ${m.labelsPerVideo} label(s), ${m.segments} segment(s), ` +
- `new/chunk ${m.newLabelsPerChunk === null ? "n/a" : m.newLabelsPerChunk.toFixed(2)}, ` +
- `top share ${(m.dominantSpeakerShare * 100).toFixed(0)}%`
- : "") +
- `, ${videoWallSeconds.toFixed(0)}s wall` +
- (calls.length ? ` / ${sum(calls.map((c) => c.engineSeconds ?? 0)).toFixed(0)}s engine` : ""),
- );
- return result;
-}
-
-// ---------------------------------------------------------------------------
-// Reporting
-// ---------------------------------------------------------------------------
-
-type CostSummary = {
- videosMeasured: number;
- chunks: number;
- chunksOk: number;
- calls: number;
- callsWithEngineTiming: number;
- audioSeconds: number;
- chunksPerAudioHour: number;
- secondsPerChunkWall: number | null;
- secondsPerChunkCallWall: number | null;
- secondsPerChunkEngine: number | null;
- secondsPerChunkEngineExLoad: number | null;
- meanLoadSeconds: number | null;
- decodeTokensPerSecond: number | null;
- unparsedLogLines: number;
-};
-
-// Throughput is computed ONLY over videos the run actually attributed: an
-// already-exists short-circuits in milliseconds and would otherwise report an
-// absurd chunk rate.
-function summarizeCost(results: VideoResult[]): CostSummary {
- const measured = results.filter((r) => r.outcome === "attributed");
- const calls = measured.flatMap((r) => r.calls);
- const withEngine = calls.filter((c) => c.engineSeconds !== undefined);
- const chunks = sum(measured.map((r) => r.recordedChunks ?? r.expectedChunks));
- const chunksOk = sum(measured.map((r) => r.recordedChunksOk ?? r.calls.length));
- const audioSeconds = sum(measured.map((r) => r.durationSeconds));
- const wall = sum(measured.map((r) => r.videoWallSeconds));
- const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0));
- const outTokens = sum(calls.map((c) => c.outputTokens ?? 0));
- const withLoad = calls.filter((c) => c.loadSeconds !== undefined);
- return {
- videosMeasured: measured.length,
- chunks,
- chunksOk,
- calls: calls.length,
- callsWithEngineTiming: withEngine.length,
- audioSeconds,
- chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds),
- secondsPerChunkWall: chunks > 0 ? wall / chunks : null,
- secondsPerChunkCallWall: calls.length ? sum(calls.map((c) => c.wallSeconds)) / calls.length : null,
- secondsPerChunkEngine: withEngine.length
- ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length
- : null,
- secondsPerChunkEngineExLoad: withEngine.length
- ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) /
- withEngine.length
- : null,
- meanLoadSeconds: withLoad.length
- ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length
- : null,
- decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null,
- unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)),
- };
-}
-
-function pct(n: number): string {
- return `${(n * 100).toFixed(0)}%`;
-}
-
-function fixed(n: number | null, digits = 2): string {
- return n === null ? "n/a" : n.toFixed(digits);
-}
-
-// Verbatim transcript at a speaker's longest segments. WITHOUT this nobody can
-// judge whether a label is right, and no aggregate metric substitutes for
-// reading it — which is why the Markdown artifact, not the JSON, is the point of
-// the exercise.
-function excerpt(cues: Cue[], start: number, end: number, maxChars = 420): string[] {
- const lines: string[] = [];
- let chars = 0;
- for (const c of cues) {
- if (c.start < start) continue;
- if (c.start > end) break;
- const text = c.text.trim().replace(/\s+/g, " ");
- if (!text) continue;
- lines.push(`[${toHms(c.start)}] ${text}`);
- chars += text.length;
- if (chars >= maxChars) break;
- }
- return lines;
-}
-
-function qualityMarkdown(
- label: string,
- method: AttributionMethod,
- results: VideoResult[],
- priced: Map<string, PricedVideo>,
- records: Map<string, AttributionRecord>,
-): string {
- const out: string[] = [];
- out.push(`# Attribution pilot — ${label} (${method}): what the labels actually say`);
- out.push("");
- out.push(
- "Read this file, not the metrics, to answer *are the labels right*. Each speaker",
- "is shown with its talk-time share and the verbatim transcript at its longest",
- "segments — the same text, in the same `[HH:MM:SS] line` form, the model saw.",
- );
- out.push("");
- for (const r of results) {
- const video = priced.get(r.slug);
- const record = records.get(r.slug);
- out.push(`## ${r.slug}${r.bucket ? ` — ${r.bucket}` : ""}`);
- out.push("");
- out.push(`**${r.title || "(untitled)"}**`);
- out.push("");
- out.push(
- `${toHms(r.durationSeconds)} · ${r.expectedChunks} chunk(s) · outcome \`${r.outcome}\`` +
- (r.recordedChunks !== undefined
- ? ` · ${r.recordedChunksOk}/${r.recordedChunks} chunk(s) ok`
- : ""),
- );
- out.push("");
- if (!record || !video || !r.metrics) {
- out.push("_No record was written for this video._");
- out.push("");
- continue;
- }
- const m = r.metrics;
- for (let i = 0; i < record.speakers.length; i++) {
- const speaker = record.speakers[i];
- const share = m.talkShares[i] ?? 0;
- const first = m.labelFirstChunk[i];
- const present = m.labelChunkCount[i];
- out.push(
- `### ${speaker.label} — ${pct(share)} of attributed time` +
- (speaker.confidence !== undefined ? `, confidence ${speaker.confidence.toFixed(2)}` : "") +
- (first !== null ? `, first seen in chunk ${first + 1}` : "") +
- (present !== null ? `, present in ${present}/${r.expectedChunks} chunk(s)` : ""),
- );
- out.push("");
- const mine = record.segments
- .filter((s) => s.speaker === speaker.index)
- .sort((a, b) => b.end - b.start - (a.end - a.start))
- .slice(0, 3)
- .sort((a, b) => a.start - b.start);
- if (mine.length === 0) {
- out.push("_No segments._");
- out.push("");
- continue;
- }
- for (const seg of mine) {
- out.push(`_${toHms(seg.start)} – ${toHms(seg.end)}_`);
- out.push("");
- out.push("```");
- const lines = excerpt(video.cues, seg.start, Math.min(seg.end, seg.start + 90));
- out.push(...(lines.length ? lines : ["(no cue text in range)"]));
- out.push("```");
- out.push("");
- }
- }
- if (m.nearDuplicatePairs.length) {
- out.push(`Near-duplicate label pairs: ${m.nearDuplicatePairs.map((p) => `\`${p}\``).join(", ")}`);
- out.push("");
- }
- out.push("- [ ] labels are right");
- out.push("- [ ] a label is wrong (which: ____)");
- out.push("- [ ] a name appears that the transcript never states");
- out.push("- [ ] one person is split across two labels");
- out.push("- [ ] two people are merged into one label");
- out.push("");
- }
- return out.join("\n");
-}
-
-function reportMarkdown(args: {
- label: string;
- method: AttributionMethod;
- sample: string;
- model: string;
- numCtx?: number;
- maxCues: number;
- results: VideoResult[];
- cost: CostSummary;
- census: { chunks: number; videos: number; chunksPerAudioHour: number; estimated: number; statsSchemaStale: boolean } | null;
- boxStart: Record<string, unknown>;
- boxEnd: Record<string, unknown>;
- startedAt: string;
-}): string {
- const { results, cost } = args;
- const out: string[] = [];
- out.push(`# Attribution pilot — ${args.label}`);
- out.push("");
- out.push(
- `Lane **${args.method}** · sample \`${args.sample}\` · ${results.length} video(s) · ` +
- `model \`${args.model}\` @ numCtx ${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`,
- );
- out.push("");
-
- const outcomes: Record<string, number> = {};
- for (const r of results) outcomes[r.outcome] = (outcomes[r.outcome] ?? 0) + 1;
- out.push(`Outcomes: ${Object.entries(outcomes).map(([k, v]) => `${k} ${v}`).join(", ")}`);
- out.push("");
-
- if (args.method === "text-only") {
- out.push("## Cross-chunk identity — the property the lane exists for");
- out.push("");
- out.push("| video | chunks | labels | new labels/chunk | singleton rate | top share | near-dup | generic |");
- out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
- for (const r of results) {
- const m = r.metrics;
- out.push(
- `| ${r.slug} | ${r.expectedChunks} | ${m?.labelsPerVideo ?? "—"} | ${
- m ? fixed(m.newLabelsPerChunk) : "—"
- } | ${m ? (m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate)) : "—"} | ${
- m ? pct(m.dominantSpeakerShare) : "—"
- } | ${m ? pct(m.nearDuplicateLabelRate) : "—"} | ${m ? pct(m.genericLabelRate) : "—"} |`,
- );
- }
- out.push("");
- const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null);
- const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null);
- if (withSeams.length) {
- out.push(
- `**Headline** — new labels per chunk after the first, over ${withSeams.length} multi-chunk video(s): ` +
- `mean ${fixed(sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length)}, ` +
- `max ${fixed(Math.max(...withSeams.map((m) => m.newLabelsPerChunk ?? 0)))}.`,
- );
- out.push("");
- out.push(
- "Read it WITH the top-share column: one label at ~100% scores a perfect 0 here",
- "and is degenerate, not good.",
- );
- out.push("");
- }
- const degenerate = scored.filter((m) => m.labelsPerVideo === 1);
- out.push(
- `Single-label (degenerate-risk) videos: ${degenerate.length} of ${scored.length}. ` +
- `Videos whose chunk reconstruction disagreed with the record: ${
- scored.filter((m) => !m.chunkReconstructionOk).length
- }.`,
- );
- out.push("");
-
- out.push("## Coverage and label quality");
- out.push("");
- out.push("| video | segments | mark-free chunks | coverage | max gap | median gap | non-ascii labels |");
- out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: |");
- for (const r of results) {
- const m = r.metrics;
- out.push(
- `| ${r.slug} | ${m?.segments ?? "—"} | ${
- m ? (m.markFreeChunks === null ? "n/a" : m.markFreeChunks) : "—"
- } | ${m ? pct(m.coverageRate) : "—"} | ${m ? toHms(Math.round(m.maxGapSeconds)) : "—"} | ${
- m ? toHms(Math.round(m.medianGapSeconds)) : "—"
- } | ${m ? pct(m.nonAsciiLabelRate) : "—"} |`,
- );
- }
- out.push("");
- out.push(
- "Coverage is ~1.0 BY CONSTRUCTION for this lane — marks are tiled, each running",
- "to the next change — so it is a structural check, not a quality signal. The",
- "honest coverage metric here is mark-free chunks.",
- );
- out.push("");
- } else {
- out.push("## Diarized lane — the claim under test is ONE call per video");
- out.push("");
- out.push("| video | calls | clusters offered | clusters named | confidences |");
- out.push("| --- | ---: | ---: | ---: | --- |");
- for (const r of results) {
- out.push(
- `| ${r.slug} | ${r.calls.length} | ${r.clustersOffered ?? "—"} | ${r.clustersNamed ?? "—"} | ${
- (r.confidences ?? []).map((c) => c.toFixed(2)).join(", ") || "—"
- } |`,
- );
- }
- out.push("");
- const calls = sum(results.map((r) => r.calls.length));
- out.push(
- `**Calls per video: ${(calls / Math.max(1, results.length)).toFixed(2)}** ` +
- `(${calls} call(s) over ${results.length} video(s)). The claim holds iff this is 1.00.`,
- );
- out.push("");
- }
-
- const warnings: Record<string, number> = {};
- for (const r of results) {
- for (const [code, n] of Object.entries(r.warningsByCode)) {
- warnings[code] = (warnings[code] ?? 0) + n;
- }
- }
- out.push(
- `Warnings by code: ${
- Object.keys(warnings).length
- ? Object.entries(warnings).map(([k, v]) => `${k} ${v}`).join(", ")
- : "none"
- }`,
- );
- out.push("");
-
- out.push("## Cost");
- out.push("");
- out.push(
- `Measured over the ${cost.videosMeasured} video(s) this run actually attributed ` +
- `(${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s), ` +
- `${cost.callsWithEngineTiming} with engine timing). Sample density ` +
- `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` +
- `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`,
- );
- out.push("");
- out.push("| quantity | value |");
- out.push("| --- | ---: |");
- out.push(`| s/chunk, wall (video wall ÷ chunks) | ${fixed(cost.secondsPerChunkWall, 1)} |`);
- out.push(`| s/chunk, per-call wall | ${fixed(cost.secondsPerChunkCallWall, 1)} |`);
- out.push(`| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} |`);
- out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} |`);
- out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} |`);
- out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} |`);
- out.push(
- `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} |`,
- );
- out.push("");
-
- if (args.method === "text-only" && args.census) {
- const idle = cost.secondsPerChunkEngineExLoad;
- const contended = cost.secondsPerChunkWall;
- out.push("## Corpus projection");
- out.push("");
- out.push(
- `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` +
- `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` +
- (args.census.estimated ? `, ${args.census.estimated} estimated` : "") +
- (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") +
- ".",
- );
- out.push("");
- out.push(
- `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` +
- `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended**` +
- ` — from n = ${cost.videosMeasured} video(s) / ${cost.chunks} chunk(s), one box, one hour.`,
- );
- out.push("");
- out.push(
- "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that",
- "has completed 0.17% of its own ~194k calls, with nothing arbitrating between",
- "them. One lane, no parallelism.",
- );
- out.push("");
- }
-
- out.push("## Box");
- out.push("");
- out.push(`Start: \`${JSON.stringify(args.boxStart)}\``);
- out.push("");
- out.push(`End: \`${JSON.stringify(args.boxEnd)}\``);
- out.push("");
- if (cost.unparsedLogLines > 0) {
- out.push(
- `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`,
- );
- out.push("");
- }
-
- out.push("## Units");
- out.push("");
- out.push(
- "- The unit of text-only work is the **chunk** (one model call). Seconds-per-",
- " audio-hour is not a unit here: chunk density varies 4× across this corpus, and",
- " pricing in it is the error behind a retracted throughput headline. This report",
- " does not print one.",
- "- The unit of diarized work is **calls per video**.",
- "- Wall and engine are different quantities and are never substituted for each",
- " other. Wall is the realistic ceiling on this shared box; engine excl. load is",
- " the floor.",
- "- Every rate carries its denominator. At n ≈ 8 videos a bare percentage would be",
- " a lie of precision.",
- );
- out.push("");
- return out.join("\n");
-}
-
-// ---------------------------------------------------------------------------
-
-async function main(): Promise<void> {
- const flags = parseFlags(process.argv.slice(2));
- const paths = getPaths();
- const sampleSpec = flags.sample ?? "bakeoff";
- const method: AttributionMethod = flags.method === "diarized" ? "diarized" : "text-only";
- const dryRun = flags["dry-run"] === "true";
- const force = flags.force === "true";
- const asJson = flags.json === "true";
- const label = flags.label ?? `${method}-${sampleSpec.replace(/[^a-zA-Z0-9]+/g, "-")}`;
-
- // THE LANE IS ENABLED IN MEMORY. settings.json has no `attribution` key at
- // all, so without this override every video returns "disabled" and the run
- // reports a flawless zero-cost hour. Nothing here is persisted.
- const pilotSettings: AttributionSettings = {
- enabled: true,
- appId: OLLAMA_DIGEST_APP_ID,
- model: "",
- diarizedEnabled: method === "diarized",
- textOnlyEnabled: method === "text-only",
- promptVersion: ATTRIBUTION_PROMPT_VERSION,
- };
- const { app, config, modelRequested } = resolveAttributionTarget(method, pilotSettings);
- const maxCues = maxCuesForContext(config.numCtx);
-
- const videos = await resolveSample(sampleSpec, paths);
- const priced: PricedVideo[] = [];
- const skipped: string[] = [];
- for (const v of videos) {
- const p = await price(v, maxCues);
- if (p) priced.push(p);
- else skipped.push(v.slug);
- }
-
- const totalChunks = sum(priced.map((p) => p.chunks));
- const totalAudio = sum(priced.map((p) => p.durationSeconds));
- const census = await readChunkCensus(paths, maxCues);
-
- console.log(
- `Attribution pilot — lane ${method}, sample ${sampleSpec}, app ${app.id}, ` +
- `model ${modelRequested}, numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`,
- );
- console.log(`Settings override (in memory only): ${JSON.stringify(pilotSettings)}`);
- console.log("");
- for (const p of priced) {
- console.log(
- ` ${p.slug}${p.bucket ? ` [${p.bucket}]` : ""} ${toHms(p.durationSeconds)} ` +
- `${p.cues.length} cues → ${p.chunks} chunk(s)`,
- );
- }
- if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`);
- console.log("");
- const projectedMinutes = (totalChunks * MEASURED_SECONDS_PER_CHUNK) / 60;
- console.log(
- `${priced.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` +
- `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour. ` +
- `At the digest lane's measured ${MEASURED_SECONDS_PER_CHUNK}s/chunk that is ` +
- `~${projectedMinutes.toFixed(0)} minute(s).`,
- );
- if (census) {
- console.log(
- `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s) ` +
- `/ ${census.chunksPerAudioHour.toFixed(2)} chunks per audio-hour` +
- (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""),
- );
- }
- console.log("");
-
- if (dryRun) {
- console.log("--dry-run: priced only, no model call and no write.");
- return;
- }
-
- const boxStart = boxState();
- const startedAt = new Date().toISOString();
- const results: VideoResult[] = [];
- const pricedBySlug = new Map(priced.map((p) => [p.slug, p]));
- const records = new Map<string, AttributionRecord>();
- for (const video of priced) {
- const r = await runVideo(video, method, pilotSettings, paths, force, (m) => console.log(m));
- results.push(r);
- const record = await loadAttribution(video.dir);
- if (record) records.set(video.slug, record);
- }
- const boxEnd = boxState();
-
- const cost = summarizeCost(results);
- const report = {
- version: 1 as const,
- label,
- method,
- sample: sampleSpec,
- startedAt,
- finishedAt: new Date().toISOString(),
- engine: {
- appId: app.id,
- modelRequested,
- numCtx: config.numCtx,
- maxCues,
- overlapCues: DIGEST_OVERLAP_CUES,
- promptVersion: ATTRIBUTION_PROMPT_VERSION,
- },
- settingsOverride: pilotSettings,
- skipped,
- box: { start: boxStart, end: boxEnd },
- census,
- cost,
- baseline: { digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE },
- videos: results,
- };
-
- const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot");
- await mkdir(outDir, { recursive: true });
- const jsonPath = path.join(outDir, `${label}.json`);
- const mdPath = path.join(outDir, `${label}.md`);
- const qualityPath = path.join(outDir, `${label}-quality.md`);
- await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n");
- await writeFile(
- mdPath,
- reportMarkdown({
- label,
- method,
- sample: sampleSpec,
- model: modelRequested,
- numCtx: config.numCtx,
- maxCues,
- results,
- cost,
- census,
- boxStart,
- boxEnd,
- startedAt,
- }),
- );
- await writeFile(qualityPath, qualityMarkdown(label, method, results, pricedBySlug, records));
-
- console.log("");
- if (asJson) {
- console.log(JSON.stringify(report, null, 2));
- } else {
- console.log(
- reportMarkdown({
- label,
- method,
- sample: sampleSpec,
- model: modelRequested,
- numCtx: config.numCtx,
- maxCues,
- results,
- cost,
- census,
- boxStart,
- boxEnd,
- startedAt,
- }),
- );
- }
- console.log(`Wrote ${jsonPath}`);
- console.log(`Wrote ${mdPath}`);
- console.log(`Wrote ${qualityPath}`);
-
- // A run in which NOTHING was attributed and nothing was already fresh is the
- // silent-no-op failure: the lane was off, or every video lacked a transcript.
- // It must not exit 0 looking like a clean measurement.
- const productive = results.filter(
- (r) => r.outcome === "attributed" || r.outcome === "already-exists",
- ).length;
- if (productive === 0) {
- console.error("No video was attributed and none was already fresh — nothing was measured.");
- process.exitCode = 1;
- }
-}
-
-main().catch((err) => {
- console.error(err);
- process.exit(1);
-});
diff --git a/common/bin/diarize-backfill.ts b/common/bin/diarize-backfill.ts
@@ -6,7 +6,7 @@
// editor/app/channels/[slug]/backfillActions.ts, sends `kindIds` and nothing
// else. So there is no way to say "diarize exactly these videos" without a
// headless entrypoint, and this is the repo's established shape for corpus work
-// (digest-plan.ts, digest-bakeoff.ts, attribution-pilot.ts).
+// (digest-plan.ts, plans/tools/digest-bakeoff.ts, plans/tools/attribution-pilot.ts).
//
// WHY IT WRITES TO THE CORPUS, unlike attribution-bakeoff.ts which refuses to.
// There the sidecar was a side effect of measuring a variant that had won
diff --git a/common/bin/digest-bakeoff.ts b/common/bin/digest-bakeoff.ts
@@ -1,1425 +0,0 @@
-#!/usr/bin/env tsx
-// Score digest engine candidates against a FIXED sample, without writing a
-// single digest to disk.
-//
-// WHY THIS IS A SCRIPT AND NOT INFRASTRUCTURE. It answers one question — which
-// (model x context x timestamp mode) should carry a multi-week sweep — and the
-// answer is a number in a table, not a feature. It therefore drives digestApps +
-// digestPrompt + digestParse DIRECTLY and scores in memory. It NEVER calls
-// digestVideo or writeDigestSection, so a losing candidate cannot leave anything
-// behind in the corpus, and no freshness record has to be invalidated afterwards
-// to undo a round.
-//
-// THROUGHPUT IS A FIRST-CLASS METRIC, not a footnote. Measured: 46 s per
-// 8,194-token chunk on qwen2.5:7b, which over the corpus is ~64 days on one
-// lane. A 14B at half the context roughly doubles the chunk count and halves the
-// token rate — order 250 days. A candidate can therefore be ruled out on
-// projected sweep days alone, however good its chapters look, which is why every
-// run prints days alongside quality.
-//
-// BOUNDARY ACCURACY, AND WHY IT WAS ADDED. Every metric this harness scored
-// originally — zeroYieldRate, chaptersPerHour, rejectionRate, maxGapSeconds,
-// genericTitleRate, duplicateTitleRate — is a DEFECT COUNTER: it says how
-// malformed the output is, never whether a boundary landed in the right place.
-// So the harness could rank candidates by which was least broken and still not
-// say which segmented better. lib/boundaryScore.ts closes that with an oracle
-// the model never sees: the UPLOADER's own chapter marks from
-// metadata.info.json, present on ~12,139 videos that also have a transcript and
-// >=4 chapters. Boundaries are facts, so scoring against them reproduces
-// nothing; the uploader's chapter TITLES are expression and are never scored
-// against, only used to spot boilerplate ("Intro", "Sponsor").
-//
-// Modes:
-//
-// --pick scan the stats cache and write the fixed stratified sample
-// (plans/bakeoff/sample.json). Run ONCE. A moving sample
-// makes the comparison between rounds meaningless.
-// --pick-chapters write a sample restricted to videos that carry >=4
-// uploader chapters, so boundary accuracy is scorable. Kept
-// as a SEPARATE file (speaker-sample.json /
-// chapter-sample.json) so rounds 1-2 stay comparable.
-// --score-existing score the ai-digest.json files ALREADY on disk against
-// their uploader chapters. Zero GPU, zero writes — the way
-// to prove the scorer works before spending a generation
-// run on it.
-// (default) run the candidates over that sample and write a JSON +
-// Markdown report under plans/bakeoff/.
-//
-// Examples:
-// tsx bin/digest-bakeoff.ts --pick
-// tsx bin/digest-bakeoff.ts --pick-chapters --sample ../plans/bakeoff/chapter-sample.json
-// tsx bin/digest-bakeoff.ts --score-existing
-// tsx bin/digest-bakeoff.ts --label round1 --buckets short,medium \
-// --candidates 'qwen2.5:7b@16384,qwen3:8b@16384,gemma2:9b@16384'
-// tsx bin/digest-bakeoff.ts --label round2 --buckets long,verylong \
-// --candidates 'qwen2.5:7b@16384' --modes absolute,chunk-local
-// tsx bin/digest-bakeoff.ts --label speakers-round1 --speakers off,on \
-// --sample ../plans/bakeoff/speaker-sample.json --candidates 'qwen2.5:7b@8192'
-
-import path from "node:path";
-import { mkdir, readdir, writeFile } from "node:fs/promises";
-import { readFile } from "node:fs/promises";
-import { open } from "lmdb";
-import { getPaths } from "../lib/paths";
-import { parseFlags } from "./_parseFlags";
-import type { VideoStat } from "../lib/stats";
-import { getDigestApp } from "../lib/digestApps";
-import type { DigestAppConfig, DigestTimestampMode } from "../lib/digest";
-import {
- CHAPTER_SYSTEM_PROMPT,
- DIGEST_MINUTES_PER_CHAPTER,
- DIGEST_OVERLAP_CUES,
- buildChapterPrompt,
- chapterSchema,
- maxCuesForContext,
- toHms,
-} from "../lib/digestPrompt";
-import { parseChapters, type DigestChunkOutput } from "../lib/digestParse";
-import { chunkCuesForContext } from "../lib/transcriptWindow";
-import { transcriptToMarkdown } from "../lib/transcriptToMarkdown";
-import { readNormalizedTranscript } from "../controller/normalizeTranscript";
-import {
- CUES_JSON_FILENAME,
- isRealAudioFile,
- readUploaderChapters,
- type UploaderChapter,
-} from "../lib/videoStatus";
-import {
- BOUNDARY_TOLERANCES_SECONDS,
- isBoilerplateChapterTitle,
- scoreBoundaries,
- type BoundaryReport,
-} from "../lib/boundaryScore";
-import { loadAttribution } from "../lib/attribution-server";
-import type { AttributionRecord } from "../lib/attribution";
-import {
- ATTRIBUTION_MAX_SPEAKERS,
- ATTRIBUTION_MIN_CLUSTER_SHARE,
-} from "../lib/attributionPrompt";
-import { loadDigest } from "../lib/digest-server";
-import type { Cue } from "../lib/vtt";
-
-// ---------------------------------------------------------------------------
-// The sample
-// ---------------------------------------------------------------------------
-
-// Duration strata. Chosen to match the corpus shape recorded in FACTS.md rather
-// than to be round numbers: the >4 h bucket is 8.2% of videos but 46% of all
-// transcript tokens, so a sample that under-represents it measures the cheap
-// half of the sweep and misses the half where chunk-seam bugs live.
-const BUCKETS = ["short", "medium", "long", "verylong"] as const;
-type Bucket = (typeof BUCKETS)[number];
-
-const BUCKET_BOUNDS: Record<Bucket, { min: number; max: number; want: number }> = {
- short: { min: 5 * 60, max: 30 * 60, want: 3 },
- medium: { min: 45 * 60, max: 90 * 60, want: 2 },
- long: { min: 3 * 3600, max: 4.5 * 3600, want: 2 },
- verylong: { min: 6 * 3600, max: 14 * 3600, want: 1 },
-};
-
-type SampleVideo = {
- slug: string;
- channelSlug: string;
- videoId: string;
- videoDir: string;
- title: string;
- bucket: Bucket;
- durationSeconds: number;
- cueCount: number;
- // Non-boilerplate uploader marks counted at pick time. Recorded for the
- // record only — the scorer re-reads them from disk, because metadata can be
- // refetched and a frozen copy would let the sample and the corpus disagree.
- uploaderChapters?: number;
-};
-
-type Sample = {
- version: 1;
- pickedAt: string;
- // Corpus-wide totals, captured at pick time. The sweep-days projection is
- // computed from measured seconds-per-audio-hour times THIS number, so the
- // projection and the sample come from one scan and can't drift apart.
- corpus: {
- videosScanned: number;
- videosWithTranscript: number;
- audioHours: number;
- longTailVideos: number;
- longTailAudioHours: number;
- };
- videos: SampleVideo[];
-};
-
-function bucketFor(seconds: number): Bucket | null {
- for (const b of BUCKETS) {
- const { min, max } = BUCKET_BOUNDS[b];
- if (seconds >= min && seconds <= max) return b;
- }
- return null;
-}
-
-// Deterministic pick, so re-running --pick on an unchanged corpus reproduces the
-// same sample. No Math.random: a sample that moves between rounds is not a
-// sample, it is noise. Videos are ordered by a stable hash of the slug and the
-// first N per bucket are taken, spreading the pick across channels instead of
-// clustering on whichever channel sorts first.
-function stableHash(s: string): number {
- let h = 2166136261;
- for (let i = 0; i < s.length; i++) {
- h ^= s.charCodeAt(i);
- h = Math.imul(h, 16777619);
- }
- return h >>> 0;
-}
-
-async function pickSample(outPath: string): Promise<void> {
- const paths = getPaths();
- const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
- const statsByPath = root.openDB<
- { metaMs: number; stat: VideoStat },
- [string, string]
- >({ name: "statsByPath", encoding: "msgpack" });
-
- const byBucket = new Map<Bucket, SampleVideo[]>();
- for (const b of BUCKETS) byBucket.set(b, []);
-
- let videosScanned = 0;
- let videosWithTranscript = 0;
- let totalSeconds = 0;
- let longTailVideos = 0;
- let longTailSeconds = 0;
-
- for (const { key, value } of statsByPath.getRange()) {
- const stat = value.stat;
- videosScanned++;
- if (!stat.hasTranscript || !(stat.duration > 0)) continue;
- videosWithTranscript++;
- totalSeconds += stat.duration;
- if (stat.duration > 4 * 3600) {
- longTailVideos++;
- longTailSeconds += stat.duration;
- }
- const bucket = bucketFor(stat.duration);
- if (!bucket) continue;
- // A digest needs cues; a transcript flagged present but empty is useless
- // here and would silently shrink a stratum.
- if (!stat.cueCount || stat.cueCount < 30) continue;
- byBucket.get(bucket)!.push({
- slug: stat.slug,
- channelSlug: stat.channelSlug,
- videoId: stat.id,
- videoDir: (key as [string, string])[1],
- title: stat.title,
- bucket,
- durationSeconds: Math.round(stat.duration),
- cueCount: stat.cueCount,
- });
- }
- await root.close();
-
- const videos: SampleVideo[] = [];
- for (const b of BUCKETS) {
- const pool = byBucket.get(b)!;
- pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug));
- // One per channel first, so a stratum can't come entirely from one
- // creator's house style — a model that happens to suit one show would
- // otherwise look like a model that suits the corpus.
- const seenChannels = new Set<string>();
- const spread: SampleVideo[] = [];
- for (const v of pool) {
- if (seenChannels.has(v.channelSlug)) continue;
- seenChannels.add(v.channelSlug);
- spread.push(v);
- }
- const want = BUCKET_BOUNDS[b].want;
- const taken = (spread.length >= want ? spread : pool).slice(0, want);
- if (taken.length < want) {
- console.warn(
- `Warning: bucket ${b} wanted ${want} videos but only ${taken.length} qualify.`,
- );
- }
- videos.push(...taken);
- }
-
- const sample: Sample = {
- version: 1,
- pickedAt: new Date().toISOString(),
- corpus: {
- videosScanned,
- videosWithTranscript,
- audioHours: Math.round(totalSeconds / 3600),
- longTailVideos,
- longTailAudioHours: Math.round(longTailSeconds / 3600),
- },
- videos,
- };
- await mkdir(path.dirname(outPath), { recursive: true });
- await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`);
- console.log(
- `Scanned ${videosScanned} videos (${videosWithTranscript} with transcripts, ` +
- `${sample.corpus.audioHours} audio-hours; ${longTailVideos} over 4 h holding ` +
- `${sample.corpus.longTailAudioHours} h).`,
- );
- for (const v of videos) {
- console.log(
- ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${v.cueCount
- .toString()
- .padStart(5)} cues ${v.slug} ${v.title.slice(0, 60)}`,
- );
- }
- console.log(`Wrote ${outPath}`);
-}
-
-// ---------------------------------------------------------------------------
-// Scoring
-// ---------------------------------------------------------------------------
-
-// Titles that carry no information about what was actually said. A cheap proxy
-// for title quality: a model that segments correctly but names every section
-// "Discussion" has produced a table of contents nobody can navigate.
-const GENERIC_TITLE_RE =
- /^(the\s+)?(intro(duction)?|outro|conclusion|discussion|continued|continuation|overview|summary|recap|closing( remarks)?|opening( remarks)?|final thoughts|misc(ellaneous)?|other|general|topics?|segment|section|chapter|part)\b/i;
-const GENERIC_TITLE_TAIL_RE = /\b(part|section|segment|chapter)\s+(\d+|one|two|three|four|five|six|seven|eight|nine|ten)$/i;
-
-function isGenericTitle(title: string): boolean {
- const t = title.trim();
- return GENERIC_TITLE_RE.test(t) || GENERIC_TITLE_TAIL_RE.test(t);
-}
-
-function normalizeTitle(title: string): string {
- return title.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, " ").trim();
-}
-
-type Candidate = {
- key: string;
- model: string;
- // Absent means "the engine's default" — the same thing an unset
- // settings.digest.apps[id].numCtx means, so the default row measures exactly
- // what a default-configured sweep would do.
- numCtx?: number;
- maxCues: number;
- timestampMode: DigestTimestampMode;
- think?: boolean;
- // Render speaker labels into the transcript from attribution.json.
- //
- // FALSE MUST BE BYTE-IDENTICAL TO PRODUCTION. The speakers-off arm is not a
- // control unless it renders exactly what the sweep renders, which is why the
- // prefix hook is a no-op that returns null rather than a second renderer.
- speakers: boolean;
-};
-
-type VideoScore = {
- slug: string;
- bucket: Bucket;
- durationSeconds: number;
- chunks: number;
- chunksFailed: number;
- zeroYieldChunks: number;
- kept: number;
- maxGapSeconds: number;
- engineSeconds: number;
- inputTokens: number;
- outputTokens: number;
- warningsByCode: Record<string, number>;
- genericTitles: number;
- duplicateTitles: number;
- // Kept so a table can be sanity-checked against real output by hand, which is
- // the only way to catch a model that scores well and reads badly.
- sampleTitles: string[];
- // Boundary accuracy against the uploader's own chapter marks. Null when the
- // video carries none, which is the normal case for ~85% of the corpus.
- boundary: BoundaryReport | null;
- // How many named speakers the render actually carried, and how many kept
- // titles mention one of them.
- //
- // NAME UPTAKE IS THE SPEAKER HYPOTHESIS' OWN METRIC. speakers-off cannot
- // produce it by construction (the roster is never in the prompt), so a
- // non-zero value on the off arm means a name leaked in some other way — a
- // useful tripwire, not a score.
- speakersRendered: number;
- titlesWithSpeakerName: number;
-};
-
-type CandidateScore = {
- candidate: Candidate;
- videos: VideoScore[];
- totals: {
- videos: number;
- audioHours: number;
- chunks: number;
- chunksFailed: number;
- zeroYieldChunks: number;
- zeroYieldRate: number;
- kept: number;
- chaptersPerHour: number;
- maxGapSeconds: number;
- meanGapSeconds: number;
- genericTitleRate: number;
- duplicateTitleRate: number;
- warningsByCode: Record<string, number>;
- rejectionRate: number;
- engineSeconds: number;
- tokensPerSecond: number;
- secondsPerAudioHour: number;
- projectedSweepDays: number;
- // Boundary accuracy, pooled over the videos that had an oracle.
- //
- // POOLED, NOT AVERAGED PER VIDEO. A per-video mean lets a 4-chapter video
- // weigh as much as a 90-chapter one, so a candidate could win by doing well
- // on the shortest lists. Pooling counts matched/generated/reference across
- // the whole sample and derives precision and recall from the totals.
- boundary: {
- videosScored: number;
- referenceBoundaries: number;
- medianOffsetSeconds: number | null;
- withinThirtySecondsRate: number;
- byTolerance: {
- toleranceSeconds: number;
- matched: number;
- precision: number;
- recall: number;
- f1: number;
- }[];
- };
- speakerNameTitleRate: number;
- };
-};
-
-// ---------------------------------------------------------------------------
-// Speaker rendering (the speakers-on arm)
-// ---------------------------------------------------------------------------
-
-// A label per cue, or null where no NAMED speaker covers it.
-//
-// CONSUMES attribution.json, NEVER diarization.json. A bare cluster index
-// rendered as "Speaker 7:" is prompt cost with no semantic content, and it
-// invites chapter titles like "Speaker 7 responds". lib/diarization.ts already
-// argues a cluster index is not an identity; this honours that.
-type SpeakerContext = {
- // Indexed by position in the FULL cue array, so a chunk can slice it.
- labelByCueIndex: (string | null)[];
- roster: string[];
- // Cues that fell inside a named turn. Reported so a report can say how much
- // of the transcript the labels actually reached.
- labelledCues: number;
-};
-
-// Speakers worth rendering: the heaviest by attributed speech, capped and
-// floored exactly as attributionPrompt caps the naming call itself.
-//
-// The cap is doing real work here. Diarization over-splits (median 35 clusters
-// per video on this corpus, max 325), and 8 of the 9 attribution records that
-// existed when this was written were the REJECTED text-only pilot, one of them
-// carrying 346 "speakers" that were raw transcript fragments. Rendering that
-// unfiltered would bury the transcript in noise and measure the noise.
-function namedSpeakers(record: AttributionRecord): Map<number, string> {
- const seconds = new Map<number, number>();
- for (const seg of record.segments) {
- const d = Math.max(0, seg.end - seg.start);
- seconds.set(seg.speaker, (seconds.get(seg.speaker) ?? 0) + d);
- }
- const total = Array.from(seconds.values()).reduce((a, b) => a + b, 0);
- if (total <= 0) return new Map();
-
- const ranked = record.speakers
- .map((s) => ({ index: s.index, label: s.label?.trim() ?? "", share: (seconds.get(s.index) ?? 0) / total }))
- .filter((s) => s.label.length > 0 && s.share >= ATTRIBUTION_MIN_CLUSTER_SHARE)
- .sort((a, b) => b.share - a.share)
- .slice(0, ATTRIBUTION_MAX_SPEAKERS);
-
- return new Map(ranked.map((s) => [s.index, s.label]));
-}
-
-// Map cues onto turns by MAXIMUM OVERLAP, not by containment.
-//
-// Cue boundaries do not align to speaker turns — measured on the diarized set,
-// only 64.4% of cues fall wholly inside a named turn. Containment would leave a
-// third of the transcript unlabelled for a reason that has nothing to do with
-// speaker identity; max-overlap assigns each cue to whoever does most of the
-// talking during it, and still yields null when no named turn touches it.
-function buildSpeakerContext(
- cues: Cue[],
- record: AttributionRecord,
-): SpeakerContext {
- const named = namedSpeakers(record);
- const segments = record.segments
- .filter((s) => named.has(s.speaker))
- .sort((a, b) => a.start - b.start);
-
- const labelByCueIndex: (string | null)[] = new Array(cues.length).fill(null);
- const rosterSeen = new Set<string>();
- const roster: string[] = [];
- let labelledCues = 0;
-
- let cursor = 0;
- for (let i = 0; i < cues.length; i++) {
- const cue = cues[i];
- const cueEnd = cue.end > cue.start ? cue.end : cue.start + 1;
- // Segments are sorted, so the scan only ever moves forward.
- while (cursor < segments.length && segments[cursor].end <= cue.start) cursor++;
- let best: { label: string; overlap: number } | null = null;
- for (let j = cursor; j < segments.length; j++) {
- const seg = segments[j];
- if (seg.start >= cueEnd) break;
- const overlap = Math.min(cueEnd, seg.end) - Math.max(cue.start, seg.start);
- if (overlap > 0 && (!best || overlap > best.overlap)) {
- best = { label: named.get(seg.speaker)!, overlap };
- }
- }
- if (best) {
- labelByCueIndex[i] = best.label;
- labelledCues++;
- if (!rosterSeen.has(best.label)) {
- rosterSeen.add(best.label);
- roster.push(best.label);
- }
- }
- }
-
- return { labelByCueIndex, roster, labelledCues };
-}
-
-// Label on speaker CHANGE only, never per line.
-//
-// MEASURED COST. Per-line labels inflate a rendered chunk by ~12% of characters
-// at the median and up to ~30%, which on the ~10-tokens-per-cue budget that
-// sizes the 600-cue chunk to an 8k window is enough to start truncating — and
-// ollama truncates SILENTLY. Change-only lands at ~3.4% median. Since the
-// speaker changes on only ~26% of cues, the two carry the same information.
-//
-// A cue with no named speaker gets NO prefix and does not count as a change, so
-// a coverage gap reads as "the previous speaker continues" rather than as a
-// fake new person.
-function speakerPrefixer(
- context: SpeakerContext,
- chunkStartIndex: number,
-): (cue: Cue, index: number) => string | null {
- let previous: string | null = null;
- return (_cue, index) => {
- const label = context.labelByCueIndex[chunkStartIndex + index] ?? null;
- if (!label) return null;
- if (label === previous) return null;
- previous = label;
- return `${label}: `;
- };
-}
-
-function renderChunk(
- meta: { id: string; title: string; channel?: string; duration?: number },
- cues: Cue[],
- offsetSeconds: number,
- prefixForCue?: (cue: Cue, index: number) => string | null,
-): string {
- return transcriptToMarkdown(
- { ...meta, cues },
- {
- timestamps: true,
- includeDescription: false,
- includeTags: false,
- stampForCue: (_clock, seconds) =>
- toHms(Math.max(0, seconds - offsetSeconds)),
- // Omitted entirely on the speakers-off arm, so that arm's bytes are the
- // sweep's bytes.
- ...(prefixForCue ? { prefixForCue } : {}),
- },
- );
-}
-
-// The cast list for ONE chunk, not for the video.
-//
-// A 12-name roster on a chunk where two people speak is misleading and wastes
-// context. The preamble also has to explain the change-only convention, or the
-// model reads an unlabelled line as an unknown speaker.
-function speakerPreamble(names: string[]): string | undefined {
- if (names.length === 0) return undefined;
- return [
- "Speakers in this section (a name before a line means that speaker begins",
- "there; unlabelled lines continue the previous speaker):",
- ...names.map((n) => `- ${n}`),
- ].join("\n");
-}
-
-// Where a sample video's sidecars live. One definition, because three call
-// sites now need it (scoring, the oracle read, attribution).
-function videoDirFor(video: SampleVideo): string {
- const paths = getPaths();
- return path.join(paths.channelsDir, video.channelSlug, "data", video.videoDir);
-}
-
-// The oracle, read FRESH at score time rather than frozen into the sample.
-//
-// A video's uploader chapters can change when metadata is refetched. Freezing
-// them would let the sample and the disk disagree silently; reading them here
-// means a report always scored against what the uploader currently says.
-//
-// Boilerplate marks are dropped. "Intro"/"Sponsor"/"Outro" are boundaries in the
-// video's FURNITURE, not in its subject, and crediting a model for finding the
-// sponsor read measures the wrong thing — the same class of junk the
-// `boilerplate` context field exists to remove.
-async function uploaderBoundaries(
- video: SampleVideo,
-): Promise<{ starts: number[]; chapters: UploaderChapter[] } | null> {
- const chapters = await readUploaderChapters(videoDirFor(video));
- if (!chapters) return null;
- const kept = chapters.filter((c) => !isBoilerplateChapterTitle(c.title));
- if (kept.length === 0) return null;
- return { starts: kept.map((c) => c.start), chapters: kept };
-}
-
-async function scoreVideo(
- video: SampleVideo,
- candidate: Candidate,
- log: (m: string) => void,
-): Promise<VideoScore | null> {
- const videoDir = videoDirFor(video);
- const cuesPath = path.join(videoDir, CUES_JSON_FILENAME);
- const transcript = await readNormalizedTranscript(cuesPath);
- if (!transcript || !transcript.cues?.length) {
- log(` ${video.slug}: no transcript on disk, skipped`);
- return null;
- }
- const cues = transcript.cues;
- const chunks = chunkCuesForContext(cues, {
- maxCues: candidate.maxCues,
- overlapCues: DIGEST_OVERLAP_CUES,
- });
-
- // Speakers are loaded per VIDEO, once, even though they are rendered per
- // chunk: the roster has to be sliced to the chunk but the cue->turn mapping
- // is a whole-video computation.
- let speakerContext: SpeakerContext | null = null;
- if (candidate.speakers) {
- const record = await loadAttribution(videoDir);
- if (record) {
- speakerContext = buildSpeakerContext(cues, record);
- log(
- ` ${video.slug}: ${speakerContext.roster.length} named speaker(s), ` +
- `${pct(cues.length > 0 ? speakerContext.labelledCues / cues.length : 0)} of cues labelled`,
- );
- } else {
- // NOT an error and NOT a skip. A video with no attribution renders
- // exactly as the off arm renders it, which is the behaviour any shipped
- // version would need for the ~98% of the corpus that can never be
- // diarized. Silently degrading is the feature.
- log(` ${video.slug}: no attribution on disk, rendering without speakers`);
- }
- }
-
- // Chunk i starts at this index in the full cue array. chunkCuesForContext
- // overlaps by DIGEST_OVERLAP_CUES, so this is not i * maxCues.
- const chunkStartIndices: number[] = [];
- {
- let cursor = 0;
- for (const chunk of chunks) {
- const first = chunk[0];
- // Cues are unique by identity here, so indexOf from the last position is
- // both correct and linear overall.
- const found = cues.indexOf(first, Math.max(0, cursor - chunk.length));
- chunkStartIndices.push(found >= 0 ? found : cursor);
- cursor = (found >= 0 ? found : cursor) + chunk.length;
- }
- }
-
- const app = getDigestApp("ollama-direct");
- const config: DigestAppConfig = {
- model: candidate.model,
- numCtx: candidate.numCtx,
- temperature: 0,
- timeoutMs: 20 * 60_000,
- ...(candidate.think !== undefined ? { think: candidate.think } : {}),
- };
-
- const outputs: DigestChunkOutput[] = [];
- const score: VideoScore = {
- slug: video.slug,
- bucket: video.bucket,
- durationSeconds: video.durationSeconds,
- chunks: chunks.length,
- chunksFailed: 0,
- zeroYieldChunks: 0,
- kept: 0,
- maxGapSeconds: 0,
- engineSeconds: 0,
- inputTokens: 0,
- outputTokens: 0,
- warningsByCode: {},
- genericTitles: 0,
- duplicateTitles: 0,
- sampleTitles: [],
- boundary: null,
- speakersRendered: speakerContext?.roster.length ?? 0,
- titlesWithSpeakerName: 0,
- };
-
- for (let i = 0; i < chunks.length; i++) {
- const chunk = chunks[i];
- const startSeconds = Math.max(0, Math.floor(chunk[0].start));
- const endSeconds = Math.max(
- startSeconds,
- Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
- );
- const offset = candidate.timestampMode === "chunk-local" ? startSeconds : 0;
-
- // The roster is per-CHUNK: the names that actually appear in THIS slice.
- // A 12-name cast list over a chunk where two people speak is misleading and
- // spends context for nothing.
- let prefixForCue: ((cue: Cue, index: number) => string | null) | undefined;
- let speakerRoster: string | undefined;
- if (speakerContext) {
- const startIndex = chunkStartIndices[i];
- const namesHere: string[] = [];
- const seenHere = new Set<string>();
- for (let k = 0; k < chunk.length; k++) {
- const label = speakerContext.labelByCueIndex[startIndex + k];
- if (label && !seenHere.has(label)) {
- seenHere.add(label);
- namesHere.push(label);
- }
- }
- if (namesHere.length > 0) {
- prefixForCue = speakerPrefixer(speakerContext, startIndex);
- speakerRoster = speakerPreamble(namesHere);
- }
- }
-
- const promptInput = {
- title: transcript.title || video.videoId,
- channel: transcript.channel || video.channelSlug,
- startSeconds,
- endSeconds,
- transcript: renderChunk(
- {
- id: transcript.id,
- title: transcript.title,
- channel: transcript.channel,
- duration: transcript.duration,
- },
- chunk,
- offset,
- prefixForCue,
- ),
- timestampMode: candidate.timestampMode,
- ...(speakerRoster ? { speakerRoster } : {}),
- };
- try {
- const result = await app.run({
- system: CHAPTER_SYSTEM_PROMPT,
- prompt: buildChapterPrompt(promptInput),
- schema: chapterSchema(endSeconds - startSeconds),
- config,
- });
- score.engineSeconds += result.durationMs / 1000;
- score.inputTokens += result.inputTokens ?? 0;
- score.outputTokens += result.outputTokens ?? 0;
- outputs.push({
- index: i,
- startSeconds,
- endSeconds,
- data: result.data,
- timestampMode: candidate.timestampMode,
- });
- } catch (err) {
- score.chunksFailed++;
- score.warningsByCode["chunk-failed"] =
- (score.warningsByCode["chunk-failed"] ?? 0) + 1;
- log(` ${video.slug} chunk ${i + 1}/${chunks.length} failed: ${(err as Error).message}`);
- }
- }
-
- // ZERO-YIELD CHUNKS — the headline defect metric. Computed by parsing each
- // chunk ALONE, because the merged parse cannot attribute a kept chapter back
- // to the chunk that produced it, and "chunk 3 produced nothing" is precisely
- // the failure this whole stage exists to fix. A chunk the engine never
- // answered counts too: from the corpus's point of view the outcome is the
- // same, an interval of the video with no chapters in it.
- for (const out of outputs) {
- if (parseChapters([out], cues).chapters.length === 0) score.zeroYieldChunks++;
- }
- score.zeroYieldChunks += score.chunksFailed;
-
- const parsed = parseChapters(outputs, cues);
- score.kept = parsed.chapters.length;
- for (const w of parsed.warnings) {
- score.warningsByCode[w.code] = (score.warningsByCode[w.code] ?? 0) + 1;
- }
-
- // MAX COVERAGE GAP — catches "summarised the tail, skipped the head". The
- // leading gap (0 -> first chapter) and the trailing one (last chapter -> end)
- // are included deliberately: a video whose chapters all sit in the last 20
- // minutes has a coverage failure that consecutive-gap-only scoring hides.
- const starts = parsed.chapters.map((c) => c.start);
- const bounds = [0, ...starts, video.durationSeconds];
- for (let i = 1; i < bounds.length; i++) {
- score.maxGapSeconds = Math.max(score.maxGapSeconds, bounds[i] - bounds[i - 1]);
- }
-
- const seen = new Set<string>();
- for (const c of parsed.chapters) {
- if (isGenericTitle(c.title)) score.genericTitles++;
- const n = normalizeTitle(c.title);
- if (seen.has(n)) score.duplicateTitles++;
- seen.add(n);
- }
- score.sampleTitles = parsed.chapters.slice(0, 8).map((c) => `${c.clock} ${c.title}`);
-
- // BOUNDARY ACCURACY against the uploader. The one metric here that measures
- // whether the segmentation is RIGHT rather than well-formed.
- const oracle = await uploaderBoundaries(video);
- if (oracle) {
- score.boundary = scoreBoundaries(
- oracle.starts,
- parsed.chapters.map((c) => c.start),
- );
- }
-
- // SPEAKER NAME UPTAKE. Counted against the roster the render actually used,
- // so the off arm can be checked for leakage: a non-zero value there means a
- // name reached the titles by some route other than the labels.
- if (speakerContext && speakerContext.roster.length > 0) {
- // WORD-BOUNDARY MATCHING, not substring. normalizeTitle collapses to
- // space-separated words, so a substring test would count "ghost stories" as
- // containing the speaker "Host" — and role labels like "Host" and "Caller"
- // are exactly what the diarized lane produces when the transcript does not
- // support a real name, so the false positives would not be rare.
- const needles = speakerContext.roster
- .map((n) => normalizeTitle(n))
- .filter((n) => n.length >= 3)
- .map((n) => ` ${n} `);
- for (const c of parsed.chapters) {
- const t = ` ${normalizeTitle(c.title)} `;
- if (needles.some((n) => t.includes(n))) score.titlesWithSpeakerName++;
- }
- }
-
- log(
- ` ${video.slug} [${video.bucket}] ${chunks.length} chunk(s) → ${score.kept} chapter(s), ` +
- `${score.zeroYieldChunks} zero-yield, ${Math.round(score.engineSeconds)}s engine` +
- (score.boundary
- ? `, boundary F1@30 ${pct(score.boundary.scores[0]?.f1 ?? 0)} ` +
- `(${score.boundary.referenceCount} uploader mark(s))`
- : ", no uploader chapters"),
- );
- return score;
-}
-
-function aggregate(candidate: Candidate, videos: VideoScore[]): CandidateScore {
- const sum = (f: (v: VideoScore) => number): number =>
- videos.reduce((a, v) => a + f(v), 0);
- const audioHours = sum((v) => v.durationSeconds) / 3600;
- const chunks = sum((v) => v.chunks);
- const kept = sum((v) => v.kept);
- const engineSeconds = sum((v) => v.engineSeconds);
- const tokens = sum((v) => v.inputTokens + v.outputTokens);
-
- const warningsByCode: Record<string, number> = {};
- for (const v of videos) {
- for (const [code, n] of Object.entries(v.warningsByCode)) {
- warningsByCode[code] = (warningsByCode[code] ?? 0) + n;
- }
- }
- const rejections = Object.entries(warningsByCode)
- .filter(([code]) => code !== "seam-duplicate")
- .reduce((a, [, n]) => a + n, 0);
-
- const secondsPerAudioHour = audioHours > 0 ? engineSeconds / audioHours : 0;
-
- // Pooled boundary accuracy. See the comment on CandidateScore.totals.boundary
- // for why this is not a mean of per-video F1s.
- const scored = videos.filter((v) => v.boundary);
- const referenceBoundaries = scored.reduce((a, v) => a + v.boundary!.referenceCount, 0);
- const generatedBoundaries = scored.reduce((a, v) => a + v.boundary!.generatedCount, 0);
- const within30 = scored.reduce((a, v) => a + v.boundary!.withinThirtySeconds, 0);
- const allOffsets: number[] = [];
- for (const v of scored) {
- // medianOffsetSeconds is per video; pooling the medians is not a median, so
- // the report quotes the median OF the per-video medians and says so.
- if (v.boundary!.medianOffsetSeconds !== null) {
- allOffsets.push(v.boundary!.medianOffsetSeconds);
- }
- }
- allOffsets.sort((a, b) => a - b);
-
- const byTolerance = BOUNDARY_TOLERANCES_SECONDS.map((tol) => {
- const matched = scored.reduce(
- (a, v) => a + (v.boundary!.scores.find((s) => s.toleranceSeconds === tol)?.matched ?? 0),
- 0,
- );
- const precision = generatedBoundaries > 0 ? matched / generatedBoundaries : 0;
- const recall = referenceBoundaries > 0 ? matched / referenceBoundaries : 0;
- return {
- toleranceSeconds: tol,
- matched,
- precision: round(precision, 4),
- recall: round(recall, 4),
- f1: round(precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0, 4),
- };
- });
-
- return {
- candidate,
- videos,
- totals: {
- videos: videos.length,
- audioHours: round(audioHours, 2),
- chunks,
- chunksFailed: sum((v) => v.chunksFailed),
- zeroYieldChunks: sum((v) => v.zeroYieldChunks),
- zeroYieldRate: chunks > 0 ? round(sum((v) => v.zeroYieldChunks) / chunks, 4) : 0,
- kept,
- chaptersPerHour: audioHours > 0 ? round(kept / audioHours, 2) : 0,
- maxGapSeconds: videos.reduce((a, v) => Math.max(a, v.maxGapSeconds), 0),
- meanGapSeconds:
- videos.length > 0 ? Math.round(sum((v) => v.maxGapSeconds) / videos.length) : 0,
- genericTitleRate: kept > 0 ? round(sum((v) => v.genericTitles) / kept, 4) : 0,
- duplicateTitleRate: kept > 0 ? round(sum((v) => v.duplicateTitles) / kept, 4) : 0,
- warningsByCode,
- // Rejections per kept chapter — the ratio that says how much of what the
- // model produced the guards had to throw away.
- rejectionRate: kept + rejections > 0 ? round(rejections / (kept + rejections), 4) : 0,
- engineSeconds: Math.round(engineSeconds),
- tokensPerSecond: engineSeconds > 0 ? round(tokens / engineSeconds, 1) : 0,
- secondsPerAudioHour: Math.round(secondsPerAudioHour),
- projectedSweepDays: 0, // filled in once corpus hours are known
- boundary: {
- videosScored: scored.length,
- referenceBoundaries,
- medianOffsetSeconds:
- allOffsets.length > 0 ? allOffsets[Math.floor(allOffsets.length / 2)] : null,
- withinThirtySecondsRate:
- referenceBoundaries > 0 ? round(within30 / referenceBoundaries, 4) : 0,
- byTolerance,
- },
- speakerNameTitleRate: kept > 0 ? round(sum((v) => v.titlesWithSpeakerName) / kept, 4) : 0,
- },
- };
-}
-
-function round(n: number, places: number): number {
- const f = 10 ** places;
- return Math.round(n * f) / f;
-}
-
-// ---------------------------------------------------------------------------
-// Report
-// ---------------------------------------------------------------------------
-
-function markdownReport(
- label: string,
- sample: Sample,
- buckets: Bucket[],
- scores: CandidateScore[],
-): string {
- const lines: string[] = [];
- lines.push(`# Digest bake-off — ${label}`);
- lines.push("");
- lines.push(
- `Sample: ${scores[0]?.totals.videos ?? 0} video(s) from \`plans/bakeoff/sample.json\`` +
- ` (buckets: ${buckets.join(", ")}), ${scores[0]?.totals.audioHours ?? 0} audio-hours.`,
- );
- lines.push(
- `Sweep days are projected as measured seconds-per-audio-hour x ` +
- `${sample.corpus.audioHours} corpus audio-hours, one lane, no parallelism.`,
- );
- lines.push("");
- lines.push(
- "| Candidate | Zero-yield chunks | Chapters/h | Rejection rate | Max gap | Generic | Dup | tok/s | s per audio-h | **Sweep days** |",
- );
- lines.push(
- "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
- );
- for (const s of scores) {
- const t = s.totals;
- lines.push(
- `| \`${s.candidate.key}\` | ${t.zeroYieldChunks}/${t.chunks} (${pct(t.zeroYieldRate)}) | ` +
- `${t.chaptersPerHour} | ${pct(t.rejectionRate)} | ${toHms(t.maxGapSeconds)} | ` +
- `${pct(t.genericTitleRate)} | ${pct(t.duplicateTitleRate)} | ${t.tokensPerSecond} | ` +
- `${t.secondsPerAudioHour} | **${t.projectedSweepDays}** |`,
- );
- }
- lines.push("");
- lines.push("## Boundary accuracy vs the uploader's own chapters");
- lines.push("");
- lines.push(
- "The oracle is `metadata.info.json.chapters` — marks a human authored while",
- "watching, who never saw our prompt. Boilerplate marks (\"Intro\", \"Sponsor\")",
- "and the boundary at 00:00 are dropped before scoring: neither carries",
- "segmentation information, and the origin would be a free hit for every",
- "candidate. Matching is one-to-one and closest-pair-first, so a cluster of",
- "boundaries around one uploader mark scores one match, not many.",
- );
- lines.push("");
- lines.push(
- "| Candidate | Videos scored | Uploader marks | Median offset | Within 30s | P@30 | R@30 | **F1@30** | F1@60 | Name-in-title |",
- );
- lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
- for (const s of scores) {
- const b = s.totals.boundary;
- const t30 = b.byTolerance.find((x) => x.toleranceSeconds === 30);
- const t60 = b.byTolerance.find((x) => x.toleranceSeconds === 60);
- lines.push(
- `| \`${s.candidate.key}\` | ${b.videosScored} | ${b.referenceBoundaries} | ` +
- `${b.medianOffsetSeconds === null ? "—" : `${b.medianOffsetSeconds}s`} | ` +
- `${pct(b.withinThirtySecondsRate)} | ${pct(t30?.precision ?? 0)} | ${pct(t30?.recall ?? 0)} | ` +
- `**${pct(t30?.f1 ?? 0)}** | ${pct(t60?.f1 ?? 0)} | ${pct(s.totals.speakerNameTitleRate)} |`,
- );
- }
- lines.push("");
- lines.push("## Rejections by guard");
- lines.push("");
- const codes = Array.from(
- new Set(scores.flatMap((s) => Object.keys(s.totals.warningsByCode))),
- ).sort();
- lines.push(`| Candidate | ${codes.join(" | ")} |`);
- lines.push(`| --- | ${codes.map(() => "---").join(" | ")} |`);
- for (const s of scores) {
- lines.push(
- `| \`${s.candidate.key}\` | ${codes
- .map((c) => s.totals.warningsByCode[c] ?? 0)
- .join(" | ")} |`,
- );
- }
- lines.push("");
- lines.push("## Per-video");
- lines.push("");
- lines.push(
- "| Candidate | Video | Bucket | Chunks | Zero-yield | Chapters | Max gap | Engine s | Marks | F1@30 | Speakers |",
- );
- lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
- for (const s of scores) {
- for (const v of s.videos) {
- const f1 = v.boundary?.scores.find((x) => x.toleranceSeconds === 30)?.f1;
- lines.push(
- `| \`${s.candidate.key}\` | \`${v.slug}\` | ${v.bucket} | ${v.chunks} | ` +
- `${v.zeroYieldChunks} | ${v.kept} | ${toHms(v.maxGapSeconds)} | ${Math.round(v.engineSeconds)} | ` +
- `${v.boundary?.referenceCount ?? "—"} | ${f1 === undefined ? "—" : pct(f1)} | ` +
- `${v.speakersRendered || "—"} |`,
- );
- }
- }
- lines.push("");
- lines.push("## Sample output (first chapters per video)");
- lines.push("");
- for (const s of scores) {
- lines.push(`### \`${s.candidate.key}\``);
- lines.push("");
- for (const v of s.videos) {
- lines.push(`**${v.slug}** (${v.bucket}, ${toHms(v.durationSeconds)})`);
- lines.push("");
- for (const t of v.sampleTitles) lines.push(`- ${t}`);
- if (v.sampleTitles.length === 0) lines.push("- _(nothing survived the guards)_");
- lines.push("");
- }
- }
- return `${lines.join("\n")}\n`;
-}
-
-function pct(v: number): string {
- return `${Math.round(v * 1000) / 10}%`;
-}
-
-// ---------------------------------------------------------------------------
-// Main
-// ---------------------------------------------------------------------------
-
-// "model@ctx" or "model@ctx:think" / "model@ctx:nothink".
-//
-// maxCues is DERIVED from the context rather than configured: the 1200-cue
-// default is sized for a 16k window (~10 tokens/cue -> ~12k tokens of transcript
-// plus room for prompt and response), so halving the context must halve the
-// slice or every call silently truncates — the exact failure that made the first
-// smoke test summarize a fragment.
-function parseCandidate(
- spec: string,
- mode: DigestTimestampMode,
- speakers: boolean,
- maxCuesOverride: number | null,
-): Candidate {
- const [modelPart, rest] = spec.split("@");
- const [ctxPart, thinkPart] = (rest ?? "").split(":");
- const numCtx = ctxPart ? Number(ctxPart) : undefined;
- // The SAME derivation production uses, not a parallel copy — otherwise the
- // bake-off scores a chunk size the sweep would never actually run.
- //
- // --max-cues breaks that tie deliberately, for one reason: a paired A/B needs
- // HEADROOM. Speaker labels add ~10% to the prompt, and on a pool of long VODs
- // the un-labelled chunk is already near the window — so at the derived size
- // the speakers-on arm would truncate where speakers-off did not, and the
- // measurement would be of truncation. Raising numCtx while pinning maxCues
- // gives both arms the same cue count with room to spare. BOTH ARMS ALWAYS GET
- // THE SAME VALUE; the report records it.
- const maxCues = maxCuesOverride ?? maxCuesForContext(numCtx);
- return {
- // The speaker axis and a cue override only show in the key when set, so
- // existing round labels keep reading the way rounds 1-2 wrote them.
- key:
- `${modelPart}@${numCtx ?? "default"}/${mode}` +
- `${maxCuesOverride ? `/${maxCuesOverride}cues` : ""}${speakers ? "/speakers" : ""}`,
- model: modelPart,
- ...(numCtx ? { numCtx } : {}),
- maxCues,
- timestampMode: mode,
- speakers,
- ...(thinkPart === "think"
- ? { think: true }
- : thinkPart === "nothink"
- ? { think: false }
- : {}),
- };
-}
-
-// ---------------------------------------------------------------------------
-// The chapter-oracle sample
-// ---------------------------------------------------------------------------
-
-// Pick a sample restricted to videos that can actually be SCORED — i.e. that
-// carry >=`minChapters` non-boilerplate uploader marks.
-//
-// WHY IT DOES NOT READ EVERY METADATA FILE. metadata.info.json runs to ~100 KB
-// and only ~19% of videos carry chapters, so parsing all 77k to find them would
-// read several GB to answer a question a stable-hash walk answers in a few
-// hundred reads: candidates are visited in the same deterministic order
-// pickSample uses, and the walk stops as soon as every stratum is full.
-//
-// KEPT IN A SEPARATE FILE from sample.json, deliberately. Rounds 1 and 2 are
-// scored against that sample; repointing it would silently invalidate the
-// comparison this harness exists to protect.
-async function pickChapterSample(
- outPath: string,
- opts: { minChapters: number; perBucket: number | null; requireAudio: boolean },
-): Promise<void> {
- const paths = getPaths();
- const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
- const statsByPath = root.openDB<
- { metaMs: number; stat: VideoStat },
- [string, string]
- >({ name: "statsByPath", encoding: "msgpack" });
-
- const byBucket = new Map<Bucket, SampleVideo[]>();
- for (const b of BUCKETS) byBucket.set(b, []);
-
- let videosScanned = 0;
- let videosWithTranscript = 0;
- let totalSeconds = 0;
- let longTailVideos = 0;
- let longTailSeconds = 0;
-
- for (const { key, value } of statsByPath.getRange()) {
- const stat = value.stat;
- videosScanned++;
- if (!stat.hasTranscript || !(stat.duration > 0)) continue;
- videosWithTranscript++;
- totalSeconds += stat.duration;
- if (stat.duration > 4 * 3600) {
- longTailVideos++;
- longTailSeconds += stat.duration;
- }
- const bucket = bucketFor(stat.duration);
- if (!bucket) continue;
- if (!stat.cueCount || stat.cueCount < 30) continue;
- byBucket.get(bucket)!.push({
- slug: stat.slug,
- channelSlug: stat.channelSlug,
- videoId: stat.id,
- videoDir: (key as [string, string])[1],
- title: stat.title,
- bucket,
- durationSeconds: Math.round(stat.duration),
- cueCount: stat.cueCount,
- });
- }
- await root.close();
-
- const videos: SampleVideo[] = [];
- for (const b of BUCKETS) {
- const pool = byBucket.get(b)!;
- pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug));
- const want = opts.perBucket ?? BUCKET_BOUNDS[b].want;
- const seenChannels = new Set<string>();
- const taken: SampleVideo[] = [];
- let probed = 0;
- for (const v of pool) {
- if (taken.length >= want) break;
- // One per channel first, for the same reason pickSample does it: a
- // stratum drawn from one creator measures a house style.
- if (seenChannels.has(v.channelSlug)) continue;
- probed++;
- const oracle = await uploaderBoundaries(v);
- if (!oracle || oracle.starts.length < opts.minChapters) continue;
- if (opts.requireAudio) {
- // isRealAudioFile, not an extension test of my own: it is the same
- // predicate the cleanup and diarization lanes use, so "has audio" means
- // here exactly what it means to the job that would diarize it.
- const files = await readdir(videoDirFor(v)).catch(() => [] as string[]);
- if (!files.some((f) => isRealAudioFile(f))) continue;
- }
- seenChannels.add(v.channelSlug);
- taken.push({ ...v, uploaderChapters: oracle.starts.length });
- }
- if (taken.length < want) {
- console.warn(
- `Warning: bucket ${b} wanted ${want} scorable videos but only ${taken.length} qualify ` +
- `(probed ${probed} candidates).`,
- );
- }
- videos.push(...taken);
- }
-
- const sample: Sample = {
- version: 1,
- pickedAt: new Date().toISOString(),
- corpus: {
- videosScanned,
- videosWithTranscript,
- audioHours: Math.round(totalSeconds / 3600),
- longTailVideos,
- longTailAudioHours: Math.round(longTailSeconds / 3600),
- },
- videos,
- };
- await mkdir(path.dirname(outPath), { recursive: true });
- await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`);
- for (const v of videos) {
- console.log(
- ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${String(v.uploaderChapters).padStart(3)} marks ` +
- `${v.slug} ${v.title.slice(0, 55)}`,
- );
- }
- console.log(`Wrote ${outPath} (${videos.length} scorable video(s))`);
-}
-
-// ---------------------------------------------------------------------------
-// Free scorer smoke test
-// ---------------------------------------------------------------------------
-
-// Score the digests ALREADY on disk against their uploader chapters.
-//
-// This is the cheapest honest check available: no model call, no write, and it
-// answers "does the metric move, and is the plumbing right" before a generation
-// run is spent finding out. If this prints nonsense — every F1 at 0 or 1 — the
-// scorer is wrong and no A/B built on it would mean anything.
-async function scoreExisting(minChapters: number, limit: number): Promise<void> {
- const paths = getPaths();
- const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
- const statsByPath = root.openDB<
- { metaMs: number; stat: VideoStat },
- [string, string]
- >({ name: "statsByPath", encoding: "msgpack" });
-
- const candidates: SampleVideo[] = [];
- for (const { key, value } of statsByPath.getRange()) {
- const stat = value.stat;
- if (!stat.hasTranscript || !(stat.duration > 0)) continue;
- candidates.push({
- slug: stat.slug,
- channelSlug: stat.channelSlug,
- videoId: stat.id,
- videoDir: (key as [string, string])[1],
- title: stat.title,
- bucket: bucketFor(stat.duration) ?? "short",
- durationSeconds: Math.round(stat.duration),
- cueCount: stat.cueCount ?? 0,
- });
- }
- await root.close();
-
- const rows: {
- slug: string;
- marks: number;
- generated: number;
- medianOffset: number | null;
- precision30: number;
- recall30: number;
- recall60: number;
- f1at30: number;
- f1at60: number;
- }[] = [];
-
- for (const v of candidates) {
- if (rows.length >= limit) break;
- const dir = videoDirFor(v);
- const digest = await loadDigest(dir);
- const items = digest?.sections?.chapters?.items ?? [];
- if (items.length === 0) continue;
- const oracle = await uploaderBoundaries(v);
- if (!oracle || oracle.starts.length < minChapters) continue;
- const report = scoreBoundaries(oracle.starts, items.map((c) => c.start));
- if (report.referenceCount === 0) continue;
- const s30 = report.scores.find((s) => s.toleranceSeconds === 30);
- const s60 = report.scores.find((s) => s.toleranceSeconds === 60);
- rows.push({
- slug: v.slug,
- marks: report.referenceCount,
- generated: report.generatedCount,
- medianOffset: report.medianOffsetSeconds,
- precision30: s30?.precision ?? 0,
- recall30: s30?.recall ?? 0,
- recall60: s60?.recall ?? 0,
- f1at30: s30?.f1 ?? 0,
- f1at60: s60?.f1 ?? 0,
- });
- }
-
- if (rows.length === 0) {
- console.log(
- `No video on disk has both a digest and >=${minChapters} non-boilerplate uploader chapters.`,
- );
- return;
- }
-
- console.log(
- `Scored ${rows.length} existing digest(s) against uploader chapters (no model calls, no writes).\n`,
- );
- console.log(
- " slug marks gen medOff P@30 R@30 R@60 F1@30 F1@60",
- );
- for (const r of rows) {
- console.log(
- ` ${r.slug.padEnd(36).slice(0, 36)} ${String(r.marks).padStart(5)} ` +
- `${String(r.generated).padStart(3)} ${String(r.medianOffset ?? "—").padStart(6)} ` +
- `${pct(r.precision30).padStart(5)} ${pct(r.recall30).padStart(5)} ` +
- `${pct(r.recall60).padStart(5)} ${pct(r.f1at30).padStart(5)} ${pct(r.f1at60).padStart(5)}`,
- );
- }
- const sumOf = (f: (r: (typeof rows)[number]) => number): number =>
- rows.reduce((a, r) => a + f(r), 0);
- // POOLED, not a mean of per-video rates: a 3-mark video must not weigh the
- // same as a 29-mark one.
- const marks = sumOf((r) => r.marks);
- const generated = sumOf((r) => r.generated);
- const matched30 = sumOf((r) => r.recall30 * r.marks);
- const matched60 = sumOf((r) => r.recall60 * r.marks);
- const p30 = generated > 0 ? matched30 / generated : 0;
- const r30 = marks > 0 ? matched30 / marks : 0;
- console.log(
- `\n POOLED: ${marks} uploader mark(s), ${generated} generated. ` +
- `P@30 ${pct(p30)}, R@30 ${pct(r30)}, R@60 ${pct(marks > 0 ? matched60 / marks : 0)}, ` +
- `F1@30 ${pct(p30 + r30 > 0 ? (2 * p30 * r30) / (p30 + r30) : 0)}`,
- );
- console.log(
- ` NOTE: the digest targets one chapter per ${DIGEST_MINUTES_PER_CHAPTER} minutes and so is\n` +
- ` deliberately DENSER than uploader chaptering (${generated} vs ${marks} here). Precision is\n` +
- ` therefore partly a density artifact; recall is the half that answers "did it find the\n` +
- ` human's boundaries". Compare candidates on recall AND on chapters/h together.`,
- );
-}
-
-async function main(): Promise<void> {
- const flags = parseFlags(process.argv.slice(2));
- const outDir = flags.outDir ?? path.join(process.cwd(), "..", "plans", "bakeoff");
- const samplePath = flags.sample ?? path.join(outDir, "sample.json");
- const minChapters = flags["min-chapters"] ? Number(flags["min-chapters"]) : 4;
-
- if (flags.pick === "true") {
- await pickSample(samplePath);
- return;
- }
-
- if (flags["pick-chapters"] === "true") {
- await pickChapterSample(samplePath, {
- minChapters,
- perBucket: flags["per-bucket"] ? Number(flags["per-bucket"]) : null,
- requireAudio: flags["require-audio"] === "true",
- });
- return;
- }
-
- if (flags["score-existing"] === "true") {
- await scoreExisting(minChapters, flags.limit ? Number(flags.limit) : 200);
- return;
- }
-
- const sample = JSON.parse(await readFile(samplePath, "utf8")) as Sample;
- const label = flags.label ?? "round";
- const buckets = (flags.buckets ?? BUCKETS.join(","))
- .split(",")
- .map((b) => b.trim())
- .filter((b): b is Bucket => (BUCKETS as readonly string[]).includes(b));
- const modes = (flags.modes ?? "absolute")
- .split(",")
- .map((m) => m.trim())
- .filter((m): m is DigestTimestampMode => m === "absolute" || m === "chunk-local");
- const specs = (flags.candidates ?? "qwen2.5:7b@16384")
- .split(",")
- .map((s) => s.trim())
- .filter(Boolean);
- // The paired A/B axis. Default "off" — the same rendering every earlier round
- // used, so omitting the flag reproduces them.
- const speakerArms = (flags.speakers ?? "off")
- .split(",")
- .map((s) => s.trim())
- .filter((s) => s === "on" || s === "off")
- .map((s) => s === "on");
-
- const videos = sample.videos.filter((v) => buckets.includes(v.bucket));
- if (videos.length === 0) throw new Error(`No sample videos in buckets ${buckets.join(",")}`);
-
- const maxCuesOverride = flags["max-cues"] ? Number(flags["max-cues"]) : null;
- const candidates: Candidate[] = [];
- for (const spec of specs) {
- for (const mode of modes) {
- for (const speakers of speakerArms) {
- candidates.push(parseCandidate(spec, mode, speakers, maxCuesOverride));
- }
- }
- }
-
- console.log(
- `Bake-off ${label}: ${candidates.length} candidate(s) x ${videos.length} video(s) ` +
- `(${Math.round(videos.reduce((a, v) => a + v.durationSeconds, 0) / 3600)} audio-hours each).`,
- );
-
- // Up front, not just before the final write, so the per-video checkpoint below
- // has somewhere to land from the very first video.
- await mkdir(outDir, { recursive: true });
-
- const scores: CandidateScore[] = [];
- for (const candidate of candidates) {
- console.log(`\n=== ${candidate.key} (maxCues ${candidate.maxCues}) ===`);
- const startedAt = Date.now();
- const perVideo: VideoScore[] = [];
- for (const video of videos) {
- const s = await scoreVideo(video, candidate, (m) => console.log(m));
- if (s) perVideo.push(s);
- // CHECKPOINT AFTER EVERY VIDEO, because the reports are only written when
- // the whole run finishes and a long run does not reliably get there. A
- // 12-video run was OOM-killed on its last video after 26 minutes of engine
- // time and left NOTHING behind — no error, no partial, just a dead process
- // (the kill is a SIGKILL, so no handler can save it either). On a box that
- // swaps, "it completed 11 of 12" has to survive.
- await writeFile(
- path.join(outDir, `${label}.partial.json`),
- `${JSON.stringify({ label, candidate: candidate.key, videos: perVideo }, null, 2)}\n`,
- ).catch(() => {});
- }
- const agg = aggregate(candidate, perVideo);
- // Days, from measured seconds-per-audio-hour against the corpus total
- // captured in the same scan that picked the sample.
- agg.totals.projectedSweepDays = round(
- (agg.totals.secondsPerAudioHour * sample.corpus.audioHours) / 86400,
- 1,
- );
- scores.push(agg);
- console.log(
- `--- ${candidate.key}: ${agg.totals.kept} chapters, ` +
- `${agg.totals.zeroYieldChunks}/${agg.totals.chunks} zero-yield, ` +
- `${agg.totals.tokensPerSecond} tok/s, ` +
- `projected ${agg.totals.projectedSweepDays} sweep days ` +
- `(wall ${Math.round((Date.now() - startedAt) / 60000)} min)`,
- );
- }
-
- scores.sort((a, b) => a.totals.zeroYieldRate - b.totals.zeroYieldRate);
-
- await mkdir(outDir, { recursive: true });
- const jsonPath = path.join(outDir, `${label}.json`);
- const mdPath = path.join(outDir, `${label}.md`);
- await writeFile(
- jsonPath,
- `${JSON.stringify({ label, sample: samplePath, corpus: sample.corpus, buckets, scores }, null, 2)}\n`,
- );
- await writeFile(mdPath, markdownReport(label, sample, buckets, scores));
- console.log(`\nWrote ${jsonPath}\nWrote ${mdPath}`);
-}
-
-main().catch((err) => {
- console.error(err);
- process.exit(1);
-});
diff --git a/common/bin/digest-validate.ts b/common/bin/digest-validate.ts
@@ -1,6 +1,6 @@
#!/usr/bin/env tsx
// Score digests that ALREADY EXIST on disk, using the same metric definitions as
-// the bake-off harness (common/bin/digest-bakeoff.ts).
+// the bake-off harness (plans/tools/digest-bakeoff.ts).
//
// The bake-off scores IN MEMORY and writes no digests at all; this reads what a
// real sweep wrote to disk. That is the difference between "how does this
diff --git a/common/bin/migrate-to-sites.ts b/common/bin/migrate-to-sites.ts
@@ -1,26 +0,0 @@
-#!/usr/bin/env tsx
-// One-time migration from the single-site layout into sites/<id>/site.json.
-// Idempotent: a no-op once any site exists. Optional first arg overrides the
-// generated site id.
-import { getPaths } from "../lib/paths";
-import { migrateToSites } from "../controller/migrateToSites";
-
-const siteId = process.argv[2];
-migrateToSites({
- paths: getPaths(),
- siteId,
- onLog: (msg) => console.log(msg),
-})
- .then((res) => {
- if (!res.migrated) {
- console.log(`Nothing to do: ${res.reason}`);
- } else {
- console.log(
- `Migrated into site "${res.siteId}" (${res.channelCount} channels).`,
- );
- }
- })
- .catch((err) => {
- console.error(err);
- process.exit(1);
- });
diff --git a/common/lib/attributionScore.ts b/common/lib/attributionScore.ts
@@ -3,7 +3,7 @@
// way.
//
// WHY THIS IS ITS OWN MODULE. These metrics were written inside
-// bin/attribution-pilot.ts, which meant (a) the headline number that rejected
+// plans/tools/attribution-pilot.ts, which meant (a) the headline number that rejected
// the text-only lane was code nobody had ever run against a known answer, and
// (b) a second harness comparing a new prompt against the pilot's baseline would
// have had to copy them. plans/FACTS.md records the digest lane already making
diff --git a/common/lib/digest.ts b/common/lib/digest.ts
@@ -79,7 +79,7 @@ export const DEFAULT_DIGEST_MAX_CUES_PER_CHUNK = 600;
// caught all nine, which is exactly its job, but a caught error is still a lost
// chunk, and >4 h videos are 8.2% of the corpus by count and 46% of its tokens.
// chunk-local removes the large offset the model has to hold. Which mode is
-// actually better was a BAKE-OFF QUESTION (common/bin/digest-bakeoff.ts), which
+// actually better was a BAKE-OFF QUESTION (plans/tools/digest-bakeoff.ts), which
// is why both were shipped rather than one being pre-applied as a fix.
//
// IT HAS BEEN ANSWERED. Round 2 (plans/bakeoff/round2.md), qwen2.5:7b@8192 over
diff --git a/common/lib/engineTiming.ts b/common/lib/engineTiming.ts
@@ -1,7 +1,7 @@
// Parse the per-call timing digestApps already logs, so a harness can price a
// run without adding instrumentation to the engine.
//
-// WHY SHARED. bin/attribution-pilot.ts and bin/attribution-bakeoff.ts both
+// WHY SHARED. plans/tools/attribution-pilot.ts and plans/tools/attribution-bakeoff.ts both
// decide on a seconds-per-chunk threshold (the pilot's 66.2 against the digest
// lane's 11.2; round 2's bar is < 25). Two copies of this regex would let the
// two harnesses report different numbers for the same run, which is the same
diff --git a/create-archives.sh b/create-archives.sh
@@ -60,12 +60,3 @@ echo "create-archives: $TARBALL ($(( BYTES / 1024 )) KiB) at ${COMMIT:0:12}"
if [ "$BYTES" -gt 26214400 ]; then
echo "create-archives: WARNING — $TARBALL exceeds Cloudflare Pages' 25 MB per-file limit." >&2
fi
-
-# Transcript archives are too big for CF pages (25mb)
-# The comments are kept since they'll likely be useful later
-#
-# cd transcripts
-# # use xz over gz to get the file below 25MB
-# git archive --format tar main | xz -9 > ../export/public/transcripts.tar.xz
-# # Uncomment this line to use .tar.gz for compatability and compression speed, but at a significantly higher file size
-# git archive -9 --format tar.gz -o "../export/public/transcripts.tar.gz" main
diff --git a/homepage/content/docs/install.md b/homepage/content/docs/install.md
@@ -179,7 +179,6 @@ launch. The ones worth knowing:
| `WHISPER_BIN` | `whisper-cli` on `PATH` | whisper.cpp binary. |
| `WHISPER_MODEL` | a path under your home directory | whisper.cpp model file. |
| `FFMPEG_BIN` / `FFPROBE_BIN` | on `PATH` | Transcode and duration checks. |
-| `PARALLEL_TRANSCRIBE_LIMIT` | `4` | Maximum simultaneous transcription jobs. |
Pointing `TRANSCRIPTS_DIR` at a large disk before you start is the one decision
worth making early — the corpus grows to whatever your channels amount to, and
diff --git a/plans/FACTS.md b/plans/FACTS.md
@@ -214,7 +214,7 @@ Transcript shards ship the **full `cues` array inline** — `buildIndex.ts:830-8
## Digest bake-off (measured 2026-07-26)
-Harness: `common/bin/digest-bakeoff.ts`. Fixed stratified sample in
+Harness: `plans/tools/digest-bakeoff.ts`. Fixed stratified sample in
`plans/bakeoff/sample.json` (8 videos, 7 channels, 13 min - 8 h), never written
to the corpus. Full tables in `plans/bakeoff/round{1,2}.{json,md}`.
@@ -781,7 +781,7 @@ anything with a settings surface gets one live run and one assertion.
### Notes for any future bake-off round (not gates now)
-- **`BUCKET_BOUNDS` has holes** (`common/bin/digest-bakeoff.ts:70-75`).
+- **`BUCKET_BOUNDS` has holes** (`plans/tools/digest-bakeoff.ts:70-75`).
`bucketFor()` returns `null` for any duration in a gap, so those videos are
silently unsamplable: **30–45 min**, **90 min – 3 h**, and **4.5 – 6 h** (`long`
caps at 4.5 h, `verylong` starts at 6 h). The gaps look deliberate — they keep
@@ -1216,9 +1216,9 @@ in-memory `settings` override):
```
cd common
-pnpm exec tsx bin/attribution-pilot.ts --sample bakeoff --dry-run
-pnpm exec tsx bin/attribution-pilot.ts --sample bakeoff --label round1-bakeoff
-pnpm exec tsx bin/attribution-pilot.ts --sample video:ObviousRises-rumble/v6z1o2g \
+pnpm exec tsx ../plans/tools/attribution-pilot.ts --sample bakeoff --dry-run
+pnpm exec tsx ../plans/tools/attribution-pilot.ts --sample bakeoff --label round1-bakeoff
+pnpm exec tsx ../plans/tools/attribution-pilot.ts --sample video:ObviousRises-rumble/v6z1o2g \
--method diarized --label round1-diarized
```
@@ -1468,7 +1468,7 @@ transcribed, so each loses its audio permanently unless capture is armed first.
## Attribution round 2 (2026-08-08) — the closed-cast schema, and it works
-Run via the new `common/bin/attribution-bakeoff.ts`, which unlike the pilot
+Run via the new `plans/tools/attribution-bakeoff.ts`, which unlike the pilot
**writes no sidecar**: it drives `attributionPrompt` + `attributionTurns` + the
digest app directly and scores in memory. The harness checks its own claim —
`attribution.json` count **9 before, 9 after**. 14 calls (2 cast + 12 chunk),
diff --git a/plans/STATE.md b/plans/STATE.md
@@ -1306,8 +1306,8 @@ Bake-off (never writes to the corpus):
```bash
cd common
-pnpm exec tsx bin/digest-bakeoff.ts --pick # ONCE — fixes the sample
-pnpm exec tsx bin/digest-bakeoff.ts --label rN --buckets long \
+pnpm exec tsx ../plans/tools/digest-bakeoff.ts --pick # ONCE — fixes the sample
+pnpm exec tsx ../plans/tools/digest-bakeoff.ts --label rN --buckets long \
--modes absolute,chunk-local --candidates 'qwen2.5:7b@8192'
```
diff --git a/plans/bakeoff/speakers-round1-notes.md b/plans/bakeoff/speakers-round1-notes.md
@@ -103,7 +103,7 @@ The generated baseline runs over a fresh 12-video, 8-channel, 4-bucket sample
*not* what the sweep runs. Re-run it with:
```sh
-cd common && pnpm exec tsx bin/digest-bakeoff.ts --label chapters-baseline \
+cd common && pnpm exec tsx ../plans/tools/digest-bakeoff.ts --label chapters-baseline \
--sample ../plans/bakeoff/chapter-sample.json --modes chunk-local \
--candidates 'qwen2.5:7b@8192'
```
@@ -213,12 +213,12 @@ for v in shondo/uVYjZKWEWmk kirsche/Y9hdSH8Ky10 shondo/pl0Wt0P3J4w shondo/ZUh5_u
shondo-vods/CcSsT1cSn8E shondo-vods/Iq3svKzK-yQ shondo-vods/F7_IyckDc3M \
shondo-vods/1d1BEZMo2F0 ; do
pnpm exec tsx bin/diarize-backfill.ts --scope "video:$v"
- pnpm exec tsx bin/attribution-pilot.ts --sample "video:$v" --method diarized --label speakers-prep
+ pnpm exec tsx ../plans/tools/attribution-pilot.ts --sample "video:$v" --method diarized --label speakers-prep
done
# 2. The paired A/B. Both arms get the SAME cue count with context headroom, so the
# speakers-on arm cannot be the only one that truncates.
-pnpm exec tsx bin/digest-bakeoff.ts --label speakers-round1 \
+pnpm exec tsx ../plans/tools/digest-bakeoff.ts --label speakers-round1 \
--sample ../plans/bakeoff/speaker-sample.json \
--modes chunk-local --candidates 'qwen2.5:7b@16384' --max-cues 600 --speakers off,on
```
diff --git a/plans/digest-context-review.md b/plans/digest-context-review.md
@@ -137,7 +137,7 @@ on the ~12,139 videos that carry uploader chapters:
```sh
cd common
-pnpm exec tsx bin/digest-bakeoff.ts --score-existing # no model calls, no writes
+pnpm exec tsx ../plans/tools/digest-bakeoff.ts --score-existing # no model calls, no writes
```
Baseline for that command as of 2026-08-12, over the 15 scorable digests on disk:
diff --git a/plans/tools/attribution-bakeoff.ts b/plans/tools/attribution-bakeoff.ts
@@ -0,0 +1,1281 @@
+#!/usr/bin/env tsx
+// Score attribution PROMPT VARIANTS against a fixed pair of videos, without
+// writing a single sidecar.
+//
+// WHY THIS EXISTS SEPARATELY FROM attribution-pilot.ts, which measures the same
+// lane. The pilot calls attributeOneVideo because the SHIPPED PATH was what was
+// under test — the freshness skip, the downgrade guard, the provenance, the
+// sidecar a human then reads. This is the opposite situation: the thing under
+// test is a prompt and a schema that have won nothing yet, so it drives
+// attributionPrompt + attributionTurns + the digest app DIRECTLY and scores in
+// memory. That is digest-bakeoff.ts's defining property and it is inherited
+// deliberately: a losing variant leaves NOTHING in the corpus and no freshness
+// record has to be invalidated to undo a round.
+//
+// find transcripts/channels -name attribution.json | wc -l
+//
+// is 9 before and after, and this harness checks that itself — see the sidecar
+// census at the end of main(). writeAttribution is never imported.
+//
+// THE QUESTION ROUND 2 ASKS. The 2026-08-08 pilot rejected the text-only lane on
+// both pre-stated thresholds: 14.70 new labels per chunk against a <= 0.3 bar,
+// and 66.2 s/chunk engine against the digest lane's 11.2. The per-chunk data
+// says the cause is NOT the knownSpeakers roster the write-up blamed — on
+// destiny/5nmDzKB23OU 15 of 16 labels first appear in chunk 0, where no roster
+// exists, and on rekietalaw/EsZhaCfc8HQ the collapse is intermittent rather than
+// monotonic. Dozens of labels sit at exactly 60 characters, the schema's own
+// maxLength. The model was continuing the transcript into a free-string field.
+//
+// So the variable is the SCHEMA, and only the schema. Same engine (qwen2.5:7b
+// @ 8192), same chunker, same videos, same scorer, same metrics as the pilot —
+// lib/attributionScore.ts exists so that last one is true by construction rather
+// than by two files agreeing.
+//
+// UNIT DISCIPLINE, inherited and not negotiable. plans/STATE.md records a
+// RETRACTED throughput headline that came from pricing a sweep in
+// seconds-per-audio-hour when the unit of work is the CHUNK (density varies 4x
+// across this corpus). This reports seconds per CHUNK, prints chunks-per-
+// audio-hour beside any rate, and never prints a bare seconds-per-audio-hour.
+//
+// Examples:
+// tsx bin/attribution-bakeoff.ts --dry-run
+// tsx bin/attribution-bakeoff.ts --variant closed-cast --label round2-closed
+// tsx bin/attribution-bakeoff.ts --variant open --label round2-open-control
+// tsx bin/attribution-bakeoff.ts --variant closed-cast --model gemma2:9b \
+// --num-ctx 8192 --label round2-gemma
+
+import os from "node:os";
+import path from "node:path";
+import { mkdir, readdir, writeFile } from "node:fs/promises";
+import { getPaths, type Paths } from "../../common/lib/paths";
+import { parseFlags } from "../../common/bin/_parseFlags";
+import {
+ ATTRIBUTION_FILENAME,
+ ATTRIBUTION_PROMPT_VERSION,
+ type AttributionRecord,
+ type AttributionWarning,
+} from "../../common/lib/attribution";
+import type { AttributionSettings } from "../../common/lib/settings";
+import { OLLAMA_DIGEST_APP_ID } from "../../common/lib/digestApps";
+import type { DigestApp, DigestAppConfig } from "../../common/lib/digestApps";
+import { resolveAttributionTarget } from "../../common/controller/attributionTarget";
+import {
+ ATTRIBUTION_CAST_SAMPLES,
+ ATTRIBUTION_MAX_SPEAKERS,
+ ATTRIBUTION_MAX_TURNS_PER_CHUNK,
+ ATTRIBUTION_UNKNOWN_SPEAKER,
+ CAST_SYSTEM_PROMPT,
+ TURN_SYSTEM_PROMPT,
+ buildCastPrompt,
+ buildClosedTurnPrompt,
+ buildTurnPrompt,
+ castSchema,
+ isUselessSpeakerLabel,
+ selectTranscriptSamples,
+ turnSchema,
+ turnSchemaClosed,
+} from "../../common/lib/attributionPrompt";
+import {
+ assembleTurns,
+ createSpeakerRoster,
+ type SpeakerMark,
+} from "../../common/lib/attributionTurns";
+import {
+ chunkRangesFor,
+ isTranscriptTextLabel,
+ scoreRecord,
+ transcriptHaystack,
+ type VideoMetrics,
+} from "../../common/lib/attributionScore";
+import { parseEngineCall, isUnparsedEngineLine, type CallTiming } from "../../common/lib/engineTiming";
+import {
+ DIGEST_OVERLAP_CUES,
+ hmsToSeconds,
+ maxCuesForContext,
+ toHms,
+} from "../../common/lib/digestPrompt";
+import { chunkCuesForContext } from "../../common/lib/transcriptWindow";
+import { transcriptToMarkdown } from "../../common/lib/transcriptToMarkdown";
+import { readDigestContext } from "../../common/lib/digestContext-server";
+import {
+ chunksPerAudioHour,
+ readChunkCensus,
+ sweepDays,
+} from "../../common/controller/digestPlan";
+import {
+ isCuesJsonFresh,
+ readNormalizedTranscript,
+} from "../../common/controller/normalizeTranscript";
+import type { Cue } from "../../common/lib/vtt";
+
+// The digest lane's own cost on the SAME model and context — round 2's 190
+// engine-seconds over 17 chunks. The bar a second sweep has to justify itself
+// against, not a footnote.
+const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2;
+
+// The pilot's measured text-only numbers on these same videos, so a round-2
+// report states what it is beating without anyone opening the round-1 file.
+const PILOT_BASELINE = {
+ newLabelsPerChunkMean: 14.7,
+ secondsPerChunkEngine: 66.2,
+ byVideo: {
+ "destiny/5nmDzKB23OU": { chunks: 4, labels: 16, newLabelsPerChunk: 0.33 },
+ "rekietalaw/EsZhaCfc8HQ": { chunks: 8, labels: 90, newLabelsPerChunk: 12.57 },
+ } as Record<string, { chunks: number; labels: number; newLabelsPerChunk: number }>,
+};
+
+// THE PROBE PAIR, and why these two. Both have a measured round-1 baseline, and
+// between them they exhibit BOTH failure shapes: destiny collapses in chunk 0
+// where no roster exists, rekietalaw collapses intermittently (chunks 1, 3 and 6
+// bad; 0, 2, 4, 5, 7 clean). 12 chunk calls + 2 cast calls.
+const DEFAULT_VIDEOS = ["destiny/5nmDzKB23OU", "rekietalaw/EsZhaCfc8HQ"];
+
+// The thresholds, written BEFORE the run — see plans/STATE.md. A bar chosen
+// after seeing the numbers is not a bar.
+const THRESHOLDS = {
+ newLabelsPerChunk: 0.3,
+ secondsPerChunkEngine: 25,
+ transcriptTextLabelRate: 0,
+};
+
+type Variant = "closed-cast" | "open";
+
+// ---------------------------------------------------------------------------
+// The sample
+// ---------------------------------------------------------------------------
+
+type ProbeVideo = {
+ slug: string;
+ channelSlug: string;
+ videoDir: string;
+ videoId: string;
+ title: string;
+ channel: string;
+ durationSeconds: number;
+ dir: string;
+ cues: Cue[];
+ chunks: Cue[][];
+ chunkRanges: { start: number; end: number }[];
+};
+
+async function priceVideo(
+ slug: string,
+ paths: Paths,
+ maxCues: number,
+): Promise<ProbeVideo | null> {
+ const slash = slug.indexOf("/");
+ if (slash <= 0) throw new Error(`--videos expects <channelSlug>/<videoDir>, got "${slug}"`);
+ const channelSlug = slug.slice(0, slash);
+ const videoDir = slug.slice(slash + 1);
+ const dir = path.join(paths.channelsDir, channelSlug, "data", videoDir);
+
+ const { fresh, cuesPath } = await isCuesJsonFresh(dir);
+ if (!fresh) return null;
+ const transcript = await readNormalizedTranscript(cuesPath);
+ const cues = transcript?.cues ?? [];
+ if (!transcript || cues.length === 0) return null;
+
+ const chunks = chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES });
+ const lastCue = cues[cues.length - 1];
+ return {
+ slug,
+ channelSlug,
+ videoDir,
+ videoId: transcript.id || videoDir,
+ title: transcript.title || videoDir,
+ channel: transcript.channel || channelSlug,
+ durationSeconds: Math.round(transcript.duration || lastCue.end || lastCue.start || 0),
+ dir,
+ cues,
+ chunks,
+ // Derived by the SHARED helper, so the scorer's chunk indexing cannot drift
+ // from the chunker this run actually used.
+ chunkRanges: chunkRangesFor(cues, maxCues),
+ };
+}
+
+// Render one chunk exactly as attributeOne.ts does — ABSOLUTE timestamps,
+// because an attribution mark lands on one shared timeline that later chunks
+// keep adding to. Re-basing per chunk would make every chunk's marks collide.
+function renderChunk(video: ProbeVideo, cues: Cue[]): string {
+ return transcriptToMarkdown(
+ {
+ id: video.videoId,
+ title: video.title,
+ channel: video.channel,
+ duration: video.durationSeconds,
+ cues,
+ },
+ {
+ timestamps: true,
+ includeDescription: false,
+ includeTags: false,
+ stampForCue: (_clock, seconds) => toHms(Math.max(0, seconds)),
+ },
+ );
+}
+
+// ---------------------------------------------------------------------------
+// The two variants
+// ---------------------------------------------------------------------------
+
+type CastMember = { name: string; confidence: number };
+
+type VariantRun = {
+ record: AttributionRecord;
+ warnings: AttributionWarning[];
+ chunksOk: number;
+ // closed-cast only
+ cast?: CastMember[];
+ castRejected?: string[];
+ castCallSeconds?: number;
+ unknownTurns?: number;
+ offCastTurns?: number;
+ // Turns whose timestamp fell outside the chunk's own range. Counted, because
+ // they are DROPPED silently and a run that produced nothing but these would
+ // otherwise be indistinguishable from a model that returned nothing at all.
+ outOfRangeTurns: number;
+ totalTurnsAccepted: number;
+};
+
+type RunCtx = {
+ app: DigestApp;
+ config: DigestAppConfig;
+ contextNote?: string;
+ log: (m: string) => void;
+};
+
+function emptyRecord(
+ video: ProbeVideo,
+ model: string,
+ chunks: number,
+ chunksOk: number,
+ warnings: AttributionWarning[],
+): AttributionRecord {
+ return {
+ videoId: video.videoId,
+ generatedAt: new Date().toISOString(),
+ speakers: [],
+ segments: [],
+ ...(warnings.length ? { warnings } : {}),
+ provenance: {
+ method: "text-only",
+ appId: "bakeoff",
+ model,
+ modelRequested: model,
+ // NOT the shipped ATTRIBUTION_PROMPT_VERSION. This record never reaches
+ // disk, and stamping it with the live version would be a lie waiting for
+ // someone to copy it somewhere that matters.
+ promptVersion: -1,
+ generatedAt: new Date().toISOString(),
+ chunks,
+ chunksOk,
+ durationMs: 0,
+ },
+ };
+}
+
+// PASS A — cast discovery. ONE call for the whole video.
+async function discoverCast(
+ video: ProbeVideo,
+ ctx: RunCtx,
+): Promise<{ cast: CastMember[]; rejected: string[]; seconds: number; model: string }> {
+ const samples = selectTranscriptSamples(video.cues, ATTRIBUTION_CAST_SAMPLES);
+ ctx.log(
+ ` ${video.slug}: cast discovery over ${samples.length} evenly-spaced excerpt(s), 1 call.`,
+ );
+ const started = Date.now();
+ const result = await ctx.app.run({
+ config: ctx.config,
+ system: CAST_SYSTEM_PROMPT,
+ prompt: buildCastPrompt({
+ title: video.title,
+ channel: video.channel,
+ samples,
+ ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
+ }),
+ schema: castSchema(),
+ onLog: ctx.log,
+ });
+ const seconds = (Date.now() - started) / 1000;
+
+ const haystack = transcriptHaystack(video.cues);
+ const cast: CastMember[] = [];
+ const rejected: string[] = [];
+ const seen = new Set<string>();
+ const raw = (result.data as { cast?: unknown })?.cast;
+ if (Array.isArray(raw)) {
+ for (const item of raw) {
+ const e = item as { name?: unknown; confidence?: unknown };
+ if (typeof e?.name !== "string") continue;
+ const name = e.name.trim();
+ // THE SAME GUARDS THE RECORD GETS, applied to the cast — pass A uses a
+ // free string, so it can fail the exact way pass B is being fixed for. A
+ // transcript fragment admitted here would be minted into the enum and
+ // become a legitimate label for the rest of the video.
+ if (isUselessSpeakerLabel(name) || isTranscriptTextLabel(name, haystack)) {
+ rejected.push(name);
+ continue;
+ }
+ const key = name.toLowerCase();
+ if (seen.has(key)) continue;
+ seen.add(key);
+ cast.push({
+ name,
+ confidence:
+ typeof e.confidence === "number" && Number.isFinite(e.confidence)
+ ? Math.min(1, Math.max(0, e.confidence))
+ : 0,
+ });
+ if (cast.length >= ATTRIBUTION_MAX_SPEAKERS) break;
+ }
+ }
+ return { cast, rejected, seconds, model: result.model };
+}
+
+// PASS B — turn assignment, per chunk, against a CLOSED label space.
+async function runClosedCast(video: ProbeVideo, ctx: RunCtx): Promise<VariantRun> {
+ const warnings: AttributionWarning[] = [];
+ const { cast, rejected, seconds: castCallSeconds, model: castModel } =
+ await discoverCast(video, ctx);
+
+ ctx.log(
+ ` ${video.slug}: cast = ${
+ cast.map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "(none)"
+ }${rejected.length ? ` — rejected ${rejected.length}` : ""}`,
+ );
+
+ if (cast.length === 0) {
+ // Pass A returning nothing usable is a RESULT, not a crash: the plan's third
+ // decision row ("Pass A itself returns garbage") escalates to gemma2:9b.
+ warnings.push({ code: "empty", detail: "cast discovery produced no usable name" });
+ return {
+ record: emptyRecord(video, castModel, video.chunks.length, 0, warnings),
+ warnings,
+ chunksOk: 0,
+ cast,
+ castRejected: rejected,
+ castCallSeconds,
+ outOfRangeTurns: 0,
+ totalTurnsAccepted: 0,
+ };
+ }
+
+ const names = cast.map((c) => c.name);
+ const schema = turnSchemaClosed(names, ATTRIBUTION_MAX_TURNS_PER_CHUNK);
+ // The roster is SEEDED with the whole cast and never grows. That is the point:
+ // cross-chunk identity stops being something the model has to remember and
+ // becomes a property of the grammar it decodes against.
+ const roster = createSpeakerRoster(names);
+ const marks: SpeakerMark[] = [];
+ let model = castModel;
+ let chunksOk = 0;
+ let unknownTurns = 0;
+ let offCastTurns = 0;
+ let outOfRange = 0;
+ let accepted = 0;
+
+ for (let i = 0; i < video.chunks.length; i++) {
+ const chunk = video.chunks[i];
+ const startSeconds = Math.max(0, Math.floor(chunk[0].start));
+ const endSeconds = Math.max(
+ startSeconds,
+ Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
+ );
+ try {
+ const result = await ctx.app.run({
+ config: ctx.config,
+ system: TURN_SYSTEM_PROMPT,
+ prompt: buildClosedTurnPrompt({
+ title: video.title,
+ channel: video.channel,
+ startSeconds,
+ endSeconds,
+ transcript: renderChunk(video, chunk),
+ cast: names,
+ ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
+ }),
+ schema,
+ onLog: ctx.log,
+ });
+ chunksOk++;
+ model = result.model || model;
+ const raw = (result.data as { turns?: unknown })?.turns;
+ if (!Array.isArray(raw)) continue;
+ for (const item of raw) {
+ const e = item as { start?: unknown; speaker?: unknown };
+ if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue;
+ const at = hmsToSeconds(e.start);
+ if (at === null) {
+ warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) });
+ continue;
+ }
+ if (at < startSeconds || at > endSeconds) {
+ outOfRange++;
+ continue;
+ }
+ if (e.speaker.trim() === ATTRIBUTION_UNKNOWN_SPEAKER) {
+ // Counted, then dropped. An honest Unknown is a good answer and the
+ // rate is worth reading, but it must not become a speaker.
+ unknownTurns++;
+ continue;
+ }
+ if (isUselessSpeakerLabel(e.speaker)) {
+ unknownTurns++;
+ continue;
+ }
+ // THE CHECK THAT MAKES THIS A MEASUREMENT. The enum pins `speaker`, and
+ // the parser verifies it anyway — the belt-and-braces nameClusters
+ // already applies to cluster indices. If a label outside the cast ever
+ // lands here, the grammar did NOT hold and the whole hypothesis is
+ // refuted; offCastTurns is that count, and it must be 0.
+ const index = roster.lookup(e.speaker);
+ if (index === undefined) {
+ offCastTurns++;
+ warnings.push({
+ code: "unknown-cluster",
+ chunk: i,
+ detail: `off-cast label: ${e.speaker.slice(0, 60)}`,
+ });
+ continue;
+ }
+ marks.push({ at, speaker: index });
+ accepted++;
+ }
+ } catch (err) {
+ const message = (err as Error)?.message ?? String(err);
+ warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) });
+ ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`);
+ }
+ }
+
+ const lastCue = video.cues[video.cues.length - 1];
+ // Only cast members that actually SPOKE become speakers. A name pass A
+ // proposed and pass B never used is not in the record — it would otherwise
+ // score as a label at 0 seconds and inflate the label count for nothing.
+ const used = [...new Set(marks.map((m) => m.speaker))].sort((a, b) => a - b);
+ const remap = new Map(used.map((old, i) => [old, i]));
+ const { speakers, segments } = assembleTurns({
+ labels: used.map((i) => roster.labels[i]),
+ marks: marks.map((m) => ({ at: m.at, speaker: remap.get(m.speaker)! })),
+ transcriptEnd: lastCue.end || lastCue.start,
+ });
+
+ const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings);
+ record.speakers = speakers;
+ record.segments = segments;
+ return {
+ record,
+ warnings,
+ chunksOk,
+ cast,
+ castRejected: rejected,
+ castCallSeconds,
+ unknownTurns,
+ offCastTurns,
+ outOfRangeTurns: outOfRange,
+ totalTurnsAccepted: accepted,
+ };
+}
+
+// The OPEN control — the shipped prompt and schema, driven in memory. Reproduces
+// round 1 without writing a sidecar, so a round-2 report can state the delta
+// under identical box conditions instead of against a number measured last week.
+async function runOpen(video: ProbeVideo, ctx: RunCtx): Promise<VariantRun> {
+ const warnings: AttributionWarning[] = [];
+ const roster = createSpeakerRoster();
+ const marks: SpeakerMark[] = [];
+ let model = "";
+ let chunksOk = 0;
+ let accepted = 0;
+ let outOfRange = 0;
+
+ for (let i = 0; i < video.chunks.length; i++) {
+ const chunk = video.chunks[i];
+ const startSeconds = Math.max(0, Math.floor(chunk[0].start));
+ const endSeconds = Math.max(
+ startSeconds,
+ Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
+ );
+ try {
+ const result = await ctx.app.run({
+ config: ctx.config,
+ system: TURN_SYSTEM_PROMPT,
+ prompt: buildTurnPrompt({
+ title: video.title,
+ channel: video.channel,
+ startSeconds,
+ endSeconds,
+ transcript: renderChunk(video, chunk),
+ ...(roster.labels.length > 0 ? { knownSpeakers: roster.labels.slice() } : {}),
+ ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}),
+ }),
+ schema: turnSchema(),
+ onLog: ctx.log,
+ });
+ chunksOk++;
+ model = result.model || model;
+ const raw = (result.data as { turns?: unknown })?.turns;
+ if (!Array.isArray(raw)) continue;
+ for (const item of raw) {
+ const e = item as { start?: unknown; speaker?: unknown };
+ if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue;
+ const at = hmsToSeconds(e.start);
+ if (at === null) {
+ warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) });
+ continue;
+ }
+ if (at < startSeconds || at > endSeconds) {
+ outOfRange++;
+ continue;
+ }
+ if (isUselessSpeakerLabel(e.speaker)) continue;
+ marks.push({ at, speaker: roster.intern(e.speaker) });
+ accepted++;
+ }
+ } catch (err) {
+ const message = (err as Error)?.message ?? String(err);
+ warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) });
+ ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`);
+ }
+ }
+
+ const lastCue = video.cues[video.cues.length - 1];
+ const { speakers, segments } = assembleTurns({
+ labels: roster.labels,
+ marks,
+ transcriptEnd: lastCue.end || lastCue.start,
+ });
+ const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings);
+ record.speakers = speakers;
+ record.segments = segments;
+ return {
+ record,
+ warnings,
+ chunksOk,
+ outOfRangeTurns: outOfRange,
+ totalTurnsAccepted: accepted,
+ };
+}
+
+// ---------------------------------------------------------------------------
+// The run
+// ---------------------------------------------------------------------------
+
+type VideoResult = {
+ slug: string;
+ title: string;
+ durationSeconds: number;
+ chunks: number;
+ chunksOk: number;
+ wallSeconds: number;
+ calls: CallTiming[];
+ unparsedLogLines: number;
+ warningsByCode: Record<string, number>;
+ metrics: VideoMetrics | null;
+ cast?: CastMember[];
+ castRejected?: string[];
+ castCallSeconds?: number;
+ unknownTurns?: number;
+ offCastTurns?: number;
+ outOfRangeTurns: number;
+ totalTurnsAccepted: number;
+ speakers: { label: string; seconds: number }[];
+};
+
+function sum(v: number[]): number {
+ return v.reduce((a, b) => a + b, 0);
+}
+
+function boxState(): Record<string, unknown> {
+ return {
+ loadavg: os.loadavg().map((n) => Number(n.toFixed(2))),
+ freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)),
+ totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)),
+ };
+}
+
+// How many attribution.json files exist right now. THE SAFETY CLAIM of this
+// harness, checked rather than asserted in a comment.
+async function countSidecars(paths: Paths): Promise<number> {
+ let total = 0;
+ let channels: string[];
+ try {
+ channels = (await readdir(paths.channelsDir, { withFileTypes: true }))
+ .filter((e) => e.isDirectory())
+ .map((e) => e.name);
+ } catch {
+ return 0;
+ }
+ for (const channel of channels) {
+ const dataDir = path.join(paths.channelsDir, channel, "data");
+ let videos;
+ try {
+ videos = await readdir(dataDir, { withFileTypes: true });
+ } catch {
+ continue;
+ }
+ for (const v of videos) {
+ if (!v.isDirectory()) continue;
+ try {
+ const files = await readdir(path.join(dataDir, v.name));
+ if (files.includes(ATTRIBUTION_FILENAME)) total++;
+ } catch {
+ // A video dir that vanished mid-walk is not a sidecar.
+ }
+ }
+ }
+ return total;
+}
+
+async function runVideo(
+ video: ProbeVideo,
+ variant: Variant,
+ ctx: Omit<RunCtx, "log">,
+ log: (m: string) => void,
+): Promise<VideoResult> {
+ const calls: CallTiming[] = [];
+ let unparsed = 0;
+ // TEES, and it has to. This one closure is both the engine-timing sink and the
+ // harness's own progress channel — the cast list, the per-chunk failures. An
+ // earlier version only parsed, which silently swallowed the cast line and made
+ // a 0-label result impossible to diagnose from the console.
+ const onLog = (line: string) => {
+ const call = parseEngineCall(line);
+ if (call) {
+ calls.push(call);
+ return;
+ }
+ if (isUnparsedEngineLine(line)) {
+ unparsed++;
+ return;
+ }
+ log(line);
+ };
+
+ const started = Date.now();
+ const out =
+ variant === "closed-cast"
+ ? await runClosedCast(video, { ...ctx, log: onLog })
+ : await runOpen(video, { ...ctx, log: onLog });
+ const wallSeconds = (Date.now() - started) / 1000;
+
+ const warningsByCode: Record<string, number> = {};
+ for (const w of out.warnings) {
+ warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1;
+ }
+
+ // Scored WITH the cues, so transcriptTextLabelRate is a number rather than
+ // null — it is the metric this round exists to drive to zero.
+ const metrics =
+ out.record.speakers.length > 0
+ ? scoreRecord(out.record, {
+ chunkRanges: video.chunkRanges,
+ durationSeconds: video.durationSeconds,
+ cues: video.cues,
+ })
+ : null;
+
+ const result: VideoResult = {
+ slug: video.slug,
+ title: video.title,
+ durationSeconds: video.durationSeconds,
+ chunks: video.chunks.length,
+ chunksOk: out.chunksOk,
+ wallSeconds,
+ calls,
+ unparsedLogLines: unparsed,
+ warningsByCode,
+ metrics,
+ ...(out.cast ? { cast: out.cast } : {}),
+ ...(out.castRejected ? { castRejected: out.castRejected } : {}),
+ ...(out.castCallSeconds !== undefined ? { castCallSeconds: out.castCallSeconds } : {}),
+ ...(out.unknownTurns !== undefined ? { unknownTurns: out.unknownTurns } : {}),
+ ...(out.offCastTurns !== undefined ? { offCastTurns: out.offCastTurns } : {}),
+ outOfRangeTurns: out.outOfRangeTurns,
+ totalTurnsAccepted: out.totalTurnsAccepted,
+ speakers: out.record.speakers.map((s) => ({ label: s.label, seconds: s.seconds ?? 0 })),
+ };
+
+ log(
+ ` ${video.slug} ${video.chunks.length} chunk(s) → ${out.chunksOk} ok, ` +
+ `${out.record.speakers.length} label(s), ${out.record.segments.length} segment(s)` +
+ (metrics
+ ? `, new/chunk ${fixed(metrics.newLabelsPerChunk)}, ` +
+ `transcript-text ${metrics.transcriptTextLabelRate === null ? "n/a" : pct(metrics.transcriptTextLabelRate)}, ` +
+ `top share ${pct(metrics.dominantSpeakerShare)}`
+ : "") +
+ (out.offCastTurns ? `, OFF-CAST ${out.offCastTurns}` : "") +
+ (out.unknownTurns ? `, Unknown ${out.unknownTurns}` : "") +
+ (out.outOfRangeTurns ? `, out-of-range ${out.outOfRangeTurns}` : "") +
+ `, ${wallSeconds.toFixed(0)}s wall`,
+ );
+ return result;
+}
+
+// ---------------------------------------------------------------------------
+// Reporting
+// ---------------------------------------------------------------------------
+
+function pct(n: number): string {
+ return `${(n * 100).toFixed(0)}%`;
+}
+function fixed(n: number | null | undefined, digits = 2): string {
+ return n === null || n === undefined ? "n/a" : n.toFixed(digits);
+}
+
+type CostSummary = {
+ chunks: number;
+ chunksOk: number;
+ calls: number;
+ callsWithEngineTiming: number;
+ audioSeconds: number;
+ chunksPerAudioHour: number;
+ // Chunk-call timings ONLY. The cast call is one per VIDEO and pricing it as a
+ // chunk would flatter or punish the variant depending on video length; it is
+ // reported separately, in its own unit.
+ secondsPerChunkWall: number | null;
+ secondsPerChunkEngine: number | null;
+ secondsPerChunkEngineExLoad: number | null;
+ meanLoadSeconds: number | null;
+ decodeTokensPerSecond: number | null;
+ outputTokens: number;
+ castCallSecondsMean: number | null;
+ unparsedLogLines: number;
+};
+
+function summarizeCost(results: VideoResult[], variant: Variant): CostSummary {
+ const calls = results.flatMap((r) => r.calls);
+ // The cast call is the FIRST call of each closed-cast video. Dropping it from
+ // the per-chunk denominators keeps "s/chunk" meaning one chunk.
+ const chunkCalls =
+ variant === "closed-cast"
+ ? results.flatMap((r) => r.calls.slice(1))
+ : calls;
+ const withEngine = chunkCalls.filter((c) => c.engineSeconds !== undefined);
+ const withLoad = chunkCalls.filter((c) => c.loadSeconds !== undefined);
+ const chunks = sum(results.map((r) => r.chunks));
+ const audioSeconds = sum(results.map((r) => r.durationSeconds));
+ const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0));
+ const outTokens = sum(calls.map((c) => c.outputTokens ?? 0));
+ const castSeconds = results
+ .map((r) => r.castCallSeconds)
+ .filter((s): s is number => s !== undefined);
+ const chunkWall = sum(
+ results.map((r) => r.wallSeconds - (r.castCallSeconds ?? 0)),
+ );
+ return {
+ chunks,
+ chunksOk: sum(results.map((r) => r.chunksOk)),
+ calls: calls.length,
+ callsWithEngineTiming: withEngine.length,
+ audioSeconds,
+ chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds),
+ secondsPerChunkWall: chunks > 0 ? chunkWall / chunks : null,
+ secondsPerChunkEngine: withEngine.length
+ ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length
+ : null,
+ secondsPerChunkEngineExLoad: withEngine.length
+ ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) /
+ withEngine.length
+ : null,
+ meanLoadSeconds: withLoad.length
+ ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length
+ : null,
+ decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null,
+ outputTokens: outTokens,
+ castCallSecondsMean: castSeconds.length ? sum(castSeconds) / castSeconds.length : null,
+ unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)),
+ };
+}
+
+// The verdict, computed against thresholds fixed before the run.
+function verdict(results: VideoResult[], cost: CostSummary) {
+ const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null);
+ const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null);
+ const newPerChunk = withSeams.length
+ ? sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length
+ : null;
+ const textRates = scored
+ .map((m) => m.transcriptTextLabelRate)
+ .filter((r): r is number => r !== null);
+ const textRate = textRates.length ? Math.max(...textRates) : null;
+ const engine = cost.secondsPerChunkEngine;
+ const offCast = sum(results.map((r) => r.offCastTurns ?? 0));
+
+ const identityOk = newPerChunk !== null && newPerChunk <= THRESHOLDS.newLabelsPerChunk;
+ const textOk = textRate !== null && textRate <= THRESHOLDS.transcriptTextLabelRate;
+ const costOk = engine !== null && engine < THRESHOLDS.secondsPerChunkEngine;
+ return {
+ newLabelsPerChunkMean: newPerChunk,
+ worstTranscriptTextLabelRate: textRate,
+ secondsPerChunkEngine: engine,
+ offCastTurns: offCast,
+ identityOk,
+ textOk,
+ costOk,
+ // A single label at ~100% scores a perfect headline and is worthless.
+ degenerateVideos: scored.filter((m) => m.labelsPerVideo === 1).length,
+ passes: identityOk && textOk && costOk && offCast === 0,
+ };
+}
+
+function reportMarkdown(args: {
+ label: string;
+ variant: Variant;
+ model: string;
+ numCtx?: number;
+ maxCues: number;
+ results: VideoResult[];
+ cost: CostSummary;
+ v: ReturnType<typeof verdict>;
+ census: { chunks: number; videos: number; chunksPerAudioHour: number; statsSchemaStale: boolean } | null;
+ boxStart: Record<string, unknown>;
+ boxEnd: Record<string, unknown>;
+ startedAt: string;
+ sidecarsBefore: number;
+ sidecarsAfter: number;
+}): string {
+ const { results, cost, v } = args;
+ const out: string[] = [];
+ out.push(`# Attribution bake-off — ${args.label}`);
+ out.push("");
+ out.push(
+ `Variant **${args.variant}** · ${results.length} video(s) · model \`${args.model}\` @ numCtx ` +
+ `${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`,
+ );
+ out.push("");
+ out.push(
+ "**No sidecar was written.** `attribution.json` count " +
+ `${args.sidecarsBefore} before, ${args.sidecarsAfter} after` +
+ (args.sidecarsBefore === args.sidecarsAfter ? "." : " — **MISMATCH, investigate.**"),
+ );
+ out.push("");
+
+ out.push("## Verdict against the thresholds fixed before the run");
+ out.push("");
+ out.push("| criterion | bar | measured | |");
+ out.push("| --- | ---: | ---: | :-: |");
+ out.push(
+ `| new labels/chunk (mean) | ≤ ${THRESHOLDS.newLabelsPerChunk} | ${fixed(v.newLabelsPerChunkMean)} | ${v.identityOk ? "PASS" : "FAIL"} |`,
+ );
+ out.push(
+ `| transcript-text labels (worst video) | ${THRESHOLDS.transcriptTextLabelRate} | ${
+ v.worstTranscriptTextLabelRate === null ? "n/a" : pct(v.worstTranscriptTextLabelRate)
+ } | ${v.textOk ? "PASS" : "FAIL"} |`,
+ );
+ out.push(
+ `| s/chunk engine | < ${THRESHOLDS.secondsPerChunkEngine} | ${fixed(v.secondsPerChunkEngine, 1)} | ${v.costOk ? "PASS" : "FAIL"} |`,
+ );
+ if (args.variant === "closed-cast") {
+ out.push(
+ `| off-cast labels (grammar held) | 0 | ${v.offCastTurns} | ${v.offCastTurns === 0 ? "PASS" : "FAIL"} |`,
+ );
+ }
+ out.push("");
+ out.push(`**${v.passes ? "ALL THRESHOLDS MET" : "NOT MET"}.**`);
+ out.push("");
+ out.push(
+ `Round 1 (open schema, same model/context/videos) measured ` +
+ `**${PILOT_BASELINE.newLabelsPerChunkMean} new labels/chunk** and ` +
+ `**${PILOT_BASELINE.secondsPerChunkEngine} s/chunk engine**. Digest lane baseline: ` +
+ `${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE} s/chunk engine.`,
+ );
+ out.push("");
+ out.push(
+ `Single-label (degenerate-risk) videos: ${v.degenerateVideos} of ${results.length}. ` +
+ "A perfect headline with one label at ~100% talk time is worthless — read the two together.",
+ );
+ out.push("");
+
+ if (args.variant === "closed-cast") {
+ out.push("## Pass A — cast discovery (1 call per video)");
+ out.push("");
+ out.push("| video | cast | rejected | call s |");
+ out.push("| --- | --- | ---: | ---: |");
+ for (const r of results) {
+ out.push(
+ `| ${r.slug} | ${
+ (r.cast ?? []).map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "—"
+ } | ${r.castRejected?.length ?? 0} | ${fixed(r.castCallSeconds, 1)} |`,
+ );
+ }
+ out.push("");
+ out.push(
+ "**Pass A is independently valuable.** If pass B churns, this alone is the",
+ 'cast-list product — ~1 call per video, order 10 days corpus-wide — shipping as',
+ '"videos featuring X" without ever claiming who said which line.',
+ );
+ out.push("");
+ }
+
+ out.push("## Cross-chunk identity, and the label space");
+ out.push("");
+ out.push(
+ "| video | chunks | labels | new/chunk | round-1 new/chunk | transcript-text | truncated | singleton | top share |",
+ );
+ out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
+ for (const r of results) {
+ const m = r.metrics;
+ const base = PILOT_BASELINE.byVideo[r.slug];
+ out.push(
+ `| ${r.slug} | ${r.chunks} | ${m?.labelsPerVideo ?? "—"} | ${fixed(m?.newLabelsPerChunk)} | ${
+ base ? base.newLabelsPerChunk.toFixed(2) : "—"
+ } | ${
+ m?.transcriptTextLabelRate === null || m === null ? "n/a" : pct(m.transcriptTextLabelRate)
+ } | ${m ? pct(m.truncatedLabelRate) : "—"} | ${
+ m === null || m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate)
+ } | ${m ? pct(m.dominantSpeakerShare) : "—"} |`,
+ );
+ }
+ out.push("");
+ const anyText = results.some((r) => (r.metrics?.transcriptTextLabels.length ?? 0) > 0);
+ if (anyText) {
+ out.push("Labels that are transcript text — the failure this round exists to eliminate:");
+ out.push("");
+ for (const r of results) {
+ for (const l of r.metrics?.transcriptTextLabels ?? []) {
+ out.push(`- \`${r.slug}\`: ${JSON.stringify(l)}`);
+ }
+ }
+ out.push("");
+ }
+
+ if (args.variant === "closed-cast") {
+ out.push("## Pass B — how often the model declined");
+ out.push("");
+ out.push("| video | turns accepted | Unknown | off-cast | out-of-range |");
+ out.push("| --- | ---: | ---: | ---: | ---: |");
+ for (const r of results) {
+ out.push(
+ `| ${r.slug} | ${r.totalTurnsAccepted} | ${r.unknownTurns ?? 0} | ${r.offCastTurns ?? 0} | ${r.outOfRangeTurns} |`,
+ );
+ }
+ out.push("");
+ out.push(
+ "`Unknown` is a GOOD answer, not a failure — the bias must run toward dropping on",
+ "uncertainty. `off-cast` must be 0: anything else means the enum did not convert to",
+ "a grammar and the whole hypothesis is refuted.",
+ );
+ out.push("");
+ }
+
+ out.push("## What the labels say");
+ out.push("");
+ for (const r of results) {
+ const total = sum(r.speakers.map((s) => s.seconds)) || 1;
+ out.push(
+ `- \`${r.slug}\` — ${
+ r.speakers
+ .map((s) => `${s.label} ${pct(s.seconds / total)}`)
+ .join(", ") || "(no speaker)"
+ }`,
+ );
+ }
+ out.push("");
+
+ const warnings: Record<string, number> = {};
+ for (const r of results) {
+ for (const [k, n] of Object.entries(r.warningsByCode)) warnings[k] = (warnings[k] ?? 0) + n;
+ }
+ out.push(
+ `Warnings by code: ${
+ Object.keys(warnings).length
+ ? Object.entries(warnings).map(([k, n]) => `${k} ${n}`).join(", ")
+ : "none"
+ }`,
+ );
+ out.push("");
+
+ out.push("## Cost");
+ out.push("");
+ out.push(
+ `${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s) total ` +
+ `(${cost.callsWithEngineTiming} chunk call(s) with engine timing). Sample density ` +
+ `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` +
+ `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`,
+ );
+ out.push("");
+ out.push("| quantity | value | round 1 |");
+ out.push("| --- | ---: | ---: |");
+ out.push(`| s/chunk, wall | ${fixed(cost.secondsPerChunkWall, 1)} | — |`);
+ out.push(
+ `| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} | ${PILOT_BASELINE.secondsPerChunkEngine} |`,
+ );
+ out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} | — |`);
+ out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} | — |`);
+ out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} | — |`);
+ out.push(`| TOTAL output tokens | ${cost.outputTokens} | — |`);
+ if (args.variant === "closed-cast") {
+ out.push(`| cast call s/VIDEO (not per chunk) | ${fixed(cost.castCallSecondsMean, 1)} | — |`);
+ }
+ out.push(
+ `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} | — |`,
+ );
+ out.push("");
+ out.push(
+ "Output tokens are the thing to watch: engine time tracked wall time within 7% in",
+ "round 1, so 66.2 s/chunk was pure decode volume (31,266 output tokens on one",
+ "13-chunk video). An enum token per turn instead of a 60-character string is the",
+ "mechanism by which this variant is meant to be cheaper as well as better.",
+ );
+ out.push("");
+
+ if (args.census) {
+ const idle = cost.secondsPerChunkEngineExLoad;
+ const contended = cost.secondsPerChunkWall;
+ out.push("## Corpus projection — per-turn lane");
+ out.push("");
+ out.push(
+ `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` +
+ `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` +
+ (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") +
+ ".",
+ );
+ out.push("");
+ out.push(
+ `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` +
+ `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended** ` +
+ `— from n = ${results.length} video(s) / ${cost.chunks} chunk(s), one box, one hour.`,
+ );
+ out.push("");
+ out.push(
+ `Cast-only, for comparison: **1 call per video** over ${args.census.videos.toLocaleString()} ` +
+ `videos at ${fixed(cost.castCallSecondsMean, 1)}s = ` +
+ `${
+ cost.castCallSecondsMean === null
+ ? "n/a"
+ : ((args.census.videos * cost.castCallSecondsMean) / 86400).toFixed(1)
+ } days.`,
+ );
+ out.push("");
+ out.push(
+ "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that has",
+ "completed 0.17% of its own ~194k calls, with nothing arbitrating between them.",
+ );
+ out.push("");
+ }
+
+ out.push("## Box");
+ out.push("");
+ out.push(`Start: \`${JSON.stringify(args.boxStart)}\``);
+ out.push("");
+ out.push(`End: \`${JSON.stringify(args.boxEnd)}\``);
+ out.push("");
+ if (cost.unparsedLogLines > 0) {
+ out.push(
+ `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`,
+ );
+ out.push("");
+ }
+
+ out.push("## Units");
+ out.push("");
+ out.push(
+ "- The unit of per-turn work is the **chunk** (one model call). Seconds-per-audio-",
+ " hour is not a unit here: chunk density varies 4× across this corpus, and pricing",
+ " in it is the error behind a retracted throughput headline. This report does not",
+ " print one.",
+ "- The unit of cast discovery is **calls per video**, priced separately and never",
+ " folded into s/chunk.",
+ "- Wall and engine are different quantities and are never substituted for each",
+ " other. Wall is the realistic ceiling on this shared box; engine excl. load is",
+ " the floor.",
+ `- Every rate carries its denominator. At n = ${results.length} videos a bare percentage`,
+ " would be a lie of precision.",
+ );
+ out.push("");
+ return out.join("\n");
+}
+
+// ---------------------------------------------------------------------------
+
+async function main(): Promise<void> {
+ const flags = parseFlags(process.argv.slice(2));
+ const paths = getPaths();
+ const variant: Variant = flags.variant === "open" ? "open" : "closed-cast";
+ const dryRun = flags["dry-run"] === "true";
+ const slugs = (flags.videos ?? DEFAULT_VIDEOS.join(",")).split(",").map((s) => s.trim()).filter(Boolean);
+ const label = flags.label ?? `${variant}-${new Date().toISOString().slice(0, 10)}`;
+
+ // The lane is enabled IN MEMORY, exactly as the pilot does it. settings.json
+ // is never written — a bake-off must not be the thing that arms a lane.
+ const probeSettings: AttributionSettings = {
+ enabled: true,
+ appId: OLLAMA_DIGEST_APP_ID,
+ model: flags.model ?? "",
+ diarizedEnabled: false,
+ textOnlyEnabled: true,
+ promptVersion: ATTRIBUTION_PROMPT_VERSION,
+ };
+ const { app, config: baseConfig, modelRequested } = resolveAttributionTarget(
+ "text-only",
+ probeSettings,
+ );
+ // Overrides land on a COPY of the app config. resolveAttributionTarget reads
+ // the real settings for the ollama URL and timeout, which we want, but the
+ // model and context are this run's variables.
+ const config: DigestAppConfig = {
+ ...baseConfig,
+ model: modelRequested,
+ ...(flags["num-ctx"] ? { numCtx: Number(flags["num-ctx"]) } : {}),
+ };
+ const maxCues = maxCuesForContext(config.numCtx);
+
+ const videos: ProbeVideo[] = [];
+ const skipped: string[] = [];
+ for (const slug of slugs) {
+ const v = await priceVideo(slug, paths, maxCues);
+ if (v) videos.push(v);
+ else skipped.push(slug);
+ }
+
+ const totalChunks = sum(videos.map((v) => v.chunks.length));
+ const totalAudio = sum(videos.map((v) => v.durationSeconds));
+ const census = await readChunkCensus(paths, maxCues);
+ const castCalls = variant === "closed-cast" ? videos.length : 0;
+
+ console.log(
+ `Attribution bake-off — variant ${variant}, app ${app.id}, model ${modelRequested}, ` +
+ `numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`,
+ );
+ console.log(`Settings override (in memory only): ${JSON.stringify(probeSettings)}`);
+ console.log("NO SIDECAR IS WRITTEN. writeAttribution is not imported by this harness.");
+ console.log("");
+ for (const v of videos) {
+ console.log(
+ ` ${v.slug} ${toHms(v.durationSeconds)} ${v.cues.length} cues → ${v.chunks.length} chunk(s)`,
+ );
+ }
+ if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`);
+ console.log("");
+ console.log(
+ `${videos.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` +
+ `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour.`,
+ );
+ console.log(
+ `Calls this run: ${castCalls} cast + ${totalChunks} chunk = ${castCalls + totalChunks}.`,
+ );
+ if (census) {
+ console.log(
+ `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s)` +
+ (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""),
+ );
+ }
+ console.log(`Box: ${JSON.stringify(boxState())}`);
+ console.log("");
+
+ if (videos.length === 0) {
+ console.error("No video had a usable transcript — nothing to measure.");
+ process.exitCode = 1;
+ return;
+ }
+
+ if (dryRun) {
+ console.log("--dry-run: priced only, no model call and no write.");
+ return;
+ }
+
+ const sidecarsBefore = await countSidecars(paths);
+ console.log(`attribution.json on disk before: ${sidecarsBefore}`);
+ console.log("");
+
+ const boxStart = boxState();
+ const startedAt = new Date().toISOString();
+ const results: VideoResult[] = [];
+ for (const video of videos) {
+ const context = await readDigestContext(paths, video.channelSlug);
+ results.push(
+ await runVideo(
+ video,
+ variant,
+ { app, config, ...(context.note ? { contextNote: context.note } : {}) },
+ (m) => console.log(m),
+ ),
+ );
+ }
+ const boxEnd = boxState();
+
+ const sidecarsAfter = await countSidecars(paths);
+ const cost = summarizeCost(results, variant);
+ const v = verdict(results, cost);
+
+ const report = {
+ version: 1 as const,
+ label,
+ variant,
+ startedAt,
+ finishedAt: new Date().toISOString(),
+ engine: {
+ appId: app.id,
+ modelRequested,
+ numCtx: config.numCtx,
+ maxCues,
+ overlapCues: DIGEST_OVERLAP_CUES,
+ castSamples: ATTRIBUTION_CAST_SAMPLES,
+ maxTurnsPerChunk: ATTRIBUTION_MAX_TURNS_PER_CHUNK,
+ },
+ thresholds: THRESHOLDS,
+ verdict: v,
+ skipped,
+ sidecars: { before: sidecarsBefore, after: sidecarsAfter },
+ box: { start: boxStart, end: boxEnd },
+ census,
+ cost,
+ baseline: {
+ pilot: PILOT_BASELINE,
+ digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE,
+ },
+ videos: results,
+ };
+
+ const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot");
+ await mkdir(outDir, { recursive: true });
+ const markdown = reportMarkdown({
+ label,
+ variant,
+ model: modelRequested,
+ numCtx: config.numCtx,
+ maxCues,
+ results,
+ cost,
+ v,
+ census,
+ boxStart,
+ boxEnd,
+ startedAt,
+ sidecarsBefore,
+ sidecarsAfter,
+ });
+ const jsonPath = path.join(outDir, `${label}.json`);
+ const mdPath = path.join(outDir, `${label}.md`);
+ await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n");
+ await writeFile(mdPath, markdown);
+
+ console.log("");
+ console.log(markdown);
+ console.log(`Wrote ${jsonPath}`);
+ console.log(`Wrote ${mdPath}`);
+
+ // THE SAFETY CLAIM, enforced. A bake-off that left a sidecar behind would have
+ // silently made a losing variant look fresh to the whole system.
+ if (sidecarsAfter !== sidecarsBefore) {
+ console.error(
+ `attribution.json count changed: ${sidecarsBefore} -> ${sidecarsAfter}. ` +
+ "This harness must never write one.",
+ );
+ process.exitCode = 1;
+ } else {
+ console.log(`attribution.json on disk after: ${sidecarsAfter} (unchanged).`);
+ }
+
+ if (sum(results.map((r) => r.chunksOk)) === 0) {
+ console.error("No chunk succeeded — nothing was measured.");
+ process.exitCode = 1;
+ }
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/plans/tools/attribution-pilot.ts b/plans/tools/attribution-pilot.ts
@@ -0,0 +1,1122 @@
+#!/usr/bin/env tsx
+// Measure the attribution lanes over a small sample BEFORE deciding whether the
+// text-only one is worth 25-55 GPU-days over the whole corpus.
+//
+// WHY THIS EXISTS. Both lanes shipped with nothing armed, deliberately: the
+// text-only lane costs roughly one model call per transcript CHUNK, which
+// corpus-wide is ~194,000 calls — the same order as the digest sweep, which has
+// itself completed 0.17%. That spend is decided on measurements, the way the
+// digest layer was (pilot -> bake-off -> measured defaults -> sweep), and this
+// is the measurement.
+//
+// WHY IT CALLS attributeOneVideo RATHER THAN BYPASSING IT, unlike
+// digest-bakeoff.ts which drives the prompt modules directly. The bake-off was
+// comparing ENGINES and had to leave nothing behind; this is measuring the
+// SHIPPED PATH — the freshness skip, the downgrade guard, the provenance it
+// records and the sidecar it writes are all part of what is under test, and the
+// record is the artifact a human then reads to judge the labels. So it writes
+// real attribution.json files into the real corpus. That is bounded (one small
+// sidecar per video, nothing consumes it yet) and reversible with one command:
+//
+// find transcripts/channels -name attribution.json -delete
+//
+// NO EDITOR, NO SETTINGS WRITE. attributeOneVideo takes an explicit `settings`
+// override and never calls writeSettings, so the pilot enables the lane
+// IN MEMORY and the operator's settings.json is untouched. A pilot must not be
+// the thing that arms a lane.
+//
+// UNIT DISCIPLINE, and it is not negotiable. plans/STATE.md records a RETRACTED
+// throughput headline that came from pricing a sweep in seconds-per-audio-hour
+// when the unit of work is the chunk (density varies 4x across this corpus). So
+// this harness reports seconds per CHUNK for the text-only lane and calls per
+// VIDEO for the diarized one, prints chunks-per-audio-hour beside any rate, and
+// REFUSES to print a bare seconds-per-audio-hour figure. sweepDaysFromAudio is
+// deliberately not imported.
+//
+// Examples:
+// tsx bin/attribution-pilot.ts --sample bakeoff --dry-run
+// tsx bin/attribution-pilot.ts --sample bakeoff --method text-only --label round1
+// tsx bin/attribution-pilot.ts --sample channel:quarteringvlogs --label autocap
+// tsx bin/attribution-pilot.ts --sample video:ObviousRises-rumble/v6z1o2g \
+// --method diarized --label diarized-smoke
+
+import os from "node:os";
+import path from "node:path";
+import { mkdir, readFile, readdir, writeFile } from "node:fs/promises";
+import { getPaths, type Paths } from "../../common/lib/paths";
+import { parseFlags } from "../../common/bin/_parseFlags";
+import {
+ ATTRIBUTION_PROMPT_VERSION,
+ attributionSpeechSeconds,
+ type AttributionRecord,
+ type AttributionMethod,
+} from "../../common/lib/attribution";
+import { loadAttribution } from "../../common/lib/attribution-server";
+import { loadDiarization } from "../../common/lib/diarization-server";
+import { selectClusterSamples } from "../../common/lib/attributionPrompt";
+import type { AttributionSettings } from "../../common/lib/settings";
+import { OLLAMA_DIGEST_APP_ID } from "../../common/lib/digestApps";
+import {
+ attributeOneVideo,
+ type AttributeOneOutcome,
+} from "../../common/controller/attributeOne";
+import { resolveAttributionTarget } from "../../common/controller/attributionTarget";
+import {
+ DIGEST_OVERLAP_CUES,
+ maxCuesForContext,
+ toHms,
+} from "../../common/lib/digestPrompt";
+import { chunkCuesForContext } from "../../common/lib/transcriptWindow";
+import {
+ chunksPerAudioHour,
+ readChunkCensus,
+ sweepDays,
+ MEASURED_SECONDS_PER_CHUNK,
+} from "../../common/controller/digestPlan";
+import {
+ isCuesJsonFresh,
+ readNormalizedTranscript,
+} from "../../common/controller/normalizeTranscript";
+import type { Cue } from "../../common/lib/vtt";
+
+// The digest lane's own cost on the SAME model, context and eight videos —
+// round 2's 190 engine-seconds over 17 chunks. Printed beside the pilot's
+// number because "is attribution more expensive per chunk than a digest" is one
+// of the two questions this run answers.
+const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2;
+
+// ---------------------------------------------------------------------------
+// The sample
+// ---------------------------------------------------------------------------
+
+type PilotVideo = {
+ slug: string;
+ channelSlug: string;
+ // The DIRECTORY name, which is not always the video id: the bake-off sample's
+ // the-quartering-rumble entry is videoDir v6xl0vu against videoId v6ve4n0 (a
+ // Rumble canonical-id reconciliation). Resolving the path off videoId returns
+ // no-transcript and silently drops a stratum from the sample.
+ videoDir: string;
+ videoId: string;
+ title: string;
+ bucket?: string;
+ durationSeconds: number;
+ dir: string;
+};
+
+// A video that made it far enough to be priced: cues on disk, chunk count known.
+type PricedVideo = PilotVideo & {
+ cues: Cue[];
+ chunks: number;
+ chunkRanges: { start: number; end: number }[];
+};
+
+async function resolveSample(
+ spec: string,
+ paths: Paths,
+): Promise<PilotVideo[]> {
+ if (spec === "bakeoff") {
+ const file = path.join(paths.monorepoRoot, "plans", "bakeoff", "sample.json");
+ const raw = JSON.parse(await readFile(file, "utf8")) as {
+ videos: {
+ slug: string;
+ channelSlug: string;
+ videoId: string;
+ videoDir: string;
+ title: string;
+ bucket: string;
+ durationSeconds: number;
+ }[];
+ };
+ return raw.videos.map((v) => ({
+ slug: v.slug,
+ channelSlug: v.channelSlug,
+ videoDir: v.videoDir,
+ videoId: v.videoId,
+ title: v.title,
+ bucket: v.bucket,
+ durationSeconds: v.durationSeconds,
+ dir: path.join(paths.channelsDir, v.channelSlug, "data", v.videoDir),
+ }));
+ }
+ if (spec.startsWith("channel:")) {
+ const channelSlug = spec.slice("channel:".length);
+ const dataDir = path.join(paths.channelsDir, channelSlug, "data");
+ const entries = await readdir(dataDir, { withFileTypes: true });
+ return entries
+ .filter((e) => e.isDirectory())
+ .map((e) => ({
+ slug: `${channelSlug}/${e.name}`,
+ channelSlug,
+ videoDir: e.name,
+ videoId: e.name,
+ title: "",
+ durationSeconds: 0,
+ dir: path.join(dataDir, e.name),
+ }))
+ .sort((a, b) => a.videoDir.localeCompare(b.videoDir));
+ }
+ if (spec.startsWith("video:")) {
+ const rest = spec.slice("video:".length);
+ const slash = rest.indexOf("/");
+ if (slash <= 0) throw new Error(`--sample video:<slug>/<dir> expected, got "${spec}"`);
+ const channelSlug = rest.slice(0, slash);
+ const videoDir = rest.slice(slash + 1);
+ return [
+ {
+ slug: rest,
+ channelSlug,
+ videoDir,
+ videoId: videoDir,
+ title: "",
+ durationSeconds: 0,
+ dir: path.join(paths.channelsDir, channelSlug, "data", videoDir),
+ },
+ ];
+ }
+ throw new Error(`unknown --sample "${spec}" (bakeoff | channel:<slug> | video:<slug>/<dir>)`);
+}
+
+// Chunk boundaries EXACTLY as findTurns computes them — same chunker, same
+// overlap, same floor/ceil. Every cross-chunk metric rests on this agreeing with
+// the runner, which is why it is derived from the resolved config's numCtx
+// rather than from a literal 600, and why the scorer asserts its count against
+// provenance.chunks and reports null (never 0) on a mismatch.
+function chunkRangesFor(cues: Cue[], maxCues: number): { start: number; end: number }[] {
+ return chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES }).map(
+ (chunk) => {
+ const start = Math.max(0, Math.floor(chunk[0].start));
+ const last = chunk[chunk.length - 1];
+ return { start, end: Math.max(start, Math.ceil(last.end || last.start)) };
+ },
+ );
+}
+
+async function price(video: PilotVideo, maxCues: number): Promise<PricedVideo | null> {
+ const { fresh, cuesPath } = await isCuesJsonFresh(video.dir);
+ if (!fresh) return null;
+ const transcript = await readNormalizedTranscript(cuesPath);
+ const cues = transcript?.cues ?? [];
+ if (!transcript || cues.length === 0) return null;
+ const ranges = chunkRangesFor(cues, maxCues);
+ const lastCue = cues[cues.length - 1];
+ return {
+ ...video,
+ videoId: video.videoId || transcript.id,
+ title: video.title || transcript.title || video.videoId,
+ durationSeconds:
+ video.durationSeconds > 0
+ ? video.durationSeconds
+ : Math.round(transcript.duration || lastCue.end || lastCue.start || 0),
+ cues,
+ chunks: ranges.length,
+ chunkRanges: ranges,
+ };
+}
+
+// ---------------------------------------------------------------------------
+// Per-call timing, parsed out of the log stream digestApps already emits
+// ---------------------------------------------------------------------------
+
+// `ollama qwen2.5:7b: 4102 in / 218 out tokens in 5.0s wall (load 0.4s, prefill
+// 0.1s, decode 3.0s, engine 3.5s) (num_ctx 8192)`
+//
+// No new instrumentation: attributeOneVideo forwards the caller's onLog straight
+// into the app's run(), so one closure sees every call of the video it wraps.
+const CALL_RE =
+ /^ollama (\S+): (\d+|\?) in \/ (\d+|\?) out tokens in ([\d.]+)s wall(?: \(([^)]*)\))?/;
+const PHASE_RE = /^(load|prefill|decode|engine) ([\d.]+)s$/;
+
+type CallTiming = {
+ wallSeconds: number;
+ loadSeconds?: number;
+ prefillSeconds?: number;
+ decodeSeconds?: number;
+ engineSeconds?: number;
+ inputTokens?: number;
+ outputTokens?: number;
+};
+
+function parseCall(line: string): CallTiming | null {
+ const m = CALL_RE.exec(line);
+ if (!m) return null;
+ const out: CallTiming = { wallSeconds: Number(m[4]) };
+ if (m[2] !== "?") out.inputTokens = Number(m[2]);
+ if (m[3] !== "?") out.outputTokens = Number(m[3]);
+ // nsToMs OMITS a phase rather than zeroing it, and the whole parenthetical is
+ // dropped when the engine reports nothing — so an absent phase stays absent
+ // here too, and every engine-derived figure is averaged over its own
+ // denominator rather than over the call count.
+ for (const part of (m[5] ?? "").split(", ")) {
+ const p = PHASE_RE.exec(part.trim());
+ if (!p) continue;
+ const seconds = Number(p[2]);
+ if (p[1] === "load") out.loadSeconds = seconds;
+ else if (p[1] === "prefill") out.prefillSeconds = seconds;
+ else if (p[1] === "decode") out.decodeSeconds = seconds;
+ else out.engineSeconds = seconds;
+ }
+ return out;
+}
+
+// ---------------------------------------------------------------------------
+// Metrics
+// ---------------------------------------------------------------------------
+
+// Roles a label can carry that name nobody. isUselessSpeakerLabel already blocks
+// "Speaker 1" before the write, so the record cannot contain one; this measures
+// the WIDER set the runner permits. Anchored whole-string, not a prefix match:
+// "Host" is generic, "Host Jane Doe" is not.
+const GENERIC_LABEL_RE =
+ /^(the\s+)?(co-?host|host|guest|caller|narrator|interviewer|moderator|panelist|announcer|commentator|reporter|audience|crowd|man|woman|male|female|person|voice|speaker)(\s*\d+)?$/i;
+
+type VideoMetrics = {
+ labels: string[];
+ labelsPerVideo: number;
+ // Null rather than 0 whenever the property could not be measured — a single
+ // chunk has no seam to cross, and a chunk-count mismatch means the
+ // reconstruction disagrees with the runner. Reporting 0 for "not measured"
+ // is how a harness reports a perfect score for something that never ran.
+ newLabelsPerChunk: number | null;
+ singletonLabelRate: number | null;
+ dominantSpeakerShare: number;
+ talkShares: number[];
+ nearDuplicateLabelRate: number;
+ nearDuplicatePairs: string[];
+ genericLabelRate: number;
+ genericSecondsShare: number;
+ nonAsciiLabelRate: number;
+ coverageRate: number;
+ maxGapSeconds: number;
+ medianGapSeconds: number;
+ markFreeChunks: number | null;
+ segments: number;
+ chunkReconstructionOk: boolean;
+ labelFirstChunk: (number | null)[];
+ labelChunkCount: (number | null)[];
+};
+
+function normalizeForCompare(label: string): string {
+ return label
+ .trim()
+ .replace(/\s+/g, " ")
+ .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "")
+ .toLowerCase();
+}
+
+// Labels the runner's deliberately conservative normalizeLabel keeps apart but
+// that are obviously one person ("Jane" / "Jane Doe"). The runner is right to
+// refuse the merge — a merged speaker silently attributes words to the wrong
+// person — but a SCORER may flag, and this rate is the direct read on whether
+// that conservatism is set right.
+function nearDuplicate(a: string, b: string): boolean {
+ const ta = new Set(normalizeForCompare(a).split(" ").filter(Boolean));
+ const tb = new Set(normalizeForCompare(b).split(" ").filter(Boolean));
+ if (ta.size === 0 || tb.size === 0) return false;
+ const [small, big] = ta.size <= tb.size ? [ta, tb] : [tb, ta];
+ let shared = 0;
+ for (const t of small) if (big.has(t)) shared++;
+ if (shared === small.size) return true; // proper token-subset
+ for (const t of small) if (t.length >= 4 && big.has(t)) return true;
+ return false;
+}
+
+function median(values: number[]): number {
+ if (values.length === 0) return 0;
+ const s = [...values].sort((a, b) => a - b);
+ const mid = Math.floor(s.length / 2);
+ return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2;
+}
+
+function scoreRecord(record: AttributionRecord, video: PricedVideo): VideoMetrics {
+ const ranges = video.chunkRanges;
+ // The runner records what it actually chunked; if our reconstruction differs,
+ // every chunk-indexed number below is meaningless and is reported as null.
+ const recordedChunks = record.provenance.chunks;
+ const reconstructionOk =
+ recordedChunks === undefined || recordedChunks === ranges.length;
+ const multiChunk = reconstructionOk && ranges.length > 1;
+
+ const chunkOf = (t: number): number => {
+ // The 40-cue overlap means a seam second sits in two chunks; first match is
+ // the earlier one, which is the chunk that ESTABLISHED the label.
+ for (let i = 0; i < ranges.length; i++) {
+ if (t >= ranges[i].start && t <= ranges[i].end) return i;
+ }
+ return t < ranges[0].start ? 0 : ranges.length - 1;
+ };
+
+ const chunksByLabel = new Map<number, Set<number>>();
+ for (const s of record.segments) {
+ const set = chunksByLabel.get(s.speaker) ?? new Set<number>();
+ set.add(chunkOf(s.start));
+ chunksByLabel.set(s.speaker, set);
+ }
+
+ const labels = record.speakers.map((s) => s.label);
+ const seconds = record.speakers.map((s) => s.seconds ?? 0);
+ const totalSeconds = seconds.reduce((a, b) => a + b, 0);
+ const labelFirstChunk = record.speakers.map((s) => {
+ const set = chunksByLabel.get(s.index);
+ if (!reconstructionOk || !set || set.size === 0) return null;
+ return Math.min(...set);
+ });
+ const labelChunkCount = record.speakers.map((s) => {
+ const set = chunksByLabel.get(s.index);
+ if (!reconstructionOk || !set) return null;
+ return set.size;
+ });
+
+ // THE HEADLINE. Labels first appearing after chunk 0, per later chunk. Near 0
+ // means the cast stabilised; >= 1 means the model reinvented it every chunk
+ // and the lane is not delivering the one property it exists for.
+ const newLabelsPerChunk = multiChunk
+ ? labelFirstChunk.filter((c) => c !== null && c > 0).length / (ranges.length - 1)
+ : null;
+ const singletonLabelRate =
+ multiChunk && labels.length > 0
+ ? labelChunkCount.filter((c) => c === 1).length / labels.length
+ : null;
+
+ // THE DEGENERATE-CASE GUARD. One label at ~100% of talk time scores a perfect
+ // 0 on the headline while being worthless, and the headline alone cannot tell
+ // those apart.
+ const dominantSpeakerShare =
+ totalSeconds > 0 ? Math.max(...seconds) / totalSeconds : 0;
+
+ const dupPairs: string[] = [];
+ const involved = new Set<number>();
+ for (let i = 0; i < labels.length; i++) {
+ for (let j = i + 1; j < labels.length; j++) {
+ if (!nearDuplicate(labels[i], labels[j])) continue;
+ dupPairs.push(`${labels[i]} ~ ${labels[j]}`);
+ involved.add(i);
+ involved.add(j);
+ }
+ }
+
+ const genericIdx = labels
+ .map((l, i) => (GENERIC_LABEL_RE.test(l.trim()) ? i : -1))
+ .filter((i) => i >= 0);
+ const genericSeconds = genericIdx.reduce((a, i) => a + seconds[i], 0);
+
+ // COVERAGE, and what it is worth. Text-only marks are TILED — each runs to the
+ // next change, the last to the end — so this is ~1.0 by construction. It is a
+ // structural check (anything else is a bug), not a quality signal. The honest
+ // coverage metric for this lane is markFreeChunks.
+ const duration = video.durationSeconds || 0;
+ const covered = attributionSpeechSeconds(record);
+ const sorted = [...record.segments].sort((a, b) => a.start - b.start);
+ const gaps: number[] = [];
+ let cursor = 0;
+ for (const s of sorted) {
+ if (s.start > cursor) gaps.push(s.start - cursor);
+ cursor = Math.max(cursor, s.end);
+ }
+ if (duration > cursor) gaps.push(duration - cursor);
+
+ const markFreeChunks = reconstructionOk
+ ? ranges.filter(
+ (r) => !record.segments.some((s) => s.start >= r.start && s.start <= r.end),
+ ).length
+ : null;
+
+ return {
+ labels,
+ labelsPerVideo: labels.length,
+ newLabelsPerChunk,
+ singletonLabelRate,
+ dominantSpeakerShare,
+ talkShares: totalSeconds > 0 ? seconds.map((s) => s / totalSeconds) : seconds.map(() => 0),
+ nearDuplicateLabelRate: labels.length > 0 ? involved.size / labels.length : 0,
+ nearDuplicatePairs: dupPairs,
+ genericLabelRate: labels.length > 0 ? genericIdx.length / labels.length : 0,
+ genericSecondsShare: totalSeconds > 0 ? genericSeconds / totalSeconds : 0,
+ nonAsciiLabelRate:
+ labels.length > 0
+ ? labels.filter((l) => /[^\x20-\x7E]/.test(l)).length / labels.length
+ : 0,
+ coverageRate: duration > 0 ? covered / duration : 0,
+ maxGapSeconds: gaps.length ? Math.max(...gaps) : 0,
+ medianGapSeconds: median(gaps),
+ markFreeChunks,
+ segments: record.segments.length,
+ chunkReconstructionOk: reconstructionOk,
+ labelFirstChunk,
+ labelChunkCount,
+ };
+}
+
+// ---------------------------------------------------------------------------
+// The run
+// ---------------------------------------------------------------------------
+
+type VideoResult = {
+ slug: string;
+ bucket?: string;
+ title: string;
+ durationSeconds: number;
+ expectedChunks: number;
+ outcome: AttributeOneOutcome;
+ // Wall time around attributeOneVideo — the realistic ceiling on this shared
+ // box, and NOT the same quantity as the engine's own number.
+ videoWallSeconds: number;
+ calls: CallTiming[];
+ unparsedLogLines: number;
+ recordedChunks?: number;
+ recordedChunksOk?: number;
+ warningsByCode: Record<string, number>;
+ metrics: VideoMetrics | null;
+ // Diarized lane only.
+ clustersOffered?: number;
+ clustersNamed?: number;
+ confidences?: number[];
+};
+
+function boxState(): Record<string, unknown> {
+ const load = os.loadavg();
+ return {
+ loadavg: load.map((n) => Number(n.toFixed(2))),
+ freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)),
+ totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)),
+ };
+}
+
+function sum(values: number[]): number {
+ return values.reduce((a, b) => a + b, 0);
+}
+
+async function runVideo(
+ video: PricedVideo,
+ method: AttributionMethod,
+ settings: AttributionSettings,
+ paths: Paths,
+ force: boolean,
+ log: (m: string) => void,
+): Promise<VideoResult> {
+ const calls: CallTiming[] = [];
+ let unparsed = 0;
+ const onLog = (line: string) => {
+ const call = parseCall(line);
+ if (call) calls.push(call);
+ // Only the ollama call lines are parseable; the runner's own progress lines
+ // are not, and are not counted as drift.
+ else if (line.startsWith("ollama ")) unparsed++;
+ };
+
+ const started = Date.now();
+ const outcome = await attributeOneVideo({
+ paths,
+ videoDir: video.dir,
+ videoId: video.videoId,
+ channelSlug: video.channelSlug,
+ method,
+ settings,
+ force,
+ onLog,
+ });
+ const videoWallSeconds = (Date.now() - started) / 1000;
+
+ const record = await loadAttribution(video.dir);
+ const warningsByCode: Record<string, number> = {};
+ for (const w of record?.warnings ?? []) {
+ warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1;
+ }
+
+ const result: VideoResult = {
+ slug: video.slug,
+ ...(video.bucket ? { bucket: video.bucket } : {}),
+ title: video.title,
+ durationSeconds: video.durationSeconds,
+ expectedChunks: video.chunks,
+ outcome,
+ videoWallSeconds,
+ calls,
+ unparsedLogLines: unparsed,
+ ...(record?.provenance.chunks !== undefined
+ ? { recordedChunks: record.provenance.chunks, recordedChunksOk: record.provenance.chunksOk }
+ : {}),
+ warningsByCode,
+ metrics: record && outcome === "attributed" ? scoreRecord(record, video) : null,
+ };
+
+ if (method === "diarized") {
+ const diarization = await loadDiarization(video.dir);
+ if (diarization) {
+ result.clustersOffered = selectClusterSamples(diarization.turns, video.cues).length;
+ }
+ result.clustersNamed = record?.speakers.length ?? 0;
+ result.confidences = (record?.speakers ?? [])
+ .map((s) => s.confidence)
+ .filter((c): c is number => typeof c === "number");
+ }
+
+ const m = result.metrics;
+ log(
+ ` ${video.slug}${video.bucket ? ` [${video.bucket}]` : ""} ${video.chunks} chunk(s) → ` +
+ `${outcome}` +
+ (m
+ ? `, ${m.labelsPerVideo} label(s), ${m.segments} segment(s), ` +
+ `new/chunk ${m.newLabelsPerChunk === null ? "n/a" : m.newLabelsPerChunk.toFixed(2)}, ` +
+ `top share ${(m.dominantSpeakerShare * 100).toFixed(0)}%`
+ : "") +
+ `, ${videoWallSeconds.toFixed(0)}s wall` +
+ (calls.length ? ` / ${sum(calls.map((c) => c.engineSeconds ?? 0)).toFixed(0)}s engine` : ""),
+ );
+ return result;
+}
+
+// ---------------------------------------------------------------------------
+// Reporting
+// ---------------------------------------------------------------------------
+
+type CostSummary = {
+ videosMeasured: number;
+ chunks: number;
+ chunksOk: number;
+ calls: number;
+ callsWithEngineTiming: number;
+ audioSeconds: number;
+ chunksPerAudioHour: number;
+ secondsPerChunkWall: number | null;
+ secondsPerChunkCallWall: number | null;
+ secondsPerChunkEngine: number | null;
+ secondsPerChunkEngineExLoad: number | null;
+ meanLoadSeconds: number | null;
+ decodeTokensPerSecond: number | null;
+ unparsedLogLines: number;
+};
+
+// Throughput is computed ONLY over videos the run actually attributed: an
+// already-exists short-circuits in milliseconds and would otherwise report an
+// absurd chunk rate.
+function summarizeCost(results: VideoResult[]): CostSummary {
+ const measured = results.filter((r) => r.outcome === "attributed");
+ const calls = measured.flatMap((r) => r.calls);
+ const withEngine = calls.filter((c) => c.engineSeconds !== undefined);
+ const chunks = sum(measured.map((r) => r.recordedChunks ?? r.expectedChunks));
+ const chunksOk = sum(measured.map((r) => r.recordedChunksOk ?? r.calls.length));
+ const audioSeconds = sum(measured.map((r) => r.durationSeconds));
+ const wall = sum(measured.map((r) => r.videoWallSeconds));
+ const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0));
+ const outTokens = sum(calls.map((c) => c.outputTokens ?? 0));
+ const withLoad = calls.filter((c) => c.loadSeconds !== undefined);
+ return {
+ videosMeasured: measured.length,
+ chunks,
+ chunksOk,
+ calls: calls.length,
+ callsWithEngineTiming: withEngine.length,
+ audioSeconds,
+ chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds),
+ secondsPerChunkWall: chunks > 0 ? wall / chunks : null,
+ secondsPerChunkCallWall: calls.length ? sum(calls.map((c) => c.wallSeconds)) / calls.length : null,
+ secondsPerChunkEngine: withEngine.length
+ ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length
+ : null,
+ secondsPerChunkEngineExLoad: withEngine.length
+ ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) /
+ withEngine.length
+ : null,
+ meanLoadSeconds: withLoad.length
+ ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length
+ : null,
+ decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null,
+ unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)),
+ };
+}
+
+function pct(n: number): string {
+ return `${(n * 100).toFixed(0)}%`;
+}
+
+function fixed(n: number | null, digits = 2): string {
+ return n === null ? "n/a" : n.toFixed(digits);
+}
+
+// Verbatim transcript at a speaker's longest segments. WITHOUT this nobody can
+// judge whether a label is right, and no aggregate metric substitutes for
+// reading it — which is why the Markdown artifact, not the JSON, is the point of
+// the exercise.
+function excerpt(cues: Cue[], start: number, end: number, maxChars = 420): string[] {
+ const lines: string[] = [];
+ let chars = 0;
+ for (const c of cues) {
+ if (c.start < start) continue;
+ if (c.start > end) break;
+ const text = c.text.trim().replace(/\s+/g, " ");
+ if (!text) continue;
+ lines.push(`[${toHms(c.start)}] ${text}`);
+ chars += text.length;
+ if (chars >= maxChars) break;
+ }
+ return lines;
+}
+
+function qualityMarkdown(
+ label: string,
+ method: AttributionMethod,
+ results: VideoResult[],
+ priced: Map<string, PricedVideo>,
+ records: Map<string, AttributionRecord>,
+): string {
+ const out: string[] = [];
+ out.push(`# Attribution pilot — ${label} (${method}): what the labels actually say`);
+ out.push("");
+ out.push(
+ "Read this file, not the metrics, to answer *are the labels right*. Each speaker",
+ "is shown with its talk-time share and the verbatim transcript at its longest",
+ "segments — the same text, in the same `[HH:MM:SS] line` form, the model saw.",
+ );
+ out.push("");
+ for (const r of results) {
+ const video = priced.get(r.slug);
+ const record = records.get(r.slug);
+ out.push(`## ${r.slug}${r.bucket ? ` — ${r.bucket}` : ""}`);
+ out.push("");
+ out.push(`**${r.title || "(untitled)"}**`);
+ out.push("");
+ out.push(
+ `${toHms(r.durationSeconds)} · ${r.expectedChunks} chunk(s) · outcome \`${r.outcome}\`` +
+ (r.recordedChunks !== undefined
+ ? ` · ${r.recordedChunksOk}/${r.recordedChunks} chunk(s) ok`
+ : ""),
+ );
+ out.push("");
+ if (!record || !video || !r.metrics) {
+ out.push("_No record was written for this video._");
+ out.push("");
+ continue;
+ }
+ const m = r.metrics;
+ for (let i = 0; i < record.speakers.length; i++) {
+ const speaker = record.speakers[i];
+ const share = m.talkShares[i] ?? 0;
+ const first = m.labelFirstChunk[i];
+ const present = m.labelChunkCount[i];
+ out.push(
+ `### ${speaker.label} — ${pct(share)} of attributed time` +
+ (speaker.confidence !== undefined ? `, confidence ${speaker.confidence.toFixed(2)}` : "") +
+ (first !== null ? `, first seen in chunk ${first + 1}` : "") +
+ (present !== null ? `, present in ${present}/${r.expectedChunks} chunk(s)` : ""),
+ );
+ out.push("");
+ const mine = record.segments
+ .filter((s) => s.speaker === speaker.index)
+ .sort((a, b) => b.end - b.start - (a.end - a.start))
+ .slice(0, 3)
+ .sort((a, b) => a.start - b.start);
+ if (mine.length === 0) {
+ out.push("_No segments._");
+ out.push("");
+ continue;
+ }
+ for (const seg of mine) {
+ out.push(`_${toHms(seg.start)} – ${toHms(seg.end)}_`);
+ out.push("");
+ out.push("```");
+ const lines = excerpt(video.cues, seg.start, Math.min(seg.end, seg.start + 90));
+ out.push(...(lines.length ? lines : ["(no cue text in range)"]));
+ out.push("```");
+ out.push("");
+ }
+ }
+ if (m.nearDuplicatePairs.length) {
+ out.push(`Near-duplicate label pairs: ${m.nearDuplicatePairs.map((p) => `\`${p}\``).join(", ")}`);
+ out.push("");
+ }
+ out.push("- [ ] labels are right");
+ out.push("- [ ] a label is wrong (which: ____)");
+ out.push("- [ ] a name appears that the transcript never states");
+ out.push("- [ ] one person is split across two labels");
+ out.push("- [ ] two people are merged into one label");
+ out.push("");
+ }
+ return out.join("\n");
+}
+
+function reportMarkdown(args: {
+ label: string;
+ method: AttributionMethod;
+ sample: string;
+ model: string;
+ numCtx?: number;
+ maxCues: number;
+ results: VideoResult[];
+ cost: CostSummary;
+ census: { chunks: number; videos: number; chunksPerAudioHour: number; estimated: number; statsSchemaStale: boolean } | null;
+ boxStart: Record<string, unknown>;
+ boxEnd: Record<string, unknown>;
+ startedAt: string;
+}): string {
+ const { results, cost } = args;
+ const out: string[] = [];
+ out.push(`# Attribution pilot — ${args.label}`);
+ out.push("");
+ out.push(
+ `Lane **${args.method}** · sample \`${args.sample}\` · ${results.length} video(s) · ` +
+ `model \`${args.model}\` @ numCtx ${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`,
+ );
+ out.push("");
+
+ const outcomes: Record<string, number> = {};
+ for (const r of results) outcomes[r.outcome] = (outcomes[r.outcome] ?? 0) + 1;
+ out.push(`Outcomes: ${Object.entries(outcomes).map(([k, v]) => `${k} ${v}`).join(", ")}`);
+ out.push("");
+
+ if (args.method === "text-only") {
+ out.push("## Cross-chunk identity — the property the lane exists for");
+ out.push("");
+ out.push("| video | chunks | labels | new labels/chunk | singleton rate | top share | near-dup | generic |");
+ out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |");
+ for (const r of results) {
+ const m = r.metrics;
+ out.push(
+ `| ${r.slug} | ${r.expectedChunks} | ${m?.labelsPerVideo ?? "—"} | ${
+ m ? fixed(m.newLabelsPerChunk) : "—"
+ } | ${m ? (m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate)) : "—"} | ${
+ m ? pct(m.dominantSpeakerShare) : "—"
+ } | ${m ? pct(m.nearDuplicateLabelRate) : "—"} | ${m ? pct(m.genericLabelRate) : "—"} |`,
+ );
+ }
+ out.push("");
+ const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null);
+ const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null);
+ if (withSeams.length) {
+ out.push(
+ `**Headline** — new labels per chunk after the first, over ${withSeams.length} multi-chunk video(s): ` +
+ `mean ${fixed(sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length)}, ` +
+ `max ${fixed(Math.max(...withSeams.map((m) => m.newLabelsPerChunk ?? 0)))}.`,
+ );
+ out.push("");
+ out.push(
+ "Read it WITH the top-share column: one label at ~100% scores a perfect 0 here",
+ "and is degenerate, not good.",
+ );
+ out.push("");
+ }
+ const degenerate = scored.filter((m) => m.labelsPerVideo === 1);
+ out.push(
+ `Single-label (degenerate-risk) videos: ${degenerate.length} of ${scored.length}. ` +
+ `Videos whose chunk reconstruction disagreed with the record: ${
+ scored.filter((m) => !m.chunkReconstructionOk).length
+ }.`,
+ );
+ out.push("");
+
+ out.push("## Coverage and label quality");
+ out.push("");
+ out.push("| video | segments | mark-free chunks | coverage | max gap | median gap | non-ascii labels |");
+ out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: |");
+ for (const r of results) {
+ const m = r.metrics;
+ out.push(
+ `| ${r.slug} | ${m?.segments ?? "—"} | ${
+ m ? (m.markFreeChunks === null ? "n/a" : m.markFreeChunks) : "—"
+ } | ${m ? pct(m.coverageRate) : "—"} | ${m ? toHms(Math.round(m.maxGapSeconds)) : "—"} | ${
+ m ? toHms(Math.round(m.medianGapSeconds)) : "—"
+ } | ${m ? pct(m.nonAsciiLabelRate) : "—"} |`,
+ );
+ }
+ out.push("");
+ out.push(
+ "Coverage is ~1.0 BY CONSTRUCTION for this lane — marks are tiled, each running",
+ "to the next change — so it is a structural check, not a quality signal. The",
+ "honest coverage metric here is mark-free chunks.",
+ );
+ out.push("");
+ } else {
+ out.push("## Diarized lane — the claim under test is ONE call per video");
+ out.push("");
+ out.push("| video | calls | clusters offered | clusters named | confidences |");
+ out.push("| --- | ---: | ---: | ---: | --- |");
+ for (const r of results) {
+ out.push(
+ `| ${r.slug} | ${r.calls.length} | ${r.clustersOffered ?? "—"} | ${r.clustersNamed ?? "—"} | ${
+ (r.confidences ?? []).map((c) => c.toFixed(2)).join(", ") || "—"
+ } |`,
+ );
+ }
+ out.push("");
+ const calls = sum(results.map((r) => r.calls.length));
+ out.push(
+ `**Calls per video: ${(calls / Math.max(1, results.length)).toFixed(2)}** ` +
+ `(${calls} call(s) over ${results.length} video(s)). The claim holds iff this is 1.00.`,
+ );
+ out.push("");
+ }
+
+ const warnings: Record<string, number> = {};
+ for (const r of results) {
+ for (const [code, n] of Object.entries(r.warningsByCode)) {
+ warnings[code] = (warnings[code] ?? 0) + n;
+ }
+ }
+ out.push(
+ `Warnings by code: ${
+ Object.keys(warnings).length
+ ? Object.entries(warnings).map(([k, v]) => `${k} ${v}`).join(", ")
+ : "none"
+ }`,
+ );
+ out.push("");
+
+ out.push("## Cost");
+ out.push("");
+ out.push(
+ `Measured over the ${cost.videosMeasured} video(s) this run actually attributed ` +
+ `(${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s), ` +
+ `${cost.callsWithEngineTiming} with engine timing). Sample density ` +
+ `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` +
+ `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`,
+ );
+ out.push("");
+ out.push("| quantity | value |");
+ out.push("| --- | ---: |");
+ out.push(`| s/chunk, wall (video wall ÷ chunks) | ${fixed(cost.secondsPerChunkWall, 1)} |`);
+ out.push(`| s/chunk, per-call wall | ${fixed(cost.secondsPerChunkCallWall, 1)} |`);
+ out.push(`| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} |`);
+ out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} |`);
+ out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} |`);
+ out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} |`);
+ out.push(
+ `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} |`,
+ );
+ out.push("");
+
+ if (args.method === "text-only" && args.census) {
+ const idle = cost.secondsPerChunkEngineExLoad;
+ const contended = cost.secondsPerChunkWall;
+ out.push("## Corpus projection");
+ out.push("");
+ out.push(
+ `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` +
+ `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` +
+ (args.census.estimated ? `, ${args.census.estimated} estimated` : "") +
+ (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") +
+ ".",
+ );
+ out.push("");
+ out.push(
+ `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` +
+ `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended**` +
+ ` — from n = ${cost.videosMeasured} video(s) / ${cost.chunks} chunk(s), one box, one hour.`,
+ );
+ out.push("");
+ out.push(
+ "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that",
+ "has completed 0.17% of its own ~194k calls, with nothing arbitrating between",
+ "them. One lane, no parallelism.",
+ );
+ out.push("");
+ }
+
+ out.push("## Box");
+ out.push("");
+ out.push(`Start: \`${JSON.stringify(args.boxStart)}\``);
+ out.push("");
+ out.push(`End: \`${JSON.stringify(args.boxEnd)}\``);
+ out.push("");
+ if (cost.unparsedLogLines > 0) {
+ out.push(
+ `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`,
+ );
+ out.push("");
+ }
+
+ out.push("## Units");
+ out.push("");
+ out.push(
+ "- The unit of text-only work is the **chunk** (one model call). Seconds-per-",
+ " audio-hour is not a unit here: chunk density varies 4× across this corpus, and",
+ " pricing in it is the error behind a retracted throughput headline. This report",
+ " does not print one.",
+ "- The unit of diarized work is **calls per video**.",
+ "- Wall and engine are different quantities and are never substituted for each",
+ " other. Wall is the realistic ceiling on this shared box; engine excl. load is",
+ " the floor.",
+ "- Every rate carries its denominator. At n ≈ 8 videos a bare percentage would be",
+ " a lie of precision.",
+ );
+ out.push("");
+ return out.join("\n");
+}
+
+// ---------------------------------------------------------------------------
+
+async function main(): Promise<void> {
+ const flags = parseFlags(process.argv.slice(2));
+ const paths = getPaths();
+ const sampleSpec = flags.sample ?? "bakeoff";
+ const method: AttributionMethod = flags.method === "diarized" ? "diarized" : "text-only";
+ const dryRun = flags["dry-run"] === "true";
+ const force = flags.force === "true";
+ const asJson = flags.json === "true";
+ const label = flags.label ?? `${method}-${sampleSpec.replace(/[^a-zA-Z0-9]+/g, "-")}`;
+
+ // THE LANE IS ENABLED IN MEMORY. settings.json has no `attribution` key at
+ // all, so without this override every video returns "disabled" and the run
+ // reports a flawless zero-cost hour. Nothing here is persisted.
+ const pilotSettings: AttributionSettings = {
+ enabled: true,
+ appId: OLLAMA_DIGEST_APP_ID,
+ model: "",
+ diarizedEnabled: method === "diarized",
+ textOnlyEnabled: method === "text-only",
+ promptVersion: ATTRIBUTION_PROMPT_VERSION,
+ };
+ const { app, config, modelRequested } = resolveAttributionTarget(method, pilotSettings);
+ const maxCues = maxCuesForContext(config.numCtx);
+
+ const videos = await resolveSample(sampleSpec, paths);
+ const priced: PricedVideo[] = [];
+ const skipped: string[] = [];
+ for (const v of videos) {
+ const p = await price(v, maxCues);
+ if (p) priced.push(p);
+ else skipped.push(v.slug);
+ }
+
+ const totalChunks = sum(priced.map((p) => p.chunks));
+ const totalAudio = sum(priced.map((p) => p.durationSeconds));
+ const census = await readChunkCensus(paths, maxCues);
+
+ console.log(
+ `Attribution pilot — lane ${method}, sample ${sampleSpec}, app ${app.id}, ` +
+ `model ${modelRequested}, numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`,
+ );
+ console.log(`Settings override (in memory only): ${JSON.stringify(pilotSettings)}`);
+ console.log("");
+ for (const p of priced) {
+ console.log(
+ ` ${p.slug}${p.bucket ? ` [${p.bucket}]` : ""} ${toHms(p.durationSeconds)} ` +
+ `${p.cues.length} cues → ${p.chunks} chunk(s)`,
+ );
+ }
+ if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`);
+ console.log("");
+ const projectedMinutes = (totalChunks * MEASURED_SECONDS_PER_CHUNK) / 60;
+ console.log(
+ `${priced.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` +
+ `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour. ` +
+ `At the digest lane's measured ${MEASURED_SECONDS_PER_CHUNK}s/chunk that is ` +
+ `~${projectedMinutes.toFixed(0)} minute(s).`,
+ );
+ if (census) {
+ console.log(
+ `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s) ` +
+ `/ ${census.chunksPerAudioHour.toFixed(2)} chunks per audio-hour` +
+ (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""),
+ );
+ }
+ console.log("");
+
+ if (dryRun) {
+ console.log("--dry-run: priced only, no model call and no write.");
+ return;
+ }
+
+ const boxStart = boxState();
+ const startedAt = new Date().toISOString();
+ const results: VideoResult[] = [];
+ const pricedBySlug = new Map(priced.map((p) => [p.slug, p]));
+ const records = new Map<string, AttributionRecord>();
+ for (const video of priced) {
+ const r = await runVideo(video, method, pilotSettings, paths, force, (m) => console.log(m));
+ results.push(r);
+ const record = await loadAttribution(video.dir);
+ if (record) records.set(video.slug, record);
+ }
+ const boxEnd = boxState();
+
+ const cost = summarizeCost(results);
+ const report = {
+ version: 1 as const,
+ label,
+ method,
+ sample: sampleSpec,
+ startedAt,
+ finishedAt: new Date().toISOString(),
+ engine: {
+ appId: app.id,
+ modelRequested,
+ numCtx: config.numCtx,
+ maxCues,
+ overlapCues: DIGEST_OVERLAP_CUES,
+ promptVersion: ATTRIBUTION_PROMPT_VERSION,
+ },
+ settingsOverride: pilotSettings,
+ skipped,
+ box: { start: boxStart, end: boxEnd },
+ census,
+ cost,
+ baseline: { digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE },
+ videos: results,
+ };
+
+ const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot");
+ await mkdir(outDir, { recursive: true });
+ const jsonPath = path.join(outDir, `${label}.json`);
+ const mdPath = path.join(outDir, `${label}.md`);
+ const qualityPath = path.join(outDir, `${label}-quality.md`);
+ await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n");
+ await writeFile(
+ mdPath,
+ reportMarkdown({
+ label,
+ method,
+ sample: sampleSpec,
+ model: modelRequested,
+ numCtx: config.numCtx,
+ maxCues,
+ results,
+ cost,
+ census,
+ boxStart,
+ boxEnd,
+ startedAt,
+ }),
+ );
+ await writeFile(qualityPath, qualityMarkdown(label, method, results, pricedBySlug, records));
+
+ console.log("");
+ if (asJson) {
+ console.log(JSON.stringify(report, null, 2));
+ } else {
+ console.log(
+ reportMarkdown({
+ label,
+ method,
+ sample: sampleSpec,
+ model: modelRequested,
+ numCtx: config.numCtx,
+ maxCues,
+ results,
+ cost,
+ census,
+ boxStart,
+ boxEnd,
+ startedAt,
+ }),
+ );
+ }
+ console.log(`Wrote ${jsonPath}`);
+ console.log(`Wrote ${mdPath}`);
+ console.log(`Wrote ${qualityPath}`);
+
+ // A run in which NOTHING was attributed and nothing was already fresh is the
+ // silent-no-op failure: the lane was off, or every video lacked a transcript.
+ // It must not exit 0 looking like a clean measurement.
+ const productive = results.filter(
+ (r) => r.outcome === "attributed" || r.outcome === "already-exists",
+ ).length;
+ if (productive === 0) {
+ console.error("No video was attributed and none was already fresh — nothing was measured.");
+ process.exitCode = 1;
+ }
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/plans/tools/digest-bakeoff.ts b/plans/tools/digest-bakeoff.ts
@@ -0,0 +1,1425 @@
+#!/usr/bin/env tsx
+// Score digest engine candidates against a FIXED sample, without writing a
+// single digest to disk.
+//
+// WHY THIS IS A SCRIPT AND NOT INFRASTRUCTURE. It answers one question — which
+// (model x context x timestamp mode) should carry a multi-week sweep — and the
+// answer is a number in a table, not a feature. It therefore drives digestApps +
+// digestPrompt + digestParse DIRECTLY and scores in memory. It NEVER calls
+// digestVideo or writeDigestSection, so a losing candidate cannot leave anything
+// behind in the corpus, and no freshness record has to be invalidated afterwards
+// to undo a round.
+//
+// THROUGHPUT IS A FIRST-CLASS METRIC, not a footnote. Measured: 46 s per
+// 8,194-token chunk on qwen2.5:7b, which over the corpus is ~64 days on one
+// lane. A 14B at half the context roughly doubles the chunk count and halves the
+// token rate — order 250 days. A candidate can therefore be ruled out on
+// projected sweep days alone, however good its chapters look, which is why every
+// run prints days alongside quality.
+//
+// BOUNDARY ACCURACY, AND WHY IT WAS ADDED. Every metric this harness scored
+// originally — zeroYieldRate, chaptersPerHour, rejectionRate, maxGapSeconds,
+// genericTitleRate, duplicateTitleRate — is a DEFECT COUNTER: it says how
+// malformed the output is, never whether a boundary landed in the right place.
+// So the harness could rank candidates by which was least broken and still not
+// say which segmented better. lib/boundaryScore.ts closes that with an oracle
+// the model never sees: the UPLOADER's own chapter marks from
+// metadata.info.json, present on ~12,139 videos that also have a transcript and
+// >=4 chapters. Boundaries are facts, so scoring against them reproduces
+// nothing; the uploader's chapter TITLES are expression and are never scored
+// against, only used to spot boilerplate ("Intro", "Sponsor").
+//
+// Modes:
+//
+// --pick scan the stats cache and write the fixed stratified sample
+// (plans/bakeoff/sample.json). Run ONCE. A moving sample
+// makes the comparison between rounds meaningless.
+// --pick-chapters write a sample restricted to videos that carry >=4
+// uploader chapters, so boundary accuracy is scorable. Kept
+// as a SEPARATE file (speaker-sample.json /
+// chapter-sample.json) so rounds 1-2 stay comparable.
+// --score-existing score the ai-digest.json files ALREADY on disk against
+// their uploader chapters. Zero GPU, zero writes — the way
+// to prove the scorer works before spending a generation
+// run on it.
+// (default) run the candidates over that sample and write a JSON +
+// Markdown report under plans/bakeoff/.
+//
+// Examples:
+// tsx bin/digest-bakeoff.ts --pick
+// tsx bin/digest-bakeoff.ts --pick-chapters --sample ../plans/bakeoff/chapter-sample.json
+// tsx bin/digest-bakeoff.ts --score-existing
+// tsx bin/digest-bakeoff.ts --label round1 --buckets short,medium \
+// --candidates 'qwen2.5:7b@16384,qwen3:8b@16384,gemma2:9b@16384'
+// tsx bin/digest-bakeoff.ts --label round2 --buckets long,verylong \
+// --candidates 'qwen2.5:7b@16384' --modes absolute,chunk-local
+// tsx bin/digest-bakeoff.ts --label speakers-round1 --speakers off,on \
+// --sample ../plans/bakeoff/speaker-sample.json --candidates 'qwen2.5:7b@8192'
+
+import path from "node:path";
+import { mkdir, readdir, writeFile } from "node:fs/promises";
+import { readFile } from "node:fs/promises";
+import { open } from "../../common/bin/_lmdb";
+import { getPaths } from "../../common/lib/paths";
+import { parseFlags } from "../../common/bin/_parseFlags";
+import type { VideoStat } from "../../common/lib/stats";
+import { getDigestApp } from "../../common/lib/digestApps";
+import type { DigestAppConfig, DigestTimestampMode } from "../../common/lib/digest";
+import {
+ CHAPTER_SYSTEM_PROMPT,
+ DIGEST_MINUTES_PER_CHAPTER,
+ DIGEST_OVERLAP_CUES,
+ buildChapterPrompt,
+ chapterSchema,
+ maxCuesForContext,
+ toHms,
+} from "../../common/lib/digestPrompt";
+import { parseChapters, type DigestChunkOutput } from "../../common/lib/digestParse";
+import { chunkCuesForContext } from "../../common/lib/transcriptWindow";
+import { transcriptToMarkdown } from "../../common/lib/transcriptToMarkdown";
+import { readNormalizedTranscript } from "../../common/controller/normalizeTranscript";
+import {
+ CUES_JSON_FILENAME,
+ isRealAudioFile,
+ readUploaderChapters,
+ type UploaderChapter,
+} from "../../common/lib/videoStatus";
+import {
+ BOUNDARY_TOLERANCES_SECONDS,
+ isBoilerplateChapterTitle,
+ scoreBoundaries,
+ type BoundaryReport,
+} from "../../common/lib/boundaryScore";
+import { loadAttribution } from "../../common/lib/attribution-server";
+import type { AttributionRecord } from "../../common/lib/attribution";
+import {
+ ATTRIBUTION_MAX_SPEAKERS,
+ ATTRIBUTION_MIN_CLUSTER_SHARE,
+} from "../../common/lib/attributionPrompt";
+import { loadDigest } from "../../common/lib/digest-server";
+import type { Cue } from "../../common/lib/vtt";
+
+// ---------------------------------------------------------------------------
+// The sample
+// ---------------------------------------------------------------------------
+
+// Duration strata. Chosen to match the corpus shape recorded in FACTS.md rather
+// than to be round numbers: the >4 h bucket is 8.2% of videos but 46% of all
+// transcript tokens, so a sample that under-represents it measures the cheap
+// half of the sweep and misses the half where chunk-seam bugs live.
+const BUCKETS = ["short", "medium", "long", "verylong"] as const;
+type Bucket = (typeof BUCKETS)[number];
+
+const BUCKET_BOUNDS: Record<Bucket, { min: number; max: number; want: number }> = {
+ short: { min: 5 * 60, max: 30 * 60, want: 3 },
+ medium: { min: 45 * 60, max: 90 * 60, want: 2 },
+ long: { min: 3 * 3600, max: 4.5 * 3600, want: 2 },
+ verylong: { min: 6 * 3600, max: 14 * 3600, want: 1 },
+};
+
+type SampleVideo = {
+ slug: string;
+ channelSlug: string;
+ videoId: string;
+ videoDir: string;
+ title: string;
+ bucket: Bucket;
+ durationSeconds: number;
+ cueCount: number;
+ // Non-boilerplate uploader marks counted at pick time. Recorded for the
+ // record only — the scorer re-reads them from disk, because metadata can be
+ // refetched and a frozen copy would let the sample and the corpus disagree.
+ uploaderChapters?: number;
+};
+
+type Sample = {
+ version: 1;
+ pickedAt: string;
+ // Corpus-wide totals, captured at pick time. The sweep-days projection is
+ // computed from measured seconds-per-audio-hour times THIS number, so the
+ // projection and the sample come from one scan and can't drift apart.
+ corpus: {
+ videosScanned: number;
+ videosWithTranscript: number;
+ audioHours: number;
+ longTailVideos: number;
+ longTailAudioHours: number;
+ };
+ videos: SampleVideo[];
+};
+
+function bucketFor(seconds: number): Bucket | null {
+ for (const b of BUCKETS) {
+ const { min, max } = BUCKET_BOUNDS[b];
+ if (seconds >= min && seconds <= max) return b;
+ }
+ return null;
+}
+
+// Deterministic pick, so re-running --pick on an unchanged corpus reproduces the
+// same sample. No Math.random: a sample that moves between rounds is not a
+// sample, it is noise. Videos are ordered by a stable hash of the slug and the
+// first N per bucket are taken, spreading the pick across channels instead of
+// clustering on whichever channel sorts first.
+function stableHash(s: string): number {
+ let h = 2166136261;
+ for (let i = 0; i < s.length; i++) {
+ h ^= s.charCodeAt(i);
+ h = Math.imul(h, 16777619);
+ }
+ return h >>> 0;
+}
+
+async function pickSample(outPath: string): Promise<void> {
+ const paths = getPaths();
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ const statsByPath = root.openDB<
+ { metaMs: number; stat: VideoStat },
+ [string, string]
+ >({ name: "statsByPath", encoding: "msgpack" });
+
+ const byBucket = new Map<Bucket, SampleVideo[]>();
+ for (const b of BUCKETS) byBucket.set(b, []);
+
+ let videosScanned = 0;
+ let videosWithTranscript = 0;
+ let totalSeconds = 0;
+ let longTailVideos = 0;
+ let longTailSeconds = 0;
+
+ for (const { key, value } of statsByPath.getRange()) {
+ const stat = value.stat;
+ videosScanned++;
+ if (!stat.hasTranscript || !(stat.duration > 0)) continue;
+ videosWithTranscript++;
+ totalSeconds += stat.duration;
+ if (stat.duration > 4 * 3600) {
+ longTailVideos++;
+ longTailSeconds += stat.duration;
+ }
+ const bucket = bucketFor(stat.duration);
+ if (!bucket) continue;
+ // A digest needs cues; a transcript flagged present but empty is useless
+ // here and would silently shrink a stratum.
+ if (!stat.cueCount || stat.cueCount < 30) continue;
+ byBucket.get(bucket)!.push({
+ slug: stat.slug,
+ channelSlug: stat.channelSlug,
+ videoId: stat.id,
+ videoDir: (key as [string, string])[1],
+ title: stat.title,
+ bucket,
+ durationSeconds: Math.round(stat.duration),
+ cueCount: stat.cueCount,
+ });
+ }
+ await root.close();
+
+ const videos: SampleVideo[] = [];
+ for (const b of BUCKETS) {
+ const pool = byBucket.get(b)!;
+ pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug));
+ // One per channel first, so a stratum can't come entirely from one
+ // creator's house style — a model that happens to suit one show would
+ // otherwise look like a model that suits the corpus.
+ const seenChannels = new Set<string>();
+ const spread: SampleVideo[] = [];
+ for (const v of pool) {
+ if (seenChannels.has(v.channelSlug)) continue;
+ seenChannels.add(v.channelSlug);
+ spread.push(v);
+ }
+ const want = BUCKET_BOUNDS[b].want;
+ const taken = (spread.length >= want ? spread : pool).slice(0, want);
+ if (taken.length < want) {
+ console.warn(
+ `Warning: bucket ${b} wanted ${want} videos but only ${taken.length} qualify.`,
+ );
+ }
+ videos.push(...taken);
+ }
+
+ const sample: Sample = {
+ version: 1,
+ pickedAt: new Date().toISOString(),
+ corpus: {
+ videosScanned,
+ videosWithTranscript,
+ audioHours: Math.round(totalSeconds / 3600),
+ longTailVideos,
+ longTailAudioHours: Math.round(longTailSeconds / 3600),
+ },
+ videos,
+ };
+ await mkdir(path.dirname(outPath), { recursive: true });
+ await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`);
+ console.log(
+ `Scanned ${videosScanned} videos (${videosWithTranscript} with transcripts, ` +
+ `${sample.corpus.audioHours} audio-hours; ${longTailVideos} over 4 h holding ` +
+ `${sample.corpus.longTailAudioHours} h).`,
+ );
+ for (const v of videos) {
+ console.log(
+ ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${v.cueCount
+ .toString()
+ .padStart(5)} cues ${v.slug} ${v.title.slice(0, 60)}`,
+ );
+ }
+ console.log(`Wrote ${outPath}`);
+}
+
+// ---------------------------------------------------------------------------
+// Scoring
+// ---------------------------------------------------------------------------
+
+// Titles that carry no information about what was actually said. A cheap proxy
+// for title quality: a model that segments correctly but names every section
+// "Discussion" has produced a table of contents nobody can navigate.
+const GENERIC_TITLE_RE =
+ /^(the\s+)?(intro(duction)?|outro|conclusion|discussion|continued|continuation|overview|summary|recap|closing( remarks)?|opening( remarks)?|final thoughts|misc(ellaneous)?|other|general|topics?|segment|section|chapter|part)\b/i;
+const GENERIC_TITLE_TAIL_RE = /\b(part|section|segment|chapter)\s+(\d+|one|two|three|four|five|six|seven|eight|nine|ten)$/i;
+
+function isGenericTitle(title: string): boolean {
+ const t = title.trim();
+ return GENERIC_TITLE_RE.test(t) || GENERIC_TITLE_TAIL_RE.test(t);
+}
+
+function normalizeTitle(title: string): string {
+ return title.toLowerCase().replace(/[^\p{L}\p{N}]+/gu, " ").trim();
+}
+
+type Candidate = {
+ key: string;
+ model: string;
+ // Absent means "the engine's default" — the same thing an unset
+ // settings.digest.apps[id].numCtx means, so the default row measures exactly
+ // what a default-configured sweep would do.
+ numCtx?: number;
+ maxCues: number;
+ timestampMode: DigestTimestampMode;
+ think?: boolean;
+ // Render speaker labels into the transcript from attribution.json.
+ //
+ // FALSE MUST BE BYTE-IDENTICAL TO PRODUCTION. The speakers-off arm is not a
+ // control unless it renders exactly what the sweep renders, which is why the
+ // prefix hook is a no-op that returns null rather than a second renderer.
+ speakers: boolean;
+};
+
+type VideoScore = {
+ slug: string;
+ bucket: Bucket;
+ durationSeconds: number;
+ chunks: number;
+ chunksFailed: number;
+ zeroYieldChunks: number;
+ kept: number;
+ maxGapSeconds: number;
+ engineSeconds: number;
+ inputTokens: number;
+ outputTokens: number;
+ warningsByCode: Record<string, number>;
+ genericTitles: number;
+ duplicateTitles: number;
+ // Kept so a table can be sanity-checked against real output by hand, which is
+ // the only way to catch a model that scores well and reads badly.
+ sampleTitles: string[];
+ // Boundary accuracy against the uploader's own chapter marks. Null when the
+ // video carries none, which is the normal case for ~85% of the corpus.
+ boundary: BoundaryReport | null;
+ // How many named speakers the render actually carried, and how many kept
+ // titles mention one of them.
+ //
+ // NAME UPTAKE IS THE SPEAKER HYPOTHESIS' OWN METRIC. speakers-off cannot
+ // produce it by construction (the roster is never in the prompt), so a
+ // non-zero value on the off arm means a name leaked in some other way — a
+ // useful tripwire, not a score.
+ speakersRendered: number;
+ titlesWithSpeakerName: number;
+};
+
+type CandidateScore = {
+ candidate: Candidate;
+ videos: VideoScore[];
+ totals: {
+ videos: number;
+ audioHours: number;
+ chunks: number;
+ chunksFailed: number;
+ zeroYieldChunks: number;
+ zeroYieldRate: number;
+ kept: number;
+ chaptersPerHour: number;
+ maxGapSeconds: number;
+ meanGapSeconds: number;
+ genericTitleRate: number;
+ duplicateTitleRate: number;
+ warningsByCode: Record<string, number>;
+ rejectionRate: number;
+ engineSeconds: number;
+ tokensPerSecond: number;
+ secondsPerAudioHour: number;
+ projectedSweepDays: number;
+ // Boundary accuracy, pooled over the videos that had an oracle.
+ //
+ // POOLED, NOT AVERAGED PER VIDEO. A per-video mean lets a 4-chapter video
+ // weigh as much as a 90-chapter one, so a candidate could win by doing well
+ // on the shortest lists. Pooling counts matched/generated/reference across
+ // the whole sample and derives precision and recall from the totals.
+ boundary: {
+ videosScored: number;
+ referenceBoundaries: number;
+ medianOffsetSeconds: number | null;
+ withinThirtySecondsRate: number;
+ byTolerance: {
+ toleranceSeconds: number;
+ matched: number;
+ precision: number;
+ recall: number;
+ f1: number;
+ }[];
+ };
+ speakerNameTitleRate: number;
+ };
+};
+
+// ---------------------------------------------------------------------------
+// Speaker rendering (the speakers-on arm)
+// ---------------------------------------------------------------------------
+
+// A label per cue, or null where no NAMED speaker covers it.
+//
+// CONSUMES attribution.json, NEVER diarization.json. A bare cluster index
+// rendered as "Speaker 7:" is prompt cost with no semantic content, and it
+// invites chapter titles like "Speaker 7 responds". lib/diarization.ts already
+// argues a cluster index is not an identity; this honours that.
+type SpeakerContext = {
+ // Indexed by position in the FULL cue array, so a chunk can slice it.
+ labelByCueIndex: (string | null)[];
+ roster: string[];
+ // Cues that fell inside a named turn. Reported so a report can say how much
+ // of the transcript the labels actually reached.
+ labelledCues: number;
+};
+
+// Speakers worth rendering: the heaviest by attributed speech, capped and
+// floored exactly as attributionPrompt caps the naming call itself.
+//
+// The cap is doing real work here. Diarization over-splits (median 35 clusters
+// per video on this corpus, max 325), and 8 of the 9 attribution records that
+// existed when this was written were the REJECTED text-only pilot, one of them
+// carrying 346 "speakers" that were raw transcript fragments. Rendering that
+// unfiltered would bury the transcript in noise and measure the noise.
+function namedSpeakers(record: AttributionRecord): Map<number, string> {
+ const seconds = new Map<number, number>();
+ for (const seg of record.segments) {
+ const d = Math.max(0, seg.end - seg.start);
+ seconds.set(seg.speaker, (seconds.get(seg.speaker) ?? 0) + d);
+ }
+ const total = Array.from(seconds.values()).reduce((a, b) => a + b, 0);
+ if (total <= 0) return new Map();
+
+ const ranked = record.speakers
+ .map((s) => ({ index: s.index, label: s.label?.trim() ?? "", share: (seconds.get(s.index) ?? 0) / total }))
+ .filter((s) => s.label.length > 0 && s.share >= ATTRIBUTION_MIN_CLUSTER_SHARE)
+ .sort((a, b) => b.share - a.share)
+ .slice(0, ATTRIBUTION_MAX_SPEAKERS);
+
+ return new Map(ranked.map((s) => [s.index, s.label]));
+}
+
+// Map cues onto turns by MAXIMUM OVERLAP, not by containment.
+//
+// Cue boundaries do not align to speaker turns — measured on the diarized set,
+// only 64.4% of cues fall wholly inside a named turn. Containment would leave a
+// third of the transcript unlabelled for a reason that has nothing to do with
+// speaker identity; max-overlap assigns each cue to whoever does most of the
+// talking during it, and still yields null when no named turn touches it.
+function buildSpeakerContext(
+ cues: Cue[],
+ record: AttributionRecord,
+): SpeakerContext {
+ const named = namedSpeakers(record);
+ const segments = record.segments
+ .filter((s) => named.has(s.speaker))
+ .sort((a, b) => a.start - b.start);
+
+ const labelByCueIndex: (string | null)[] = new Array(cues.length).fill(null);
+ const rosterSeen = new Set<string>();
+ const roster: string[] = [];
+ let labelledCues = 0;
+
+ let cursor = 0;
+ for (let i = 0; i < cues.length; i++) {
+ const cue = cues[i];
+ const cueEnd = cue.end > cue.start ? cue.end : cue.start + 1;
+ // Segments are sorted, so the scan only ever moves forward.
+ while (cursor < segments.length && segments[cursor].end <= cue.start) cursor++;
+ let best: { label: string; overlap: number } | null = null;
+ for (let j = cursor; j < segments.length; j++) {
+ const seg = segments[j];
+ if (seg.start >= cueEnd) break;
+ const overlap = Math.min(cueEnd, seg.end) - Math.max(cue.start, seg.start);
+ if (overlap > 0 && (!best || overlap > best.overlap)) {
+ best = { label: named.get(seg.speaker)!, overlap };
+ }
+ }
+ if (best) {
+ labelByCueIndex[i] = best.label;
+ labelledCues++;
+ if (!rosterSeen.has(best.label)) {
+ rosterSeen.add(best.label);
+ roster.push(best.label);
+ }
+ }
+ }
+
+ return { labelByCueIndex, roster, labelledCues };
+}
+
+// Label on speaker CHANGE only, never per line.
+//
+// MEASURED COST. Per-line labels inflate a rendered chunk by ~12% of characters
+// at the median and up to ~30%, which on the ~10-tokens-per-cue budget that
+// sizes the 600-cue chunk to an 8k window is enough to start truncating — and
+// ollama truncates SILENTLY. Change-only lands at ~3.4% median. Since the
+// speaker changes on only ~26% of cues, the two carry the same information.
+//
+// A cue with no named speaker gets NO prefix and does not count as a change, so
+// a coverage gap reads as "the previous speaker continues" rather than as a
+// fake new person.
+function speakerPrefixer(
+ context: SpeakerContext,
+ chunkStartIndex: number,
+): (cue: Cue, index: number) => string | null {
+ let previous: string | null = null;
+ return (_cue, index) => {
+ const label = context.labelByCueIndex[chunkStartIndex + index] ?? null;
+ if (!label) return null;
+ if (label === previous) return null;
+ previous = label;
+ return `${label}: `;
+ };
+}
+
+function renderChunk(
+ meta: { id: string; title: string; channel?: string; duration?: number },
+ cues: Cue[],
+ offsetSeconds: number,
+ prefixForCue?: (cue: Cue, index: number) => string | null,
+): string {
+ return transcriptToMarkdown(
+ { ...meta, cues },
+ {
+ timestamps: true,
+ includeDescription: false,
+ includeTags: false,
+ stampForCue: (_clock, seconds) =>
+ toHms(Math.max(0, seconds - offsetSeconds)),
+ // Omitted entirely on the speakers-off arm, so that arm's bytes are the
+ // sweep's bytes.
+ ...(prefixForCue ? { prefixForCue } : {}),
+ },
+ );
+}
+
+// The cast list for ONE chunk, not for the video.
+//
+// A 12-name roster on a chunk where two people speak is misleading and wastes
+// context. The preamble also has to explain the change-only convention, or the
+// model reads an unlabelled line as an unknown speaker.
+function speakerPreamble(names: string[]): string | undefined {
+ if (names.length === 0) return undefined;
+ return [
+ "Speakers in this section (a name before a line means that speaker begins",
+ "there; unlabelled lines continue the previous speaker):",
+ ...names.map((n) => `- ${n}`),
+ ].join("\n");
+}
+
+// Where a sample video's sidecars live. One definition, because three call
+// sites now need it (scoring, the oracle read, attribution).
+function videoDirFor(video: SampleVideo): string {
+ const paths = getPaths();
+ return path.join(paths.channelsDir, video.channelSlug, "data", video.videoDir);
+}
+
+// The oracle, read FRESH at score time rather than frozen into the sample.
+//
+// A video's uploader chapters can change when metadata is refetched. Freezing
+// them would let the sample and the disk disagree silently; reading them here
+// means a report always scored against what the uploader currently says.
+//
+// Boilerplate marks are dropped. "Intro"/"Sponsor"/"Outro" are boundaries in the
+// video's FURNITURE, not in its subject, and crediting a model for finding the
+// sponsor read measures the wrong thing — the same class of junk the
+// `boilerplate` context field exists to remove.
+async function uploaderBoundaries(
+ video: SampleVideo,
+): Promise<{ starts: number[]; chapters: UploaderChapter[] } | null> {
+ const chapters = await readUploaderChapters(videoDirFor(video));
+ if (!chapters) return null;
+ const kept = chapters.filter((c) => !isBoilerplateChapterTitle(c.title));
+ if (kept.length === 0) return null;
+ return { starts: kept.map((c) => c.start), chapters: kept };
+}
+
+async function scoreVideo(
+ video: SampleVideo,
+ candidate: Candidate,
+ log: (m: string) => void,
+): Promise<VideoScore | null> {
+ const videoDir = videoDirFor(video);
+ const cuesPath = path.join(videoDir, CUES_JSON_FILENAME);
+ const transcript = await readNormalizedTranscript(cuesPath);
+ if (!transcript || !transcript.cues?.length) {
+ log(` ${video.slug}: no transcript on disk, skipped`);
+ return null;
+ }
+ const cues = transcript.cues;
+ const chunks = chunkCuesForContext(cues, {
+ maxCues: candidate.maxCues,
+ overlapCues: DIGEST_OVERLAP_CUES,
+ });
+
+ // Speakers are loaded per VIDEO, once, even though they are rendered per
+ // chunk: the roster has to be sliced to the chunk but the cue->turn mapping
+ // is a whole-video computation.
+ let speakerContext: SpeakerContext | null = null;
+ if (candidate.speakers) {
+ const record = await loadAttribution(videoDir);
+ if (record) {
+ speakerContext = buildSpeakerContext(cues, record);
+ log(
+ ` ${video.slug}: ${speakerContext.roster.length} named speaker(s), ` +
+ `${pct(cues.length > 0 ? speakerContext.labelledCues / cues.length : 0)} of cues labelled`,
+ );
+ } else {
+ // NOT an error and NOT a skip. A video with no attribution renders
+ // exactly as the off arm renders it, which is the behaviour any shipped
+ // version would need for the ~98% of the corpus that can never be
+ // diarized. Silently degrading is the feature.
+ log(` ${video.slug}: no attribution on disk, rendering without speakers`);
+ }
+ }
+
+ // Chunk i starts at this index in the full cue array. chunkCuesForContext
+ // overlaps by DIGEST_OVERLAP_CUES, so this is not i * maxCues.
+ const chunkStartIndices: number[] = [];
+ {
+ let cursor = 0;
+ for (const chunk of chunks) {
+ const first = chunk[0];
+ // Cues are unique by identity here, so indexOf from the last position is
+ // both correct and linear overall.
+ const found = cues.indexOf(first, Math.max(0, cursor - chunk.length));
+ chunkStartIndices.push(found >= 0 ? found : cursor);
+ cursor = (found >= 0 ? found : cursor) + chunk.length;
+ }
+ }
+
+ const app = getDigestApp("ollama-direct");
+ const config: DigestAppConfig = {
+ model: candidate.model,
+ numCtx: candidate.numCtx,
+ temperature: 0,
+ timeoutMs: 20 * 60_000,
+ ...(candidate.think !== undefined ? { think: candidate.think } : {}),
+ };
+
+ const outputs: DigestChunkOutput[] = [];
+ const score: VideoScore = {
+ slug: video.slug,
+ bucket: video.bucket,
+ durationSeconds: video.durationSeconds,
+ chunks: chunks.length,
+ chunksFailed: 0,
+ zeroYieldChunks: 0,
+ kept: 0,
+ maxGapSeconds: 0,
+ engineSeconds: 0,
+ inputTokens: 0,
+ outputTokens: 0,
+ warningsByCode: {},
+ genericTitles: 0,
+ duplicateTitles: 0,
+ sampleTitles: [],
+ boundary: null,
+ speakersRendered: speakerContext?.roster.length ?? 0,
+ titlesWithSpeakerName: 0,
+ };
+
+ for (let i = 0; i < chunks.length; i++) {
+ const chunk = chunks[i];
+ const startSeconds = Math.max(0, Math.floor(chunk[0].start));
+ const endSeconds = Math.max(
+ startSeconds,
+ Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start),
+ );
+ const offset = candidate.timestampMode === "chunk-local" ? startSeconds : 0;
+
+ // The roster is per-CHUNK: the names that actually appear in THIS slice.
+ // A 12-name cast list over a chunk where two people speak is misleading and
+ // spends context for nothing.
+ let prefixForCue: ((cue: Cue, index: number) => string | null) | undefined;
+ let speakerRoster: string | undefined;
+ if (speakerContext) {
+ const startIndex = chunkStartIndices[i];
+ const namesHere: string[] = [];
+ const seenHere = new Set<string>();
+ for (let k = 0; k < chunk.length; k++) {
+ const label = speakerContext.labelByCueIndex[startIndex + k];
+ if (label && !seenHere.has(label)) {
+ seenHere.add(label);
+ namesHere.push(label);
+ }
+ }
+ if (namesHere.length > 0) {
+ prefixForCue = speakerPrefixer(speakerContext, startIndex);
+ speakerRoster = speakerPreamble(namesHere);
+ }
+ }
+
+ const promptInput = {
+ title: transcript.title || video.videoId,
+ channel: transcript.channel || video.channelSlug,
+ startSeconds,
+ endSeconds,
+ transcript: renderChunk(
+ {
+ id: transcript.id,
+ title: transcript.title,
+ channel: transcript.channel,
+ duration: transcript.duration,
+ },
+ chunk,
+ offset,
+ prefixForCue,
+ ),
+ timestampMode: candidate.timestampMode,
+ ...(speakerRoster ? { speakerRoster } : {}),
+ };
+ try {
+ const result = await app.run({
+ system: CHAPTER_SYSTEM_PROMPT,
+ prompt: buildChapterPrompt(promptInput),
+ schema: chapterSchema(endSeconds - startSeconds),
+ config,
+ });
+ score.engineSeconds += result.durationMs / 1000;
+ score.inputTokens += result.inputTokens ?? 0;
+ score.outputTokens += result.outputTokens ?? 0;
+ outputs.push({
+ index: i,
+ startSeconds,
+ endSeconds,
+ data: result.data,
+ timestampMode: candidate.timestampMode,
+ });
+ } catch (err) {
+ score.chunksFailed++;
+ score.warningsByCode["chunk-failed"] =
+ (score.warningsByCode["chunk-failed"] ?? 0) + 1;
+ log(` ${video.slug} chunk ${i + 1}/${chunks.length} failed: ${(err as Error).message}`);
+ }
+ }
+
+ // ZERO-YIELD CHUNKS — the headline defect metric. Computed by parsing each
+ // chunk ALONE, because the merged parse cannot attribute a kept chapter back
+ // to the chunk that produced it, and "chunk 3 produced nothing" is precisely
+ // the failure this whole stage exists to fix. A chunk the engine never
+ // answered counts too: from the corpus's point of view the outcome is the
+ // same, an interval of the video with no chapters in it.
+ for (const out of outputs) {
+ if (parseChapters([out], cues).chapters.length === 0) score.zeroYieldChunks++;
+ }
+ score.zeroYieldChunks += score.chunksFailed;
+
+ const parsed = parseChapters(outputs, cues);
+ score.kept = parsed.chapters.length;
+ for (const w of parsed.warnings) {
+ score.warningsByCode[w.code] = (score.warningsByCode[w.code] ?? 0) + 1;
+ }
+
+ // MAX COVERAGE GAP — catches "summarised the tail, skipped the head". The
+ // leading gap (0 -> first chapter) and the trailing one (last chapter -> end)
+ // are included deliberately: a video whose chapters all sit in the last 20
+ // minutes has a coverage failure that consecutive-gap-only scoring hides.
+ const starts = parsed.chapters.map((c) => c.start);
+ const bounds = [0, ...starts, video.durationSeconds];
+ for (let i = 1; i < bounds.length; i++) {
+ score.maxGapSeconds = Math.max(score.maxGapSeconds, bounds[i] - bounds[i - 1]);
+ }
+
+ const seen = new Set<string>();
+ for (const c of parsed.chapters) {
+ if (isGenericTitle(c.title)) score.genericTitles++;
+ const n = normalizeTitle(c.title);
+ if (seen.has(n)) score.duplicateTitles++;
+ seen.add(n);
+ }
+ score.sampleTitles = parsed.chapters.slice(0, 8).map((c) => `${c.clock} ${c.title}`);
+
+ // BOUNDARY ACCURACY against the uploader. The one metric here that measures
+ // whether the segmentation is RIGHT rather than well-formed.
+ const oracle = await uploaderBoundaries(video);
+ if (oracle) {
+ score.boundary = scoreBoundaries(
+ oracle.starts,
+ parsed.chapters.map((c) => c.start),
+ );
+ }
+
+ // SPEAKER NAME UPTAKE. Counted against the roster the render actually used,
+ // so the off arm can be checked for leakage: a non-zero value there means a
+ // name reached the titles by some route other than the labels.
+ if (speakerContext && speakerContext.roster.length > 0) {
+ // WORD-BOUNDARY MATCHING, not substring. normalizeTitle collapses to
+ // space-separated words, so a substring test would count "ghost stories" as
+ // containing the speaker "Host" — and role labels like "Host" and "Caller"
+ // are exactly what the diarized lane produces when the transcript does not
+ // support a real name, so the false positives would not be rare.
+ const needles = speakerContext.roster
+ .map((n) => normalizeTitle(n))
+ .filter((n) => n.length >= 3)
+ .map((n) => ` ${n} `);
+ for (const c of parsed.chapters) {
+ const t = ` ${normalizeTitle(c.title)} `;
+ if (needles.some((n) => t.includes(n))) score.titlesWithSpeakerName++;
+ }
+ }
+
+ log(
+ ` ${video.slug} [${video.bucket}] ${chunks.length} chunk(s) → ${score.kept} chapter(s), ` +
+ `${score.zeroYieldChunks} zero-yield, ${Math.round(score.engineSeconds)}s engine` +
+ (score.boundary
+ ? `, boundary F1@30 ${pct(score.boundary.scores[0]?.f1 ?? 0)} ` +
+ `(${score.boundary.referenceCount} uploader mark(s))`
+ : ", no uploader chapters"),
+ );
+ return score;
+}
+
+function aggregate(candidate: Candidate, videos: VideoScore[]): CandidateScore {
+ const sum = (f: (v: VideoScore) => number): number =>
+ videos.reduce((a, v) => a + f(v), 0);
+ const audioHours = sum((v) => v.durationSeconds) / 3600;
+ const chunks = sum((v) => v.chunks);
+ const kept = sum((v) => v.kept);
+ const engineSeconds = sum((v) => v.engineSeconds);
+ const tokens = sum((v) => v.inputTokens + v.outputTokens);
+
+ const warningsByCode: Record<string, number> = {};
+ for (const v of videos) {
+ for (const [code, n] of Object.entries(v.warningsByCode)) {
+ warningsByCode[code] = (warningsByCode[code] ?? 0) + n;
+ }
+ }
+ const rejections = Object.entries(warningsByCode)
+ .filter(([code]) => code !== "seam-duplicate")
+ .reduce((a, [, n]) => a + n, 0);
+
+ const secondsPerAudioHour = audioHours > 0 ? engineSeconds / audioHours : 0;
+
+ // Pooled boundary accuracy. See the comment on CandidateScore.totals.boundary
+ // for why this is not a mean of per-video F1s.
+ const scored = videos.filter((v) => v.boundary);
+ const referenceBoundaries = scored.reduce((a, v) => a + v.boundary!.referenceCount, 0);
+ const generatedBoundaries = scored.reduce((a, v) => a + v.boundary!.generatedCount, 0);
+ const within30 = scored.reduce((a, v) => a + v.boundary!.withinThirtySeconds, 0);
+ const allOffsets: number[] = [];
+ for (const v of scored) {
+ // medianOffsetSeconds is per video; pooling the medians is not a median, so
+ // the report quotes the median OF the per-video medians and says so.
+ if (v.boundary!.medianOffsetSeconds !== null) {
+ allOffsets.push(v.boundary!.medianOffsetSeconds);
+ }
+ }
+ allOffsets.sort((a, b) => a - b);
+
+ const byTolerance = BOUNDARY_TOLERANCES_SECONDS.map((tol) => {
+ const matched = scored.reduce(
+ (a, v) => a + (v.boundary!.scores.find((s) => s.toleranceSeconds === tol)?.matched ?? 0),
+ 0,
+ );
+ const precision = generatedBoundaries > 0 ? matched / generatedBoundaries : 0;
+ const recall = referenceBoundaries > 0 ? matched / referenceBoundaries : 0;
+ return {
+ toleranceSeconds: tol,
+ matched,
+ precision: round(precision, 4),
+ recall: round(recall, 4),
+ f1: round(precision + recall > 0 ? (2 * precision * recall) / (precision + recall) : 0, 4),
+ };
+ });
+
+ return {
+ candidate,
+ videos,
+ totals: {
+ videos: videos.length,
+ audioHours: round(audioHours, 2),
+ chunks,
+ chunksFailed: sum((v) => v.chunksFailed),
+ zeroYieldChunks: sum((v) => v.zeroYieldChunks),
+ zeroYieldRate: chunks > 0 ? round(sum((v) => v.zeroYieldChunks) / chunks, 4) : 0,
+ kept,
+ chaptersPerHour: audioHours > 0 ? round(kept / audioHours, 2) : 0,
+ maxGapSeconds: videos.reduce((a, v) => Math.max(a, v.maxGapSeconds), 0),
+ meanGapSeconds:
+ videos.length > 0 ? Math.round(sum((v) => v.maxGapSeconds) / videos.length) : 0,
+ genericTitleRate: kept > 0 ? round(sum((v) => v.genericTitles) / kept, 4) : 0,
+ duplicateTitleRate: kept > 0 ? round(sum((v) => v.duplicateTitles) / kept, 4) : 0,
+ warningsByCode,
+ // Rejections per kept chapter — the ratio that says how much of what the
+ // model produced the guards had to throw away.
+ rejectionRate: kept + rejections > 0 ? round(rejections / (kept + rejections), 4) : 0,
+ engineSeconds: Math.round(engineSeconds),
+ tokensPerSecond: engineSeconds > 0 ? round(tokens / engineSeconds, 1) : 0,
+ secondsPerAudioHour: Math.round(secondsPerAudioHour),
+ projectedSweepDays: 0, // filled in once corpus hours are known
+ boundary: {
+ videosScored: scored.length,
+ referenceBoundaries,
+ medianOffsetSeconds:
+ allOffsets.length > 0 ? allOffsets[Math.floor(allOffsets.length / 2)] : null,
+ withinThirtySecondsRate:
+ referenceBoundaries > 0 ? round(within30 / referenceBoundaries, 4) : 0,
+ byTolerance,
+ },
+ speakerNameTitleRate: kept > 0 ? round(sum((v) => v.titlesWithSpeakerName) / kept, 4) : 0,
+ },
+ };
+}
+
+function round(n: number, places: number): number {
+ const f = 10 ** places;
+ return Math.round(n * f) / f;
+}
+
+// ---------------------------------------------------------------------------
+// Report
+// ---------------------------------------------------------------------------
+
+function markdownReport(
+ label: string,
+ sample: Sample,
+ buckets: Bucket[],
+ scores: CandidateScore[],
+): string {
+ const lines: string[] = [];
+ lines.push(`# Digest bake-off — ${label}`);
+ lines.push("");
+ lines.push(
+ `Sample: ${scores[0]?.totals.videos ?? 0} video(s) from \`plans/bakeoff/sample.json\`` +
+ ` (buckets: ${buckets.join(", ")}), ${scores[0]?.totals.audioHours ?? 0} audio-hours.`,
+ );
+ lines.push(
+ `Sweep days are projected as measured seconds-per-audio-hour x ` +
+ `${sample.corpus.audioHours} corpus audio-hours, one lane, no parallelism.`,
+ );
+ lines.push("");
+ lines.push(
+ "| Candidate | Zero-yield chunks | Chapters/h | Rejection rate | Max gap | Generic | Dup | tok/s | s per audio-h | **Sweep days** |",
+ );
+ lines.push(
+ "| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
+ );
+ for (const s of scores) {
+ const t = s.totals;
+ lines.push(
+ `| \`${s.candidate.key}\` | ${t.zeroYieldChunks}/${t.chunks} (${pct(t.zeroYieldRate)}) | ` +
+ `${t.chaptersPerHour} | ${pct(t.rejectionRate)} | ${toHms(t.maxGapSeconds)} | ` +
+ `${pct(t.genericTitleRate)} | ${pct(t.duplicateTitleRate)} | ${t.tokensPerSecond} | ` +
+ `${t.secondsPerAudioHour} | **${t.projectedSweepDays}** |`,
+ );
+ }
+ lines.push("");
+ lines.push("## Boundary accuracy vs the uploader's own chapters");
+ lines.push("");
+ lines.push(
+ "The oracle is `metadata.info.json.chapters` — marks a human authored while",
+ "watching, who never saw our prompt. Boilerplate marks (\"Intro\", \"Sponsor\")",
+ "and the boundary at 00:00 are dropped before scoring: neither carries",
+ "segmentation information, and the origin would be a free hit for every",
+ "candidate. Matching is one-to-one and closest-pair-first, so a cluster of",
+ "boundaries around one uploader mark scores one match, not many.",
+ );
+ lines.push("");
+ lines.push(
+ "| Candidate | Videos scored | Uploader marks | Median offset | Within 30s | P@30 | R@30 | **F1@30** | F1@60 | Name-in-title |",
+ );
+ lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
+ for (const s of scores) {
+ const b = s.totals.boundary;
+ const t30 = b.byTolerance.find((x) => x.toleranceSeconds === 30);
+ const t60 = b.byTolerance.find((x) => x.toleranceSeconds === 60);
+ lines.push(
+ `| \`${s.candidate.key}\` | ${b.videosScored} | ${b.referenceBoundaries} | ` +
+ `${b.medianOffsetSeconds === null ? "—" : `${b.medianOffsetSeconds}s`} | ` +
+ `${pct(b.withinThirtySecondsRate)} | ${pct(t30?.precision ?? 0)} | ${pct(t30?.recall ?? 0)} | ` +
+ `**${pct(t30?.f1 ?? 0)}** | ${pct(t60?.f1 ?? 0)} | ${pct(s.totals.speakerNameTitleRate)} |`,
+ );
+ }
+ lines.push("");
+ lines.push("## Rejections by guard");
+ lines.push("");
+ const codes = Array.from(
+ new Set(scores.flatMap((s) => Object.keys(s.totals.warningsByCode))),
+ ).sort();
+ lines.push(`| Candidate | ${codes.join(" | ")} |`);
+ lines.push(`| --- | ${codes.map(() => "---").join(" | ")} |`);
+ for (const s of scores) {
+ lines.push(
+ `| \`${s.candidate.key}\` | ${codes
+ .map((c) => s.totals.warningsByCode[c] ?? 0)
+ .join(" | ")} |`,
+ );
+ }
+ lines.push("");
+ lines.push("## Per-video");
+ lines.push("");
+ lines.push(
+ "| Candidate | Video | Bucket | Chunks | Zero-yield | Chapters | Max gap | Engine s | Marks | F1@30 | Speakers |",
+ );
+ lines.push("| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |");
+ for (const s of scores) {
+ for (const v of s.videos) {
+ const f1 = v.boundary?.scores.find((x) => x.toleranceSeconds === 30)?.f1;
+ lines.push(
+ `| \`${s.candidate.key}\` | \`${v.slug}\` | ${v.bucket} | ${v.chunks} | ` +
+ `${v.zeroYieldChunks} | ${v.kept} | ${toHms(v.maxGapSeconds)} | ${Math.round(v.engineSeconds)} | ` +
+ `${v.boundary?.referenceCount ?? "—"} | ${f1 === undefined ? "—" : pct(f1)} | ` +
+ `${v.speakersRendered || "—"} |`,
+ );
+ }
+ }
+ lines.push("");
+ lines.push("## Sample output (first chapters per video)");
+ lines.push("");
+ for (const s of scores) {
+ lines.push(`### \`${s.candidate.key}\``);
+ lines.push("");
+ for (const v of s.videos) {
+ lines.push(`**${v.slug}** (${v.bucket}, ${toHms(v.durationSeconds)})`);
+ lines.push("");
+ for (const t of v.sampleTitles) lines.push(`- ${t}`);
+ if (v.sampleTitles.length === 0) lines.push("- _(nothing survived the guards)_");
+ lines.push("");
+ }
+ }
+ return `${lines.join("\n")}\n`;
+}
+
+function pct(v: number): string {
+ return `${Math.round(v * 1000) / 10}%`;
+}
+
+// ---------------------------------------------------------------------------
+// Main
+// ---------------------------------------------------------------------------
+
+// "model@ctx" or "model@ctx:think" / "model@ctx:nothink".
+//
+// maxCues is DERIVED from the context rather than configured: the 1200-cue
+// default is sized for a 16k window (~10 tokens/cue -> ~12k tokens of transcript
+// plus room for prompt and response), so halving the context must halve the
+// slice or every call silently truncates — the exact failure that made the first
+// smoke test summarize a fragment.
+function parseCandidate(
+ spec: string,
+ mode: DigestTimestampMode,
+ speakers: boolean,
+ maxCuesOverride: number | null,
+): Candidate {
+ const [modelPart, rest] = spec.split("@");
+ const [ctxPart, thinkPart] = (rest ?? "").split(":");
+ const numCtx = ctxPart ? Number(ctxPart) : undefined;
+ // The SAME derivation production uses, not a parallel copy — otherwise the
+ // bake-off scores a chunk size the sweep would never actually run.
+ //
+ // --max-cues breaks that tie deliberately, for one reason: a paired A/B needs
+ // HEADROOM. Speaker labels add ~10% to the prompt, and on a pool of long VODs
+ // the un-labelled chunk is already near the window — so at the derived size
+ // the speakers-on arm would truncate where speakers-off did not, and the
+ // measurement would be of truncation. Raising numCtx while pinning maxCues
+ // gives both arms the same cue count with room to spare. BOTH ARMS ALWAYS GET
+ // THE SAME VALUE; the report records it.
+ const maxCues = maxCuesOverride ?? maxCuesForContext(numCtx);
+ return {
+ // The speaker axis and a cue override only show in the key when set, so
+ // existing round labels keep reading the way rounds 1-2 wrote them.
+ key:
+ `${modelPart}@${numCtx ?? "default"}/${mode}` +
+ `${maxCuesOverride ? `/${maxCuesOverride}cues` : ""}${speakers ? "/speakers" : ""}`,
+ model: modelPart,
+ ...(numCtx ? { numCtx } : {}),
+ maxCues,
+ timestampMode: mode,
+ speakers,
+ ...(thinkPart === "think"
+ ? { think: true }
+ : thinkPart === "nothink"
+ ? { think: false }
+ : {}),
+ };
+}
+
+// ---------------------------------------------------------------------------
+// The chapter-oracle sample
+// ---------------------------------------------------------------------------
+
+// Pick a sample restricted to videos that can actually be SCORED — i.e. that
+// carry >=`minChapters` non-boilerplate uploader marks.
+//
+// WHY IT DOES NOT READ EVERY METADATA FILE. metadata.info.json runs to ~100 KB
+// and only ~19% of videos carry chapters, so parsing all 77k to find them would
+// read several GB to answer a question a stable-hash walk answers in a few
+// hundred reads: candidates are visited in the same deterministic order
+// pickSample uses, and the walk stops as soon as every stratum is full.
+//
+// KEPT IN A SEPARATE FILE from sample.json, deliberately. Rounds 1 and 2 are
+// scored against that sample; repointing it would silently invalidate the
+// comparison this harness exists to protect.
+async function pickChapterSample(
+ outPath: string,
+ opts: { minChapters: number; perBucket: number | null; requireAudio: boolean },
+): Promise<void> {
+ const paths = getPaths();
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ const statsByPath = root.openDB<
+ { metaMs: number; stat: VideoStat },
+ [string, string]
+ >({ name: "statsByPath", encoding: "msgpack" });
+
+ const byBucket = new Map<Bucket, SampleVideo[]>();
+ for (const b of BUCKETS) byBucket.set(b, []);
+
+ let videosScanned = 0;
+ let videosWithTranscript = 0;
+ let totalSeconds = 0;
+ let longTailVideos = 0;
+ let longTailSeconds = 0;
+
+ for (const { key, value } of statsByPath.getRange()) {
+ const stat = value.stat;
+ videosScanned++;
+ if (!stat.hasTranscript || !(stat.duration > 0)) continue;
+ videosWithTranscript++;
+ totalSeconds += stat.duration;
+ if (stat.duration > 4 * 3600) {
+ longTailVideos++;
+ longTailSeconds += stat.duration;
+ }
+ const bucket = bucketFor(stat.duration);
+ if (!bucket) continue;
+ if (!stat.cueCount || stat.cueCount < 30) continue;
+ byBucket.get(bucket)!.push({
+ slug: stat.slug,
+ channelSlug: stat.channelSlug,
+ videoId: stat.id,
+ videoDir: (key as [string, string])[1],
+ title: stat.title,
+ bucket,
+ durationSeconds: Math.round(stat.duration),
+ cueCount: stat.cueCount,
+ });
+ }
+ await root.close();
+
+ const videos: SampleVideo[] = [];
+ for (const b of BUCKETS) {
+ const pool = byBucket.get(b)!;
+ pool.sort((a, c) => stableHash(a.slug) - stableHash(c.slug));
+ const want = opts.perBucket ?? BUCKET_BOUNDS[b].want;
+ const seenChannels = new Set<string>();
+ const taken: SampleVideo[] = [];
+ let probed = 0;
+ for (const v of pool) {
+ if (taken.length >= want) break;
+ // One per channel first, for the same reason pickSample does it: a
+ // stratum drawn from one creator measures a house style.
+ if (seenChannels.has(v.channelSlug)) continue;
+ probed++;
+ const oracle = await uploaderBoundaries(v);
+ if (!oracle || oracle.starts.length < opts.minChapters) continue;
+ if (opts.requireAudio) {
+ // isRealAudioFile, not an extension test of my own: it is the same
+ // predicate the cleanup and diarization lanes use, so "has audio" means
+ // here exactly what it means to the job that would diarize it.
+ const files = await readdir(videoDirFor(v)).catch(() => [] as string[]);
+ if (!files.some((f) => isRealAudioFile(f))) continue;
+ }
+ seenChannels.add(v.channelSlug);
+ taken.push({ ...v, uploaderChapters: oracle.starts.length });
+ }
+ if (taken.length < want) {
+ console.warn(
+ `Warning: bucket ${b} wanted ${want} scorable videos but only ${taken.length} qualify ` +
+ `(probed ${probed} candidates).`,
+ );
+ }
+ videos.push(...taken);
+ }
+
+ const sample: Sample = {
+ version: 1,
+ pickedAt: new Date().toISOString(),
+ corpus: {
+ videosScanned,
+ videosWithTranscript,
+ audioHours: Math.round(totalSeconds / 3600),
+ longTailVideos,
+ longTailAudioHours: Math.round(longTailSeconds / 3600),
+ },
+ videos,
+ };
+ await mkdir(path.dirname(outPath), { recursive: true });
+ await writeFile(outPath, `${JSON.stringify(sample, null, 2)}\n`);
+ for (const v of videos) {
+ console.log(
+ ` ${v.bucket.padEnd(8)} ${toHms(v.durationSeconds)} ${String(v.uploaderChapters).padStart(3)} marks ` +
+ `${v.slug} ${v.title.slice(0, 55)}`,
+ );
+ }
+ console.log(`Wrote ${outPath} (${videos.length} scorable video(s))`);
+}
+
+// ---------------------------------------------------------------------------
+// Free scorer smoke test
+// ---------------------------------------------------------------------------
+
+// Score the digests ALREADY on disk against their uploader chapters.
+//
+// This is the cheapest honest check available: no model call, no write, and it
+// answers "does the metric move, and is the plumbing right" before a generation
+// run is spent finding out. If this prints nonsense — every F1 at 0 or 1 — the
+// scorer is wrong and no A/B built on it would mean anything.
+async function scoreExisting(minChapters: number, limit: number): Promise<void> {
+ const paths = getPaths();
+ const root = open({ path: paths.lmdbPath, maxDbs: 12, compression: true });
+ const statsByPath = root.openDB<
+ { metaMs: number; stat: VideoStat },
+ [string, string]
+ >({ name: "statsByPath", encoding: "msgpack" });
+
+ const candidates: SampleVideo[] = [];
+ for (const { key, value } of statsByPath.getRange()) {
+ const stat = value.stat;
+ if (!stat.hasTranscript || !(stat.duration > 0)) continue;
+ candidates.push({
+ slug: stat.slug,
+ channelSlug: stat.channelSlug,
+ videoId: stat.id,
+ videoDir: (key as [string, string])[1],
+ title: stat.title,
+ bucket: bucketFor(stat.duration) ?? "short",
+ durationSeconds: Math.round(stat.duration),
+ cueCount: stat.cueCount ?? 0,
+ });
+ }
+ await root.close();
+
+ const rows: {
+ slug: string;
+ marks: number;
+ generated: number;
+ medianOffset: number | null;
+ precision30: number;
+ recall30: number;
+ recall60: number;
+ f1at30: number;
+ f1at60: number;
+ }[] = [];
+
+ for (const v of candidates) {
+ if (rows.length >= limit) break;
+ const dir = videoDirFor(v);
+ const digest = await loadDigest(dir);
+ const items = digest?.sections?.chapters?.items ?? [];
+ if (items.length === 0) continue;
+ const oracle = await uploaderBoundaries(v);
+ if (!oracle || oracle.starts.length < minChapters) continue;
+ const report = scoreBoundaries(oracle.starts, items.map((c) => c.start));
+ if (report.referenceCount === 0) continue;
+ const s30 = report.scores.find((s) => s.toleranceSeconds === 30);
+ const s60 = report.scores.find((s) => s.toleranceSeconds === 60);
+ rows.push({
+ slug: v.slug,
+ marks: report.referenceCount,
+ generated: report.generatedCount,
+ medianOffset: report.medianOffsetSeconds,
+ precision30: s30?.precision ?? 0,
+ recall30: s30?.recall ?? 0,
+ recall60: s60?.recall ?? 0,
+ f1at30: s30?.f1 ?? 0,
+ f1at60: s60?.f1 ?? 0,
+ });
+ }
+
+ if (rows.length === 0) {
+ console.log(
+ `No video on disk has both a digest and >=${minChapters} non-boilerplate uploader chapters.`,
+ );
+ return;
+ }
+
+ console.log(
+ `Scored ${rows.length} existing digest(s) against uploader chapters (no model calls, no writes).\n`,
+ );
+ console.log(
+ " slug marks gen medOff P@30 R@30 R@60 F1@30 F1@60",
+ );
+ for (const r of rows) {
+ console.log(
+ ` ${r.slug.padEnd(36).slice(0, 36)} ${String(r.marks).padStart(5)} ` +
+ `${String(r.generated).padStart(3)} ${String(r.medianOffset ?? "—").padStart(6)} ` +
+ `${pct(r.precision30).padStart(5)} ${pct(r.recall30).padStart(5)} ` +
+ `${pct(r.recall60).padStart(5)} ${pct(r.f1at30).padStart(5)} ${pct(r.f1at60).padStart(5)}`,
+ );
+ }
+ const sumOf = (f: (r: (typeof rows)[number]) => number): number =>
+ rows.reduce((a, r) => a + f(r), 0);
+ // POOLED, not a mean of per-video rates: a 3-mark video must not weigh the
+ // same as a 29-mark one.
+ const marks = sumOf((r) => r.marks);
+ const generated = sumOf((r) => r.generated);
+ const matched30 = sumOf((r) => r.recall30 * r.marks);
+ const matched60 = sumOf((r) => r.recall60 * r.marks);
+ const p30 = generated > 0 ? matched30 / generated : 0;
+ const r30 = marks > 0 ? matched30 / marks : 0;
+ console.log(
+ `\n POOLED: ${marks} uploader mark(s), ${generated} generated. ` +
+ `P@30 ${pct(p30)}, R@30 ${pct(r30)}, R@60 ${pct(marks > 0 ? matched60 / marks : 0)}, ` +
+ `F1@30 ${pct(p30 + r30 > 0 ? (2 * p30 * r30) / (p30 + r30) : 0)}`,
+ );
+ console.log(
+ ` NOTE: the digest targets one chapter per ${DIGEST_MINUTES_PER_CHAPTER} minutes and so is\n` +
+ ` deliberately DENSER than uploader chaptering (${generated} vs ${marks} here). Precision is\n` +
+ ` therefore partly a density artifact; recall is the half that answers "did it find the\n` +
+ ` human's boundaries". Compare candidates on recall AND on chapters/h together.`,
+ );
+}
+
+async function main(): Promise<void> {
+ const flags = parseFlags(process.argv.slice(2));
+ const outDir = flags.outDir ?? path.join(process.cwd(), "..", "plans", "bakeoff");
+ const samplePath = flags.sample ?? path.join(outDir, "sample.json");
+ const minChapters = flags["min-chapters"] ? Number(flags["min-chapters"]) : 4;
+
+ if (flags.pick === "true") {
+ await pickSample(samplePath);
+ return;
+ }
+
+ if (flags["pick-chapters"] === "true") {
+ await pickChapterSample(samplePath, {
+ minChapters,
+ perBucket: flags["per-bucket"] ? Number(flags["per-bucket"]) : null,
+ requireAudio: flags["require-audio"] === "true",
+ });
+ return;
+ }
+
+ if (flags["score-existing"] === "true") {
+ await scoreExisting(minChapters, flags.limit ? Number(flags.limit) : 200);
+ return;
+ }
+
+ const sample = JSON.parse(await readFile(samplePath, "utf8")) as Sample;
+ const label = flags.label ?? "round";
+ const buckets = (flags.buckets ?? BUCKETS.join(","))
+ .split(",")
+ .map((b) => b.trim())
+ .filter((b): b is Bucket => (BUCKETS as readonly string[]).includes(b));
+ const modes = (flags.modes ?? "absolute")
+ .split(",")
+ .map((m) => m.trim())
+ .filter((m): m is DigestTimestampMode => m === "absolute" || m === "chunk-local");
+ const specs = (flags.candidates ?? "qwen2.5:7b@16384")
+ .split(",")
+ .map((s) => s.trim())
+ .filter(Boolean);
+ // The paired A/B axis. Default "off" — the same rendering every earlier round
+ // used, so omitting the flag reproduces them.
+ const speakerArms = (flags.speakers ?? "off")
+ .split(",")
+ .map((s) => s.trim())
+ .filter((s) => s === "on" || s === "off")
+ .map((s) => s === "on");
+
+ const videos = sample.videos.filter((v) => buckets.includes(v.bucket));
+ if (videos.length === 0) throw new Error(`No sample videos in buckets ${buckets.join(",")}`);
+
+ const maxCuesOverride = flags["max-cues"] ? Number(flags["max-cues"]) : null;
+ const candidates: Candidate[] = [];
+ for (const spec of specs) {
+ for (const mode of modes) {
+ for (const speakers of speakerArms) {
+ candidates.push(parseCandidate(spec, mode, speakers, maxCuesOverride));
+ }
+ }
+ }
+
+ console.log(
+ `Bake-off ${label}: ${candidates.length} candidate(s) x ${videos.length} video(s) ` +
+ `(${Math.round(videos.reduce((a, v) => a + v.durationSeconds, 0) / 3600)} audio-hours each).`,
+ );
+
+ // Up front, not just before the final write, so the per-video checkpoint below
+ // has somewhere to land from the very first video.
+ await mkdir(outDir, { recursive: true });
+
+ const scores: CandidateScore[] = [];
+ for (const candidate of candidates) {
+ console.log(`\n=== ${candidate.key} (maxCues ${candidate.maxCues}) ===`);
+ const startedAt = Date.now();
+ const perVideo: VideoScore[] = [];
+ for (const video of videos) {
+ const s = await scoreVideo(video, candidate, (m) => console.log(m));
+ if (s) perVideo.push(s);
+ // CHECKPOINT AFTER EVERY VIDEO, because the reports are only written when
+ // the whole run finishes and a long run does not reliably get there. A
+ // 12-video run was OOM-killed on its last video after 26 minutes of engine
+ // time and left NOTHING behind — no error, no partial, just a dead process
+ // (the kill is a SIGKILL, so no handler can save it either). On a box that
+ // swaps, "it completed 11 of 12" has to survive.
+ await writeFile(
+ path.join(outDir, `${label}.partial.json`),
+ `${JSON.stringify({ label, candidate: candidate.key, videos: perVideo }, null, 2)}\n`,
+ ).catch(() => {});
+ }
+ const agg = aggregate(candidate, perVideo);
+ // Days, from measured seconds-per-audio-hour against the corpus total
+ // captured in the same scan that picked the sample.
+ agg.totals.projectedSweepDays = round(
+ (agg.totals.secondsPerAudioHour * sample.corpus.audioHours) / 86400,
+ 1,
+ );
+ scores.push(agg);
+ console.log(
+ `--- ${candidate.key}: ${agg.totals.kept} chapters, ` +
+ `${agg.totals.zeroYieldChunks}/${agg.totals.chunks} zero-yield, ` +
+ `${agg.totals.tokensPerSecond} tok/s, ` +
+ `projected ${agg.totals.projectedSweepDays} sweep days ` +
+ `(wall ${Math.round((Date.now() - startedAt) / 60000)} min)`,
+ );
+ }
+
+ scores.sort((a, b) => a.totals.zeroYieldRate - b.totals.zeroYieldRate);
+
+ await mkdir(outDir, { recursive: true });
+ const jsonPath = path.join(outDir, `${label}.json`);
+ const mdPath = path.join(outDir, `${label}.md`);
+ await writeFile(
+ jsonPath,
+ `${JSON.stringify({ label, sample: samplePath, corpus: sample.corpus, buckets, scores }, null, 2)}\n`,
+ );
+ await writeFile(mdPath, markdownReport(label, sample, buckets, scores));
+ console.log(`\nWrote ${jsonPath}\nWrote ${mdPath}`);
+}
+
+main().catch((err) => {
+ console.error(err);
+ process.exit(1);
+});
diff --git a/plans/tools/tsconfig.json b/plans/tools/tsconfig.json
@@ -0,0 +1,8 @@
+{
+ "extends": "../../tsconfig.base.json",
+ "compilerOptions": {
+ "types": ["node"],
+ "typeRoots": ["../../common/node_modules/@types"]
+ },
+ "include": ["*.ts"]
+}
diff --git a/sites/jeralyzer/site.json.example b/sites/jeralyzer/site.json.example
@@ -1,37 +0,0 @@
-{
- "siteId": "jeralyzer",
- "siteTitle": "Jeralyzer",
- "siteDescription": "Browse and search Jeremy Hambly's video transcripts",
- "headerTitle": "Jeralyzer",
- "homeTagline": "",
- "cloudflareProject": "jeralyzer",
- "groups": [
- { "id": "default", "name": "", "selectedByDefault": true, "order": 0 },
- {
- "id": "archives",
- "name": "Archives",
- "selectedByDefault": true,
- "description": "Channels that archive Jeremy's content, out of his reach",
- "order": 1
- },
- {
- "id": "other",
- "name": "Extended Universe",
- "selectedByDefault": false,
- "description": "Interesting related characters who aren't Jeremy himself",
- "order": 2
- }
- ],
- "defaultGroupId": "default",
- "channels": [
- { "slug": "jeremy-hambly", "groupId": "default" },
- { "slug": "the-quartering", "groupId": "archives" }
- ],
- "socialLinks": [
- {
- "label": "x.com",
- "url": "https://x.com/IMIJS_KF",
- "svg": "<svg aria-hidden=\"true\" viewBox=\"0 0 1200 1227\" fill=\"currentColor\" xmlns=\"http://www.w3.org/2000/svg\"><path d=\"M714.163 519.284L1160.89 0H1055.03L667.137 450.887L357.328 0H0L468.492 681.821L0 1226.37H105.866L515.491 750.218L842.672 1226.37H1200L714.137 519.284H714.163ZM569.165 687.828L521.697 619.934L144.011 79.6944H306.615L611.412 515.685L658.88 583.579L1055.08 1150.3H892.476L569.165 687.854V687.828Z\"/></svg>"
- }
- ]
-}