#!/usr/bin/env tsx // Score attribution PROMPT VARIANTS against a fixed pair of videos, without // writing a single sidecar. // // WHY THIS EXISTS SEPARATELY FROM attribution-pilot.ts, which measures the same // lane. The pilot calls attributeOneVideo because the SHIPPED PATH was what was // under test — the freshness skip, the downgrade guard, the provenance, the // sidecar a human then reads. This is the opposite situation: the thing under // test is a prompt and a schema that have won nothing yet, so it drives // attributionPrompt + attributionTurns + the digest app DIRECTLY and scores in // memory. That is digest-bakeoff.ts's defining property and it is inherited // deliberately: a losing variant leaves NOTHING in the corpus and no freshness // record has to be invalidated to undo a round. // // find transcripts/channels -name attribution.json | wc -l // // is 9 before and after, and this harness checks that itself — see the sidecar // census at the end of main(). writeAttribution is never imported. // // THE QUESTION ROUND 2 ASKS. The 2026-08-08 pilot rejected the text-only lane on // both pre-stated thresholds: 14.70 new labels per chunk against a <= 0.3 bar, // and 66.2 s/chunk engine against the digest lane's 11.2. The per-chunk data // says the cause is NOT the knownSpeakers roster the write-up blamed — on // destiny/5nmDzKB23OU 15 of 16 labels first appear in chunk 0, where no roster // exists, and on rekietalaw/EsZhaCfc8HQ the collapse is intermittent rather than // monotonic. Dozens of labels sit at exactly 60 characters, the schema's own // maxLength. The model was continuing the transcript into a free-string field. // // So the variable is the SCHEMA, and only the schema. Same engine (qwen2.5:7b // @ 8192), same chunker, same videos, same scorer, same metrics as the pilot — // lib/attributionScore.ts exists so that last one is true by construction rather // than by two files agreeing. // // UNIT DISCIPLINE, inherited and not negotiable. plans/STATE.md records a // RETRACTED throughput headline that came from pricing a sweep in // seconds-per-audio-hour when the unit of work is the CHUNK (density varies 4x // across this corpus). This reports seconds per CHUNK, prints chunks-per- // audio-hour beside any rate, and never prints a bare seconds-per-audio-hour. // // Examples: // tsx bin/attribution-bakeoff.ts --dry-run // tsx bin/attribution-bakeoff.ts --variant closed-cast --label round2-closed // tsx bin/attribution-bakeoff.ts --variant open --label round2-open-control // tsx bin/attribution-bakeoff.ts --variant closed-cast --model gemma2:9b \ // --num-ctx 8192 --label round2-gemma import os from "node:os"; import path from "node:path"; import { mkdir, readdir, writeFile } from "node:fs/promises"; import { getPaths, type Paths } from "../../common/lib/paths"; import { parseFlags } from "../../common/bin/_parseFlags"; import { ATTRIBUTION_FILENAME, ATTRIBUTION_PROMPT_VERSION, type AttributionRecord, type AttributionWarning, } from "../../common/lib/attribution"; import type { AttributionSettings } from "../../common/lib/settings"; import { OLLAMA_DIGEST_APP_ID } from "../../common/lib/digestApps"; import type { DigestApp, DigestAppConfig } from "../../common/lib/digestApps"; import { resolveAttributionTarget } from "../../common/controller/attributionTarget"; import { ATTRIBUTION_CAST_SAMPLES, ATTRIBUTION_MAX_SPEAKERS, ATTRIBUTION_MAX_TURNS_PER_CHUNK, ATTRIBUTION_UNKNOWN_SPEAKER, CAST_SYSTEM_PROMPT, TURN_SYSTEM_PROMPT, buildCastPrompt, buildClosedTurnPrompt, buildTurnPrompt, castSchema, isUselessSpeakerLabel, selectTranscriptSamples, turnSchema, turnSchemaClosed, } from "../../common/lib/attributionPrompt"; import { assembleTurns, createSpeakerRoster, type SpeakerMark, } from "../../common/lib/attributionTurns"; import { chunkRangesFor, isTranscriptTextLabel, scoreRecord, transcriptHaystack, type VideoMetrics, } from "../../common/lib/attributionScore"; import { parseEngineCall, isUnparsedEngineLine, type CallTiming } from "../../common/lib/engineTiming"; import { DIGEST_OVERLAP_CUES, hmsToSeconds, maxCuesForContext, toHms, } from "../../common/lib/digestPrompt"; import { chunkCuesForContext } from "../../common/lib/transcriptWindow"; import { transcriptToMarkdown } from "../../common/lib/transcriptToMarkdown"; import { readDigestContext } from "../../common/lib/digestContext-server"; import { chunksPerAudioHour, readChunkCensus, sweepDays, } from "../../common/controller/digestPlan"; import { isCuesJsonFresh, readNormalizedTranscript, } from "../../common/controller/normalizeTranscript"; import type { Cue } from "../../common/lib/vtt"; // The digest lane's own cost on the SAME model and context — round 2's 190 // engine-seconds over 17 chunks. The bar a second sweep has to justify itself // against, not a footnote. const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2; // The pilot's measured text-only numbers on these same videos, so a round-2 // report states what it is beating without anyone opening the round-1 file. const PILOT_BASELINE = { newLabelsPerChunkMean: 14.7, secondsPerChunkEngine: 66.2, byVideo: { "destiny/5nmDzKB23OU": { chunks: 4, labels: 16, newLabelsPerChunk: 0.33 }, "rekietalaw/EsZhaCfc8HQ": { chunks: 8, labels: 90, newLabelsPerChunk: 12.57 }, } as Record, }; // THE PROBE PAIR, and why these two. Both have a measured round-1 baseline, and // between them they exhibit BOTH failure shapes: destiny collapses in chunk 0 // where no roster exists, rekietalaw collapses intermittently (chunks 1, 3 and 6 // bad; 0, 2, 4, 5, 7 clean). 12 chunk calls + 2 cast calls. const DEFAULT_VIDEOS = ["destiny/5nmDzKB23OU", "rekietalaw/EsZhaCfc8HQ"]; // The thresholds, written BEFORE the run — see plans/STATE.md. A bar chosen // after seeing the numbers is not a bar. const THRESHOLDS = { newLabelsPerChunk: 0.3, secondsPerChunkEngine: 25, transcriptTextLabelRate: 0, }; type Variant = "closed-cast" | "open"; // --------------------------------------------------------------------------- // The sample // --------------------------------------------------------------------------- type ProbeVideo = { slug: string; channelSlug: string; videoDir: string; videoId: string; title: string; channel: string; durationSeconds: number; dir: string; cues: Cue[]; chunks: Cue[][]; chunkRanges: { start: number; end: number }[]; }; async function priceVideo( slug: string, paths: Paths, maxCues: number, ): Promise { const slash = slug.indexOf("/"); if (slash <= 0) throw new Error(`--videos expects /, got "${slug}"`); const channelSlug = slug.slice(0, slash); const videoDir = slug.slice(slash + 1); const dir = path.join(paths.channelsDir, channelSlug, "data", videoDir); const { fresh, cuesPath } = await isCuesJsonFresh(dir); if (!fresh) return null; const transcript = await readNormalizedTranscript(cuesPath); const cues = transcript?.cues ?? []; if (!transcript || cues.length === 0) return null; const chunks = chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES }); const lastCue = cues[cues.length - 1]; return { slug, channelSlug, videoDir, videoId: transcript.id || videoDir, title: transcript.title || videoDir, channel: transcript.channel || channelSlug, durationSeconds: Math.round(transcript.duration || lastCue.end || lastCue.start || 0), dir, cues, chunks, // Derived by the SHARED helper, so the scorer's chunk indexing cannot drift // from the chunker this run actually used. chunkRanges: chunkRangesFor(cues, maxCues), }; } // Render one chunk exactly as attributeOne.ts does — ABSOLUTE timestamps, // because an attribution mark lands on one shared timeline that later chunks // keep adding to. Re-basing per chunk would make every chunk's marks collide. function renderChunk(video: ProbeVideo, cues: Cue[]): string { return transcriptToMarkdown( { id: video.videoId, title: video.title, channel: video.channel, duration: video.durationSeconds, cues, }, { timestamps: true, includeDescription: false, includeTags: false, stampForCue: (_clock, seconds) => toHms(Math.max(0, seconds)), }, ); } // --------------------------------------------------------------------------- // The two variants // --------------------------------------------------------------------------- type CastMember = { name: string; confidence: number }; type VariantRun = { record: AttributionRecord; warnings: AttributionWarning[]; chunksOk: number; // closed-cast only cast?: CastMember[]; castRejected?: string[]; castCallSeconds?: number; unknownTurns?: number; offCastTurns?: number; // Turns whose timestamp fell outside the chunk's own range. Counted, because // they are DROPPED silently and a run that produced nothing but these would // otherwise be indistinguishable from a model that returned nothing at all. outOfRangeTurns: number; totalTurnsAccepted: number; }; type RunCtx = { app: DigestApp; config: DigestAppConfig; contextNote?: string; log: (m: string) => void; }; function emptyRecord( video: ProbeVideo, model: string, chunks: number, chunksOk: number, warnings: AttributionWarning[], ): AttributionRecord { return { videoId: video.videoId, generatedAt: new Date().toISOString(), speakers: [], segments: [], ...(warnings.length ? { warnings } : {}), provenance: { method: "text-only", appId: "bakeoff", model, modelRequested: model, // NOT the shipped ATTRIBUTION_PROMPT_VERSION. This record never reaches // disk, and stamping it with the live version would be a lie waiting for // someone to copy it somewhere that matters. promptVersion: -1, generatedAt: new Date().toISOString(), chunks, chunksOk, durationMs: 0, }, }; } // PASS A — cast discovery. ONE call for the whole video. async function discoverCast( video: ProbeVideo, ctx: RunCtx, ): Promise<{ cast: CastMember[]; rejected: string[]; seconds: number; model: string }> { const samples = selectTranscriptSamples(video.cues, ATTRIBUTION_CAST_SAMPLES); ctx.log( ` ${video.slug}: cast discovery over ${samples.length} evenly-spaced excerpt(s), 1 call.`, ); const started = Date.now(); const result = await ctx.app.run({ config: ctx.config, system: CAST_SYSTEM_PROMPT, prompt: buildCastPrompt({ title: video.title, channel: video.channel, samples, ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}), }), schema: castSchema(), onLog: ctx.log, }); const seconds = (Date.now() - started) / 1000; const haystack = transcriptHaystack(video.cues); const cast: CastMember[] = []; const rejected: string[] = []; const seen = new Set(); const raw = (result.data as { cast?: unknown })?.cast; if (Array.isArray(raw)) { for (const item of raw) { const e = item as { name?: unknown; confidence?: unknown }; if (typeof e?.name !== "string") continue; const name = e.name.trim(); // THE SAME GUARDS THE RECORD GETS, applied to the cast — pass A uses a // free string, so it can fail the exact way pass B is being fixed for. A // transcript fragment admitted here would be minted into the enum and // become a legitimate label for the rest of the video. if (isUselessSpeakerLabel(name) || isTranscriptTextLabel(name, haystack)) { rejected.push(name); continue; } const key = name.toLowerCase(); if (seen.has(key)) continue; seen.add(key); cast.push({ name, confidence: typeof e.confidence === "number" && Number.isFinite(e.confidence) ? Math.min(1, Math.max(0, e.confidence)) : 0, }); if (cast.length >= ATTRIBUTION_MAX_SPEAKERS) break; } } return { cast, rejected, seconds, model: result.model }; } // PASS B — turn assignment, per chunk, against a CLOSED label space. async function runClosedCast(video: ProbeVideo, ctx: RunCtx): Promise { const warnings: AttributionWarning[] = []; const { cast, rejected, seconds: castCallSeconds, model: castModel } = await discoverCast(video, ctx); ctx.log( ` ${video.slug}: cast = ${ cast.map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "(none)" }${rejected.length ? ` — rejected ${rejected.length}` : ""}`, ); if (cast.length === 0) { // Pass A returning nothing usable is a RESULT, not a crash: the plan's third // decision row ("Pass A itself returns garbage") escalates to gemma2:9b. warnings.push({ code: "empty", detail: "cast discovery produced no usable name" }); return { record: emptyRecord(video, castModel, video.chunks.length, 0, warnings), warnings, chunksOk: 0, cast, castRejected: rejected, castCallSeconds, outOfRangeTurns: 0, totalTurnsAccepted: 0, }; } const names = cast.map((c) => c.name); const schema = turnSchemaClosed(names, ATTRIBUTION_MAX_TURNS_PER_CHUNK); // The roster is SEEDED with the whole cast and never grows. That is the point: // cross-chunk identity stops being something the model has to remember and // becomes a property of the grammar it decodes against. const roster = createSpeakerRoster(names); const marks: SpeakerMark[] = []; let model = castModel; let chunksOk = 0; let unknownTurns = 0; let offCastTurns = 0; let outOfRange = 0; let accepted = 0; for (let i = 0; i < video.chunks.length; i++) { const chunk = video.chunks[i]; const startSeconds = Math.max(0, Math.floor(chunk[0].start)); const endSeconds = Math.max( startSeconds, Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start), ); try { const result = await ctx.app.run({ config: ctx.config, system: TURN_SYSTEM_PROMPT, prompt: buildClosedTurnPrompt({ title: video.title, channel: video.channel, startSeconds, endSeconds, transcript: renderChunk(video, chunk), cast: names, ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}), }), schema, onLog: ctx.log, }); chunksOk++; model = result.model || model; const raw = (result.data as { turns?: unknown })?.turns; if (!Array.isArray(raw)) continue; for (const item of raw) { const e = item as { start?: unknown; speaker?: unknown }; if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue; const at = hmsToSeconds(e.start); if (at === null) { warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) }); continue; } if (at < startSeconds || at > endSeconds) { outOfRange++; continue; } if (e.speaker.trim() === ATTRIBUTION_UNKNOWN_SPEAKER) { // Counted, then dropped. An honest Unknown is a good answer and the // rate is worth reading, but it must not become a speaker. unknownTurns++; continue; } if (isUselessSpeakerLabel(e.speaker)) { unknownTurns++; continue; } // THE CHECK THAT MAKES THIS A MEASUREMENT. The enum pins `speaker`, and // the parser verifies it anyway — the belt-and-braces nameClusters // already applies to cluster indices. If a label outside the cast ever // lands here, the grammar did NOT hold and the whole hypothesis is // refuted; offCastTurns is that count, and it must be 0. const index = roster.lookup(e.speaker); if (index === undefined) { offCastTurns++; warnings.push({ code: "unknown-cluster", chunk: i, detail: `off-cast label: ${e.speaker.slice(0, 60)}`, }); continue; } marks.push({ at, speaker: index }); accepted++; } } catch (err) { const message = (err as Error)?.message ?? String(err); warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) }); ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`); } } const lastCue = video.cues[video.cues.length - 1]; // Only cast members that actually SPOKE become speakers. A name pass A // proposed and pass B never used is not in the record — it would otherwise // score as a label at 0 seconds and inflate the label count for nothing. const used = [...new Set(marks.map((m) => m.speaker))].sort((a, b) => a - b); const remap = new Map(used.map((old, i) => [old, i])); const { speakers, segments } = assembleTurns({ labels: used.map((i) => roster.labels[i]), marks: marks.map((m) => ({ at: m.at, speaker: remap.get(m.speaker)! })), transcriptEnd: lastCue.end || lastCue.start, }); const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings); record.speakers = speakers; record.segments = segments; return { record, warnings, chunksOk, cast, castRejected: rejected, castCallSeconds, unknownTurns, offCastTurns, outOfRangeTurns: outOfRange, totalTurnsAccepted: accepted, }; } // The OPEN control — the shipped prompt and schema, driven in memory. Reproduces // round 1 without writing a sidecar, so a round-2 report can state the delta // under identical box conditions instead of against a number measured last week. async function runOpen(video: ProbeVideo, ctx: RunCtx): Promise { const warnings: AttributionWarning[] = []; const roster = createSpeakerRoster(); const marks: SpeakerMark[] = []; let model = ""; let chunksOk = 0; let accepted = 0; let outOfRange = 0; for (let i = 0; i < video.chunks.length; i++) { const chunk = video.chunks[i]; const startSeconds = Math.max(0, Math.floor(chunk[0].start)); const endSeconds = Math.max( startSeconds, Math.ceil(chunk[chunk.length - 1].end || chunk[chunk.length - 1].start), ); try { const result = await ctx.app.run({ config: ctx.config, system: TURN_SYSTEM_PROMPT, prompt: buildTurnPrompt({ title: video.title, channel: video.channel, startSeconds, endSeconds, transcript: renderChunk(video, chunk), ...(roster.labels.length > 0 ? { knownSpeakers: roster.labels.slice() } : {}), ...(ctx.contextNote ? { contextNote: ctx.contextNote } : {}), }), schema: turnSchema(), onLog: ctx.log, }); chunksOk++; model = result.model || model; const raw = (result.data as { turns?: unknown })?.turns; if (!Array.isArray(raw)) continue; for (const item of raw) { const e = item as { start?: unknown; speaker?: unknown }; if (typeof e?.start !== "string" || typeof e.speaker !== "string") continue; const at = hmsToSeconds(e.start); if (at === null) { warnings.push({ code: "bad-timestamp", chunk: i, detail: e.start.slice(0, 40) }); continue; } if (at < startSeconds || at > endSeconds) { outOfRange++; continue; } if (isUselessSpeakerLabel(e.speaker)) continue; marks.push({ at, speaker: roster.intern(e.speaker) }); accepted++; } } catch (err) { const message = (err as Error)?.message ?? String(err); warnings.push({ code: "chunk-failed", chunk: i, detail: message.slice(0, 300) }); ctx.log(` ${video.slug}: chunk ${i + 1}/${video.chunks.length} failed: ${message}`); } } const lastCue = video.cues[video.cues.length - 1]; const { speakers, segments } = assembleTurns({ labels: roster.labels, marks, transcriptEnd: lastCue.end || lastCue.start, }); const record = emptyRecord(video, model, video.chunks.length, chunksOk, warnings); record.speakers = speakers; record.segments = segments; return { record, warnings, chunksOk, outOfRangeTurns: outOfRange, totalTurnsAccepted: accepted, }; } // --------------------------------------------------------------------------- // The run // --------------------------------------------------------------------------- type VideoResult = { slug: string; title: string; durationSeconds: number; chunks: number; chunksOk: number; wallSeconds: number; calls: CallTiming[]; unparsedLogLines: number; warningsByCode: Record; metrics: VideoMetrics | null; cast?: CastMember[]; castRejected?: string[]; castCallSeconds?: number; unknownTurns?: number; offCastTurns?: number; outOfRangeTurns: number; totalTurnsAccepted: number; speakers: { label: string; seconds: number }[]; }; function sum(v: number[]): number { return v.reduce((a, b) => a + b, 0); } function boxState(): Record { return { loadavg: os.loadavg().map((n) => Number(n.toFixed(2))), freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)), totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)), }; } // How many attribution.json files exist right now. THE SAFETY CLAIM of this // harness, checked rather than asserted in a comment. async function countSidecars(paths: Paths): Promise { let total = 0; let channels: string[]; try { channels = (await readdir(paths.channelsDir, { withFileTypes: true })) .filter((e) => e.isDirectory()) .map((e) => e.name); } catch { return 0; } for (const channel of channels) { const dataDir = path.join(paths.channelsDir, channel, "data"); let videos; try { videos = await readdir(dataDir, { withFileTypes: true }); } catch { continue; } for (const v of videos) { if (!v.isDirectory()) continue; try { const files = await readdir(path.join(dataDir, v.name)); if (files.includes(ATTRIBUTION_FILENAME)) total++; } catch { // A video dir that vanished mid-walk is not a sidecar. } } } return total; } async function runVideo( video: ProbeVideo, variant: Variant, ctx: Omit, log: (m: string) => void, ): Promise { const calls: CallTiming[] = []; let unparsed = 0; // TEES, and it has to. This one closure is both the engine-timing sink and the // harness's own progress channel — the cast list, the per-chunk failures. An // earlier version only parsed, which silently swallowed the cast line and made // a 0-label result impossible to diagnose from the console. const onLog = (line: string) => { const call = parseEngineCall(line); if (call) { calls.push(call); return; } if (isUnparsedEngineLine(line)) { unparsed++; return; } log(line); }; const started = Date.now(); const out = variant === "closed-cast" ? await runClosedCast(video, { ...ctx, log: onLog }) : await runOpen(video, { ...ctx, log: onLog }); const wallSeconds = (Date.now() - started) / 1000; const warningsByCode: Record = {}; for (const w of out.warnings) { warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1; } // Scored WITH the cues, so transcriptTextLabelRate is a number rather than // null — it is the metric this round exists to drive to zero. const metrics = out.record.speakers.length > 0 ? scoreRecord(out.record, { chunkRanges: video.chunkRanges, durationSeconds: video.durationSeconds, cues: video.cues, }) : null; const result: VideoResult = { slug: video.slug, title: video.title, durationSeconds: video.durationSeconds, chunks: video.chunks.length, chunksOk: out.chunksOk, wallSeconds, calls, unparsedLogLines: unparsed, warningsByCode, metrics, ...(out.cast ? { cast: out.cast } : {}), ...(out.castRejected ? { castRejected: out.castRejected } : {}), ...(out.castCallSeconds !== undefined ? { castCallSeconds: out.castCallSeconds } : {}), ...(out.unknownTurns !== undefined ? { unknownTurns: out.unknownTurns } : {}), ...(out.offCastTurns !== undefined ? { offCastTurns: out.offCastTurns } : {}), outOfRangeTurns: out.outOfRangeTurns, totalTurnsAccepted: out.totalTurnsAccepted, speakers: out.record.speakers.map((s) => ({ label: s.label, seconds: s.seconds ?? 0 })), }; log( ` ${video.slug} ${video.chunks.length} chunk(s) → ${out.chunksOk} ok, ` + `${out.record.speakers.length} label(s), ${out.record.segments.length} segment(s)` + (metrics ? `, new/chunk ${fixed(metrics.newLabelsPerChunk)}, ` + `transcript-text ${metrics.transcriptTextLabelRate === null ? "n/a" : pct(metrics.transcriptTextLabelRate)}, ` + `top share ${pct(metrics.dominantSpeakerShare)}` : "") + (out.offCastTurns ? `, OFF-CAST ${out.offCastTurns}` : "") + (out.unknownTurns ? `, Unknown ${out.unknownTurns}` : "") + (out.outOfRangeTurns ? `, out-of-range ${out.outOfRangeTurns}` : "") + `, ${wallSeconds.toFixed(0)}s wall`, ); return result; } // --------------------------------------------------------------------------- // Reporting // --------------------------------------------------------------------------- function pct(n: number): string { return `${(n * 100).toFixed(0)}%`; } function fixed(n: number | null | undefined, digits = 2): string { return n === null || n === undefined ? "n/a" : n.toFixed(digits); } type CostSummary = { chunks: number; chunksOk: number; calls: number; callsWithEngineTiming: number; audioSeconds: number; chunksPerAudioHour: number; // Chunk-call timings ONLY. The cast call is one per VIDEO and pricing it as a // chunk would flatter or punish the variant depending on video length; it is // reported separately, in its own unit. secondsPerChunkWall: number | null; secondsPerChunkEngine: number | null; secondsPerChunkEngineExLoad: number | null; meanLoadSeconds: number | null; decodeTokensPerSecond: number | null; outputTokens: number; castCallSecondsMean: number | null; unparsedLogLines: number; }; function summarizeCost(results: VideoResult[], variant: Variant): CostSummary { const calls = results.flatMap((r) => r.calls); // The cast call is the FIRST call of each closed-cast video. Dropping it from // the per-chunk denominators keeps "s/chunk" meaning one chunk. const chunkCalls = variant === "closed-cast" ? results.flatMap((r) => r.calls.slice(1)) : calls; const withEngine = chunkCalls.filter((c) => c.engineSeconds !== undefined); const withLoad = chunkCalls.filter((c) => c.loadSeconds !== undefined); const chunks = sum(results.map((r) => r.chunks)); const audioSeconds = sum(results.map((r) => r.durationSeconds)); const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0)); const outTokens = sum(calls.map((c) => c.outputTokens ?? 0)); const castSeconds = results .map((r) => r.castCallSeconds) .filter((s): s is number => s !== undefined); const chunkWall = sum( results.map((r) => r.wallSeconds - (r.castCallSeconds ?? 0)), ); return { chunks, chunksOk: sum(results.map((r) => r.chunksOk)), calls: calls.length, callsWithEngineTiming: withEngine.length, audioSeconds, chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds), secondsPerChunkWall: chunks > 0 ? chunkWall / chunks : null, secondsPerChunkEngine: withEngine.length ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length : null, secondsPerChunkEngineExLoad: withEngine.length ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) / withEngine.length : null, meanLoadSeconds: withLoad.length ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length : null, decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null, outputTokens: outTokens, castCallSecondsMean: castSeconds.length ? sum(castSeconds) / castSeconds.length : null, unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)), }; } // The verdict, computed against thresholds fixed before the run. function verdict(results: VideoResult[], cost: CostSummary) { const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null); const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null); const newPerChunk = withSeams.length ? sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length : null; const textRates = scored .map((m) => m.transcriptTextLabelRate) .filter((r): r is number => r !== null); const textRate = textRates.length ? Math.max(...textRates) : null; const engine = cost.secondsPerChunkEngine; const offCast = sum(results.map((r) => r.offCastTurns ?? 0)); const identityOk = newPerChunk !== null && newPerChunk <= THRESHOLDS.newLabelsPerChunk; const textOk = textRate !== null && textRate <= THRESHOLDS.transcriptTextLabelRate; const costOk = engine !== null && engine < THRESHOLDS.secondsPerChunkEngine; return { newLabelsPerChunkMean: newPerChunk, worstTranscriptTextLabelRate: textRate, secondsPerChunkEngine: engine, offCastTurns: offCast, identityOk, textOk, costOk, // A single label at ~100% scores a perfect headline and is worthless. degenerateVideos: scored.filter((m) => m.labelsPerVideo === 1).length, passes: identityOk && textOk && costOk && offCast === 0, }; } function reportMarkdown(args: { label: string; variant: Variant; model: string; numCtx?: number; maxCues: number; results: VideoResult[]; cost: CostSummary; v: ReturnType; census: { chunks: number; videos: number; chunksPerAudioHour: number; statsSchemaStale: boolean } | null; boxStart: Record; boxEnd: Record; startedAt: string; sidecarsBefore: number; sidecarsAfter: number; }): string { const { results, cost, v } = args; const out: string[] = []; out.push(`# Attribution bake-off — ${args.label}`); out.push(""); out.push( `Variant **${args.variant}** · ${results.length} video(s) · model \`${args.model}\` @ numCtx ` + `${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`, ); out.push(""); out.push( "**No sidecar was written.** `attribution.json` count " + `${args.sidecarsBefore} before, ${args.sidecarsAfter} after` + (args.sidecarsBefore === args.sidecarsAfter ? "." : " — **MISMATCH, investigate.**"), ); out.push(""); out.push("## Verdict against the thresholds fixed before the run"); out.push(""); out.push("| criterion | bar | measured | |"); out.push("| --- | ---: | ---: | :-: |"); out.push( `| new labels/chunk (mean) | ≤ ${THRESHOLDS.newLabelsPerChunk} | ${fixed(v.newLabelsPerChunkMean)} | ${v.identityOk ? "PASS" : "FAIL"} |`, ); out.push( `| transcript-text labels (worst video) | ${THRESHOLDS.transcriptTextLabelRate} | ${ v.worstTranscriptTextLabelRate === null ? "n/a" : pct(v.worstTranscriptTextLabelRate) } | ${v.textOk ? "PASS" : "FAIL"} |`, ); out.push( `| s/chunk engine | < ${THRESHOLDS.secondsPerChunkEngine} | ${fixed(v.secondsPerChunkEngine, 1)} | ${v.costOk ? "PASS" : "FAIL"} |`, ); if (args.variant === "closed-cast") { out.push( `| off-cast labels (grammar held) | 0 | ${v.offCastTurns} | ${v.offCastTurns === 0 ? "PASS" : "FAIL"} |`, ); } out.push(""); out.push(`**${v.passes ? "ALL THRESHOLDS MET" : "NOT MET"}.**`); out.push(""); out.push( `Round 1 (open schema, same model/context/videos) measured ` + `**${PILOT_BASELINE.newLabelsPerChunkMean} new labels/chunk** and ` + `**${PILOT_BASELINE.secondsPerChunkEngine} s/chunk engine**. Digest lane baseline: ` + `${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE} s/chunk engine.`, ); out.push(""); out.push( `Single-label (degenerate-risk) videos: ${v.degenerateVideos} of ${results.length}. ` + "A perfect headline with one label at ~100% talk time is worthless — read the two together.", ); out.push(""); if (args.variant === "closed-cast") { out.push("## Pass A — cast discovery (1 call per video)"); out.push(""); out.push("| video | cast | rejected | call s |"); out.push("| --- | --- | ---: | ---: |"); for (const r of results) { out.push( `| ${r.slug} | ${ (r.cast ?? []).map((c) => `${c.name} (${c.confidence.toFixed(2)})`).join(", ") || "—" } | ${r.castRejected?.length ?? 0} | ${fixed(r.castCallSeconds, 1)} |`, ); } out.push(""); out.push( "**Pass A is independently valuable.** If pass B churns, this alone is the", 'cast-list product — ~1 call per video, order 10 days corpus-wide — shipping as', '"videos featuring X" without ever claiming who said which line.', ); out.push(""); } out.push("## Cross-chunk identity, and the label space"); out.push(""); out.push( "| video | chunks | labels | new/chunk | round-1 new/chunk | transcript-text | truncated | singleton | top share |", ); out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |"); for (const r of results) { const m = r.metrics; const base = PILOT_BASELINE.byVideo[r.slug]; out.push( `| ${r.slug} | ${r.chunks} | ${m?.labelsPerVideo ?? "—"} | ${fixed(m?.newLabelsPerChunk)} | ${ base ? base.newLabelsPerChunk.toFixed(2) : "—" } | ${ m?.transcriptTextLabelRate === null || m === null ? "n/a" : pct(m.transcriptTextLabelRate) } | ${m ? pct(m.truncatedLabelRate) : "—"} | ${ m === null || m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate) } | ${m ? pct(m.dominantSpeakerShare) : "—"} |`, ); } out.push(""); const anyText = results.some((r) => (r.metrics?.transcriptTextLabels.length ?? 0) > 0); if (anyText) { out.push("Labels that are transcript text — the failure this round exists to eliminate:"); out.push(""); for (const r of results) { for (const l of r.metrics?.transcriptTextLabels ?? []) { out.push(`- \`${r.slug}\`: ${JSON.stringify(l)}`); } } out.push(""); } if (args.variant === "closed-cast") { out.push("## Pass B — how often the model declined"); out.push(""); out.push("| video | turns accepted | Unknown | off-cast | out-of-range |"); out.push("| --- | ---: | ---: | ---: | ---: |"); for (const r of results) { out.push( `| ${r.slug} | ${r.totalTurnsAccepted} | ${r.unknownTurns ?? 0} | ${r.offCastTurns ?? 0} | ${r.outOfRangeTurns} |`, ); } out.push(""); out.push( "`Unknown` is a GOOD answer, not a failure — the bias must run toward dropping on", "uncertainty. `off-cast` must be 0: anything else means the enum did not convert to", "a grammar and the whole hypothesis is refuted.", ); out.push(""); } out.push("## What the labels say"); out.push(""); for (const r of results) { const total = sum(r.speakers.map((s) => s.seconds)) || 1; out.push( `- \`${r.slug}\` — ${ r.speakers .map((s) => `${s.label} ${pct(s.seconds / total)}`) .join(", ") || "(no speaker)" }`, ); } out.push(""); const warnings: Record = {}; for (const r of results) { for (const [k, n] of Object.entries(r.warningsByCode)) warnings[k] = (warnings[k] ?? 0) + n; } out.push( `Warnings by code: ${ Object.keys(warnings).length ? Object.entries(warnings).map(([k, n]) => `${k} ${n}`).join(", ") : "none" }`, ); out.push(""); out.push("## Cost"); out.push(""); out.push( `${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s) total ` + `(${cost.callsWithEngineTiming} chunk call(s) with engine timing). Sample density ` + `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` + `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`, ); out.push(""); out.push("| quantity | value | round 1 |"); out.push("| --- | ---: | ---: |"); out.push(`| s/chunk, wall | ${fixed(cost.secondsPerChunkWall, 1)} | — |`); out.push( `| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} | ${PILOT_BASELINE.secondsPerChunkEngine} |`, ); out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} | — |`); out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} | — |`); out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} | — |`); out.push(`| TOTAL output tokens | ${cost.outputTokens} | — |`); if (args.variant === "closed-cast") { out.push(`| cast call s/VIDEO (not per chunk) | ${fixed(cost.castCallSecondsMean, 1)} | — |`); } out.push( `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} | — |`, ); out.push(""); out.push( "Output tokens are the thing to watch: engine time tracked wall time within 7% in", "round 1, so 66.2 s/chunk was pure decode volume (31,266 output tokens on one", "13-chunk video). An enum token per turn instead of a 60-character string is the", "mechanism by which this variant is meant to be cheaper as well as better.", ); out.push(""); if (args.census) { const idle = cost.secondsPerChunkEngineExLoad; const contended = cost.secondsPerChunkWall; out.push("## Corpus projection — per-turn lane"); out.push(""); out.push( `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` + `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` + (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") + ".", ); out.push(""); out.push( `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` + `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended** ` + `— from n = ${results.length} video(s) / ${cost.chunks} chunk(s), one box, one hour.`, ); out.push(""); out.push( `Cast-only, for comparison: **1 call per video** over ${args.census.videos.toLocaleString()} ` + `videos at ${fixed(cost.castCallSecondsMean, 1)}s = ` + `${ cost.castCallSecondsMean === null ? "n/a" : ((args.census.videos * cost.castCallSecondsMean) / 86400).toFixed(1) } days.`, ); out.push(""); out.push( "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that has", "completed 0.17% of its own ~194k calls, with nothing arbitrating between them.", ); out.push(""); } out.push("## Box"); out.push(""); out.push(`Start: \`${JSON.stringify(args.boxStart)}\``); out.push(""); out.push(`End: \`${JSON.stringify(args.boxEnd)}\``); out.push(""); if (cost.unparsedLogLines > 0) { out.push( `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`, ); out.push(""); } out.push("## Units"); out.push(""); out.push( "- The unit of per-turn work is the **chunk** (one model call). Seconds-per-audio-", " hour is not a unit here: chunk density varies 4× across this corpus, and pricing", " in it is the error behind a retracted throughput headline. This report does not", " print one.", "- The unit of cast discovery is **calls per video**, priced separately and never", " folded into s/chunk.", "- Wall and engine are different quantities and are never substituted for each", " other. Wall is the realistic ceiling on this shared box; engine excl. load is", " the floor.", `- Every rate carries its denominator. At n = ${results.length} videos a bare percentage`, " would be a lie of precision.", ); out.push(""); return out.join("\n"); } // --------------------------------------------------------------------------- async function main(): Promise { const flags = parseFlags(process.argv.slice(2)); const paths = getPaths(); const variant: Variant = flags.variant === "open" ? "open" : "closed-cast"; const dryRun = flags["dry-run"] === "true"; const slugs = (flags.videos ?? DEFAULT_VIDEOS.join(",")).split(",").map((s) => s.trim()).filter(Boolean); const label = flags.label ?? `${variant}-${new Date().toISOString().slice(0, 10)}`; // The lane is enabled IN MEMORY, exactly as the pilot does it. settings.json // is never written — a bake-off must not be the thing that arms a lane. const probeSettings: AttributionSettings = { enabled: true, appId: OLLAMA_DIGEST_APP_ID, model: flags.model ?? "", diarizedEnabled: false, textOnlyEnabled: true, promptVersion: ATTRIBUTION_PROMPT_VERSION, }; const { app, config: baseConfig, modelRequested } = resolveAttributionTarget( "text-only", probeSettings, ); // Overrides land on a COPY of the app config. resolveAttributionTarget reads // the real settings for the ollama URL and timeout, which we want, but the // model and context are this run's variables. const config: DigestAppConfig = { ...baseConfig, model: modelRequested, ...(flags["num-ctx"] ? { numCtx: Number(flags["num-ctx"]) } : {}), }; const maxCues = maxCuesForContext(config.numCtx); const videos: ProbeVideo[] = []; const skipped: string[] = []; for (const slug of slugs) { const v = await priceVideo(slug, paths, maxCues); if (v) videos.push(v); else skipped.push(slug); } const totalChunks = sum(videos.map((v) => v.chunks.length)); const totalAudio = sum(videos.map((v) => v.durationSeconds)); const census = await readChunkCensus(paths, maxCues); const castCalls = variant === "closed-cast" ? videos.length : 0; console.log( `Attribution bake-off — variant ${variant}, app ${app.id}, model ${modelRequested}, ` + `numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`, ); console.log(`Settings override (in memory only): ${JSON.stringify(probeSettings)}`); console.log("NO SIDECAR IS WRITTEN. writeAttribution is not imported by this harness."); console.log(""); for (const v of videos) { console.log( ` ${v.slug} ${toHms(v.durationSeconds)} ${v.cues.length} cues → ${v.chunks.length} chunk(s)`, ); } if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`); console.log(""); console.log( `${videos.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` + `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour.`, ); console.log( `Calls this run: ${castCalls} cast + ${totalChunks} chunk = ${castCalls + totalChunks}.`, ); if (census) { console.log( `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s)` + (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""), ); } console.log(`Box: ${JSON.stringify(boxState())}`); console.log(""); if (videos.length === 0) { console.error("No video had a usable transcript — nothing to measure."); process.exitCode = 1; return; } if (dryRun) { console.log("--dry-run: priced only, no model call and no write."); return; } const sidecarsBefore = await countSidecars(paths); console.log(`attribution.json on disk before: ${sidecarsBefore}`); console.log(""); const boxStart = boxState(); const startedAt = new Date().toISOString(); const results: VideoResult[] = []; for (const video of videos) { const context = await readDigestContext(paths, video.channelSlug); results.push( await runVideo( video, variant, { app, config, ...(context.note ? { contextNote: context.note } : {}) }, (m) => console.log(m), ), ); } const boxEnd = boxState(); const sidecarsAfter = await countSidecars(paths); const cost = summarizeCost(results, variant); const v = verdict(results, cost); const report = { version: 1 as const, label, variant, startedAt, finishedAt: new Date().toISOString(), engine: { appId: app.id, modelRequested, numCtx: config.numCtx, maxCues, overlapCues: DIGEST_OVERLAP_CUES, castSamples: ATTRIBUTION_CAST_SAMPLES, maxTurnsPerChunk: ATTRIBUTION_MAX_TURNS_PER_CHUNK, }, thresholds: THRESHOLDS, verdict: v, skipped, sidecars: { before: sidecarsBefore, after: sidecarsAfter }, box: { start: boxStart, end: boxEnd }, census, cost, baseline: { pilot: PILOT_BASELINE, digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE, }, videos: results, }; const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot"); await mkdir(outDir, { recursive: true }); const markdown = reportMarkdown({ label, variant, model: modelRequested, numCtx: config.numCtx, maxCues, results, cost, v, census, boxStart, boxEnd, startedAt, sidecarsBefore, sidecarsAfter, }); const jsonPath = path.join(outDir, `${label}.json`); const mdPath = path.join(outDir, `${label}.md`); await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n"); await writeFile(mdPath, markdown); console.log(""); console.log(markdown); console.log(`Wrote ${jsonPath}`); console.log(`Wrote ${mdPath}`); // THE SAFETY CLAIM, enforced. A bake-off that left a sidecar behind would have // silently made a losing variant look fresh to the whole system. if (sidecarsAfter !== sidecarsBefore) { console.error( `attribution.json count changed: ${sidecarsBefore} -> ${sidecarsAfter}. ` + "This harness must never write one.", ); process.exitCode = 1; } else { console.log(`attribution.json on disk after: ${sidecarsAfter} (unchanged).`); } if (sum(results.map((r) => r.chunksOk)) === 0) { console.error("No chunk succeeded — nothing was measured."); process.exitCode = 1; } } main().catch((err) => { console.error(err); process.exit(1); });