#!/usr/bin/env tsx // Measure the attribution lanes over a small sample BEFORE deciding whether the // text-only one is worth 25-55 GPU-days over the whole corpus. // // WHY THIS EXISTS. Both lanes shipped with nothing armed, deliberately: the // text-only lane costs roughly one model call per transcript CHUNK, which // corpus-wide is ~194,000 calls — the same order as the digest sweep, which has // itself completed 0.17%. That spend is decided on measurements, the way the // digest layer was (pilot -> bake-off -> measured defaults -> sweep), and this // is the measurement. // // WHY IT CALLS attributeOneVideo RATHER THAN BYPASSING IT, unlike // digest-bakeoff.ts which drives the prompt modules directly. The bake-off was // comparing ENGINES and had to leave nothing behind; this is measuring the // SHIPPED PATH — the freshness skip, the downgrade guard, the provenance it // records and the sidecar it writes are all part of what is under test, and the // record is the artifact a human then reads to judge the labels. So it writes // real attribution.json files into the real corpus. That is bounded (one small // sidecar per video, nothing consumes it yet) and reversible with one command: // // find transcripts/channels -name attribution.json -delete // // NO EDITOR, NO SETTINGS WRITE. attributeOneVideo takes an explicit `settings` // override and never calls writeSettings, so the pilot enables the lane // IN MEMORY and the operator's settings.json is untouched. A pilot must not be // the thing that arms a lane. // // UNIT DISCIPLINE, and it is not negotiable. plans/STATE.md records a RETRACTED // throughput headline that came from pricing a sweep in seconds-per-audio-hour // when the unit of work is the chunk (density varies 4x across this corpus). So // this harness reports seconds per CHUNK for the text-only lane and calls per // VIDEO for the diarized one, prints chunks-per-audio-hour beside any rate, and // REFUSES to print a bare seconds-per-audio-hour figure. sweepDaysFromAudio is // deliberately not imported. // // Examples: // tsx bin/attribution-pilot.ts --sample bakeoff --dry-run // tsx bin/attribution-pilot.ts --sample bakeoff --method text-only --label round1 // tsx bin/attribution-pilot.ts --sample channel:quarteringvlogs --label autocap // tsx bin/attribution-pilot.ts --sample video:ObviousRises-rumble/v6z1o2g \ // --method diarized --label diarized-smoke import os from "node:os"; import path from "node:path"; import { mkdir, readFile, readdir, writeFile } from "node:fs/promises"; import { getPaths, type Paths } from "../../common/lib/paths"; import { parseFlags } from "../../common/bin/_parseFlags"; import { ATTRIBUTION_PROMPT_VERSION, attributionSpeechSeconds, type AttributionRecord, type AttributionMethod, } from "../../common/lib/attribution"; import { loadAttribution } from "../../common/lib/attribution-server"; import { loadDiarization } from "../../common/lib/diarization-server"; import { selectClusterSamples } from "../../common/lib/attributionPrompt"; import type { AttributionSettings } from "../../common/lib/settings"; import { OLLAMA_DIGEST_APP_ID } from "../../common/lib/digestApps"; import { attributeOneVideo, type AttributeOneOutcome, } from "../../common/controller/attributeOne"; import { resolveAttributionTarget } from "../../common/controller/attributionTarget"; import { DIGEST_OVERLAP_CUES, maxCuesForContext, toHms, } from "../../common/lib/digestPrompt"; import { chunkCuesForContext } from "../../common/lib/transcriptWindow"; import { chunksPerAudioHour, readChunkCensus, sweepDays, MEASURED_SECONDS_PER_CHUNK, } from "../../common/controller/digestPlan"; import { isCuesJsonFresh, readNormalizedTranscript, } from "../../common/controller/normalizeTranscript"; import type { Cue } from "../../common/lib/vtt"; // The digest lane's own cost on the SAME model, context and eight videos — // round 2's 190 engine-seconds over 17 chunks. Printed beside the pilot's // number because "is attribution more expensive per chunk than a digest" is one // of the two questions this run answers. const DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE = 11.2; // --------------------------------------------------------------------------- // The sample // --------------------------------------------------------------------------- type PilotVideo = { slug: string; channelSlug: string; // The DIRECTORY name, which is not always the video id: the bake-off sample's // the-quartering-rumble entry is videoDir v6xl0vu against videoId v6ve4n0 (a // Rumble canonical-id reconciliation). Resolving the path off videoId returns // no-transcript and silently drops a stratum from the sample. videoDir: string; videoId: string; title: string; bucket?: string; durationSeconds: number; dir: string; }; // A video that made it far enough to be priced: cues on disk, chunk count known. type PricedVideo = PilotVideo & { cues: Cue[]; chunks: number; chunkRanges: { start: number; end: number }[]; }; async function resolveSample( spec: string, paths: Paths, ): Promise { if (spec === "bakeoff") { const file = path.join(paths.monorepoRoot, "plans", "bakeoff", "sample.json"); const raw = JSON.parse(await readFile(file, "utf8")) as { videos: { slug: string; channelSlug: string; videoId: string; videoDir: string; title: string; bucket: string; durationSeconds: number; }[]; }; return raw.videos.map((v) => ({ slug: v.slug, channelSlug: v.channelSlug, videoDir: v.videoDir, videoId: v.videoId, title: v.title, bucket: v.bucket, durationSeconds: v.durationSeconds, dir: path.join(paths.channelsDir, v.channelSlug, "data", v.videoDir), })); } if (spec.startsWith("channel:")) { const channelSlug = spec.slice("channel:".length); const dataDir = path.join(paths.channelsDir, channelSlug, "data"); const entries = await readdir(dataDir, { withFileTypes: true }); return entries .filter((e) => e.isDirectory()) .map((e) => ({ slug: `${channelSlug}/${e.name}`, channelSlug, videoDir: e.name, videoId: e.name, title: "", durationSeconds: 0, dir: path.join(dataDir, e.name), })) .sort((a, b) => a.videoDir.localeCompare(b.videoDir)); } if (spec.startsWith("video:")) { const rest = spec.slice("video:".length); const slash = rest.indexOf("/"); if (slash <= 0) throw new Error(`--sample video:/ expected, got "${spec}"`); const channelSlug = rest.slice(0, slash); const videoDir = rest.slice(slash + 1); return [ { slug: rest, channelSlug, videoDir, videoId: videoDir, title: "", durationSeconds: 0, dir: path.join(paths.channelsDir, channelSlug, "data", videoDir), }, ]; } throw new Error(`unknown --sample "${spec}" (bakeoff | channel: | video:/)`); } // Chunk boundaries EXACTLY as findTurns computes them — same chunker, same // overlap, same floor/ceil. Every cross-chunk metric rests on this agreeing with // the runner, which is why it is derived from the resolved config's numCtx // rather than from a literal 600, and why the scorer asserts its count against // provenance.chunks and reports null (never 0) on a mismatch. function chunkRangesFor(cues: Cue[], maxCues: number): { start: number; end: number }[] { return chunkCuesForContext(cues, { maxCues, overlapCues: DIGEST_OVERLAP_CUES }).map( (chunk) => { const start = Math.max(0, Math.floor(chunk[0].start)); const last = chunk[chunk.length - 1]; return { start, end: Math.max(start, Math.ceil(last.end || last.start)) }; }, ); } async function price(video: PilotVideo, maxCues: number): Promise { const { fresh, cuesPath } = await isCuesJsonFresh(video.dir); if (!fresh) return null; const transcript = await readNormalizedTranscript(cuesPath); const cues = transcript?.cues ?? []; if (!transcript || cues.length === 0) return null; const ranges = chunkRangesFor(cues, maxCues); const lastCue = cues[cues.length - 1]; return { ...video, videoId: video.videoId || transcript.id, title: video.title || transcript.title || video.videoId, durationSeconds: video.durationSeconds > 0 ? video.durationSeconds : Math.round(transcript.duration || lastCue.end || lastCue.start || 0), cues, chunks: ranges.length, chunkRanges: ranges, }; } // --------------------------------------------------------------------------- // Per-call timing, parsed out of the log stream digestApps already emits // --------------------------------------------------------------------------- // `ollama qwen2.5:7b: 4102 in / 218 out tokens in 5.0s wall (load 0.4s, prefill // 0.1s, decode 3.0s, engine 3.5s) (num_ctx 8192)` // // No new instrumentation: attributeOneVideo forwards the caller's onLog straight // into the app's run(), so one closure sees every call of the video it wraps. const CALL_RE = /^ollama (\S+): (\d+|\?) in \/ (\d+|\?) out tokens in ([\d.]+)s wall(?: \(([^)]*)\))?/; const PHASE_RE = /^(load|prefill|decode|engine) ([\d.]+)s$/; type CallTiming = { wallSeconds: number; loadSeconds?: number; prefillSeconds?: number; decodeSeconds?: number; engineSeconds?: number; inputTokens?: number; outputTokens?: number; }; function parseCall(line: string): CallTiming | null { const m = CALL_RE.exec(line); if (!m) return null; const out: CallTiming = { wallSeconds: Number(m[4]) }; if (m[2] !== "?") out.inputTokens = Number(m[2]); if (m[3] !== "?") out.outputTokens = Number(m[3]); // nsToMs OMITS a phase rather than zeroing it, and the whole parenthetical is // dropped when the engine reports nothing — so an absent phase stays absent // here too, and every engine-derived figure is averaged over its own // denominator rather than over the call count. for (const part of (m[5] ?? "").split(", ")) { const p = PHASE_RE.exec(part.trim()); if (!p) continue; const seconds = Number(p[2]); if (p[1] === "load") out.loadSeconds = seconds; else if (p[1] === "prefill") out.prefillSeconds = seconds; else if (p[1] === "decode") out.decodeSeconds = seconds; else out.engineSeconds = seconds; } return out; } // --------------------------------------------------------------------------- // Metrics // --------------------------------------------------------------------------- // Roles a label can carry that name nobody. isUselessSpeakerLabel already blocks // "Speaker 1" before the write, so the record cannot contain one; this measures // the WIDER set the runner permits. Anchored whole-string, not a prefix match: // "Host" is generic, "Host Jane Doe" is not. const GENERIC_LABEL_RE = /^(the\s+)?(co-?host|host|guest|caller|narrator|interviewer|moderator|panelist|announcer|commentator|reporter|audience|crowd|man|woman|male|female|person|voice|speaker)(\s*\d+)?$/i; type VideoMetrics = { labels: string[]; labelsPerVideo: number; // Null rather than 0 whenever the property could not be measured — a single // chunk has no seam to cross, and a chunk-count mismatch means the // reconstruction disagrees with the runner. Reporting 0 for "not measured" // is how a harness reports a perfect score for something that never ran. newLabelsPerChunk: number | null; singletonLabelRate: number | null; dominantSpeakerShare: number; talkShares: number[]; nearDuplicateLabelRate: number; nearDuplicatePairs: string[]; genericLabelRate: number; genericSecondsShare: number; nonAsciiLabelRate: number; coverageRate: number; maxGapSeconds: number; medianGapSeconds: number; markFreeChunks: number | null; segments: number; chunkReconstructionOk: boolean; labelFirstChunk: (number | null)[]; labelChunkCount: (number | null)[]; }; function normalizeForCompare(label: string): string { return label .trim() .replace(/\s+/g, " ") .replace(/^[^\p{L}\p{N}]+|[^\p{L}\p{N}]+$/gu, "") .toLowerCase(); } // Labels the runner's deliberately conservative normalizeLabel keeps apart but // that are obviously one person ("Jane" / "Jane Doe"). The runner is right to // refuse the merge — a merged speaker silently attributes words to the wrong // person — but a SCORER may flag, and this rate is the direct read on whether // that conservatism is set right. function nearDuplicate(a: string, b: string): boolean { const ta = new Set(normalizeForCompare(a).split(" ").filter(Boolean)); const tb = new Set(normalizeForCompare(b).split(" ").filter(Boolean)); if (ta.size === 0 || tb.size === 0) return false; const [small, big] = ta.size <= tb.size ? [ta, tb] : [tb, ta]; let shared = 0; for (const t of small) if (big.has(t)) shared++; if (shared === small.size) return true; // proper token-subset for (const t of small) if (t.length >= 4 && big.has(t)) return true; return false; } function median(values: number[]): number { if (values.length === 0) return 0; const s = [...values].sort((a, b) => a - b); const mid = Math.floor(s.length / 2); return s.length % 2 ? s[mid] : (s[mid - 1] + s[mid]) / 2; } function scoreRecord(record: AttributionRecord, video: PricedVideo): VideoMetrics { const ranges = video.chunkRanges; // The runner records what it actually chunked; if our reconstruction differs, // every chunk-indexed number below is meaningless and is reported as null. const recordedChunks = record.provenance.chunks; const reconstructionOk = recordedChunks === undefined || recordedChunks === ranges.length; const multiChunk = reconstructionOk && ranges.length > 1; const chunkOf = (t: number): number => { // The 40-cue overlap means a seam second sits in two chunks; first match is // the earlier one, which is the chunk that ESTABLISHED the label. for (let i = 0; i < ranges.length; i++) { if (t >= ranges[i].start && t <= ranges[i].end) return i; } return t < ranges[0].start ? 0 : ranges.length - 1; }; const chunksByLabel = new Map>(); for (const s of record.segments) { const set = chunksByLabel.get(s.speaker) ?? new Set(); set.add(chunkOf(s.start)); chunksByLabel.set(s.speaker, set); } const labels = record.speakers.map((s) => s.label); const seconds = record.speakers.map((s) => s.seconds ?? 0); const totalSeconds = seconds.reduce((a, b) => a + b, 0); const labelFirstChunk = record.speakers.map((s) => { const set = chunksByLabel.get(s.index); if (!reconstructionOk || !set || set.size === 0) return null; return Math.min(...set); }); const labelChunkCount = record.speakers.map((s) => { const set = chunksByLabel.get(s.index); if (!reconstructionOk || !set) return null; return set.size; }); // THE HEADLINE. Labels first appearing after chunk 0, per later chunk. Near 0 // means the cast stabilised; >= 1 means the model reinvented it every chunk // and the lane is not delivering the one property it exists for. const newLabelsPerChunk = multiChunk ? labelFirstChunk.filter((c) => c !== null && c > 0).length / (ranges.length - 1) : null; const singletonLabelRate = multiChunk && labels.length > 0 ? labelChunkCount.filter((c) => c === 1).length / labels.length : null; // THE DEGENERATE-CASE GUARD. One label at ~100% of talk time scores a perfect // 0 on the headline while being worthless, and the headline alone cannot tell // those apart. const dominantSpeakerShare = totalSeconds > 0 ? Math.max(...seconds) / totalSeconds : 0; const dupPairs: string[] = []; const involved = new Set(); for (let i = 0; i < labels.length; i++) { for (let j = i + 1; j < labels.length; j++) { if (!nearDuplicate(labels[i], labels[j])) continue; dupPairs.push(`${labels[i]} ~ ${labels[j]}`); involved.add(i); involved.add(j); } } const genericIdx = labels .map((l, i) => (GENERIC_LABEL_RE.test(l.trim()) ? i : -1)) .filter((i) => i >= 0); const genericSeconds = genericIdx.reduce((a, i) => a + seconds[i], 0); // COVERAGE, and what it is worth. Text-only marks are TILED — each runs to the // next change, the last to the end — so this is ~1.0 by construction. It is a // structural check (anything else is a bug), not a quality signal. The honest // coverage metric for this lane is markFreeChunks. const duration = video.durationSeconds || 0; const covered = attributionSpeechSeconds(record); const sorted = [...record.segments].sort((a, b) => a.start - b.start); const gaps: number[] = []; let cursor = 0; for (const s of sorted) { if (s.start > cursor) gaps.push(s.start - cursor); cursor = Math.max(cursor, s.end); } if (duration > cursor) gaps.push(duration - cursor); const markFreeChunks = reconstructionOk ? ranges.filter( (r) => !record.segments.some((s) => s.start >= r.start && s.start <= r.end), ).length : null; return { labels, labelsPerVideo: labels.length, newLabelsPerChunk, singletonLabelRate, dominantSpeakerShare, talkShares: totalSeconds > 0 ? seconds.map((s) => s / totalSeconds) : seconds.map(() => 0), nearDuplicateLabelRate: labels.length > 0 ? involved.size / labels.length : 0, nearDuplicatePairs: dupPairs, genericLabelRate: labels.length > 0 ? genericIdx.length / labels.length : 0, genericSecondsShare: totalSeconds > 0 ? genericSeconds / totalSeconds : 0, nonAsciiLabelRate: labels.length > 0 ? labels.filter((l) => /[^\x20-\x7E]/.test(l)).length / labels.length : 0, coverageRate: duration > 0 ? covered / duration : 0, maxGapSeconds: gaps.length ? Math.max(...gaps) : 0, medianGapSeconds: median(gaps), markFreeChunks, segments: record.segments.length, chunkReconstructionOk: reconstructionOk, labelFirstChunk, labelChunkCount, }; } // --------------------------------------------------------------------------- // The run // --------------------------------------------------------------------------- type VideoResult = { slug: string; bucket?: string; title: string; durationSeconds: number; expectedChunks: number; outcome: AttributeOneOutcome; // Wall time around attributeOneVideo — the realistic ceiling on this shared // box, and NOT the same quantity as the engine's own number. videoWallSeconds: number; calls: CallTiming[]; unparsedLogLines: number; recordedChunks?: number; recordedChunksOk?: number; warningsByCode: Record; metrics: VideoMetrics | null; // Diarized lane only. clustersOffered?: number; clustersNamed?: number; confidences?: number[]; }; function boxState(): Record { const load = os.loadavg(); return { loadavg: load.map((n) => Number(n.toFixed(2))), freeMemGb: Number((os.freemem() / 2 ** 30).toFixed(2)), totalMemGb: Number((os.totalmem() / 2 ** 30).toFixed(2)), }; } function sum(values: number[]): number { return values.reduce((a, b) => a + b, 0); } async function runVideo( video: PricedVideo, method: AttributionMethod, settings: AttributionSettings, paths: Paths, force: boolean, log: (m: string) => void, ): Promise { const calls: CallTiming[] = []; let unparsed = 0; const onLog = (line: string) => { const call = parseCall(line); if (call) calls.push(call); // Only the ollama call lines are parseable; the runner's own progress lines // are not, and are not counted as drift. else if (line.startsWith("ollama ")) unparsed++; }; const started = Date.now(); const outcome = await attributeOneVideo({ paths, videoDir: video.dir, videoId: video.videoId, channelSlug: video.channelSlug, method, settings, force, onLog, }); const videoWallSeconds = (Date.now() - started) / 1000; const record = await loadAttribution(video.dir); const warningsByCode: Record = {}; for (const w of record?.warnings ?? []) { warningsByCode[w.code] = (warningsByCode[w.code] ?? 0) + 1; } const result: VideoResult = { slug: video.slug, ...(video.bucket ? { bucket: video.bucket } : {}), title: video.title, durationSeconds: video.durationSeconds, expectedChunks: video.chunks, outcome, videoWallSeconds, calls, unparsedLogLines: unparsed, ...(record?.provenance.chunks !== undefined ? { recordedChunks: record.provenance.chunks, recordedChunksOk: record.provenance.chunksOk } : {}), warningsByCode, metrics: record && outcome === "attributed" ? scoreRecord(record, video) : null, }; if (method === "diarized") { const diarization = await loadDiarization(video.dir); if (diarization) { result.clustersOffered = selectClusterSamples(diarization.turns, video.cues).length; } result.clustersNamed = record?.speakers.length ?? 0; result.confidences = (record?.speakers ?? []) .map((s) => s.confidence) .filter((c): c is number => typeof c === "number"); } const m = result.metrics; log( ` ${video.slug}${video.bucket ? ` [${video.bucket}]` : ""} ${video.chunks} chunk(s) → ` + `${outcome}` + (m ? `, ${m.labelsPerVideo} label(s), ${m.segments} segment(s), ` + `new/chunk ${m.newLabelsPerChunk === null ? "n/a" : m.newLabelsPerChunk.toFixed(2)}, ` + `top share ${(m.dominantSpeakerShare * 100).toFixed(0)}%` : "") + `, ${videoWallSeconds.toFixed(0)}s wall` + (calls.length ? ` / ${sum(calls.map((c) => c.engineSeconds ?? 0)).toFixed(0)}s engine` : ""), ); return result; } // --------------------------------------------------------------------------- // Reporting // --------------------------------------------------------------------------- type CostSummary = { videosMeasured: number; chunks: number; chunksOk: number; calls: number; callsWithEngineTiming: number; audioSeconds: number; chunksPerAudioHour: number; secondsPerChunkWall: number | null; secondsPerChunkCallWall: number | null; secondsPerChunkEngine: number | null; secondsPerChunkEngineExLoad: number | null; meanLoadSeconds: number | null; decodeTokensPerSecond: number | null; unparsedLogLines: number; }; // Throughput is computed ONLY over videos the run actually attributed: an // already-exists short-circuits in milliseconds and would otherwise report an // absurd chunk rate. function summarizeCost(results: VideoResult[]): CostSummary { const measured = results.filter((r) => r.outcome === "attributed"); const calls = measured.flatMap((r) => r.calls); const withEngine = calls.filter((c) => c.engineSeconds !== undefined); const chunks = sum(measured.map((r) => r.recordedChunks ?? r.expectedChunks)); const chunksOk = sum(measured.map((r) => r.recordedChunksOk ?? r.calls.length)); const audioSeconds = sum(measured.map((r) => r.durationSeconds)); const wall = sum(measured.map((r) => r.videoWallSeconds)); const decodeSeconds = sum(calls.map((c) => c.decodeSeconds ?? 0)); const outTokens = sum(calls.map((c) => c.outputTokens ?? 0)); const withLoad = calls.filter((c) => c.loadSeconds !== undefined); return { videosMeasured: measured.length, chunks, chunksOk, calls: calls.length, callsWithEngineTiming: withEngine.length, audioSeconds, chunksPerAudioHour: chunksPerAudioHour(chunks, audioSeconds), secondsPerChunkWall: chunks > 0 ? wall / chunks : null, secondsPerChunkCallWall: calls.length ? sum(calls.map((c) => c.wallSeconds)) / calls.length : null, secondsPerChunkEngine: withEngine.length ? sum(withEngine.map((c) => c.engineSeconds ?? 0)) / withEngine.length : null, secondsPerChunkEngineExLoad: withEngine.length ? sum(withEngine.map((c) => (c.prefillSeconds ?? 0) + (c.decodeSeconds ?? 0))) / withEngine.length : null, meanLoadSeconds: withLoad.length ? sum(withLoad.map((c) => c.loadSeconds ?? 0)) / withLoad.length : null, decodeTokensPerSecond: decodeSeconds > 0 ? outTokens / decodeSeconds : null, unparsedLogLines: sum(results.map((r) => r.unparsedLogLines)), }; } function pct(n: number): string { return `${(n * 100).toFixed(0)}%`; } function fixed(n: number | null, digits = 2): string { return n === null ? "n/a" : n.toFixed(digits); } // Verbatim transcript at a speaker's longest segments. WITHOUT this nobody can // judge whether a label is right, and no aggregate metric substitutes for // reading it — which is why the Markdown artifact, not the JSON, is the point of // the exercise. function excerpt(cues: Cue[], start: number, end: number, maxChars = 420): string[] { const lines: string[] = []; let chars = 0; for (const c of cues) { if (c.start < start) continue; if (c.start > end) break; const text = c.text.trim().replace(/\s+/g, " "); if (!text) continue; lines.push(`[${toHms(c.start)}] ${text}`); chars += text.length; if (chars >= maxChars) break; } return lines; } function qualityMarkdown( label: string, method: AttributionMethod, results: VideoResult[], priced: Map, records: Map, ): string { const out: string[] = []; out.push(`# Attribution pilot — ${label} (${method}): what the labels actually say`); out.push(""); out.push( "Read this file, not the metrics, to answer *are the labels right*. Each speaker", "is shown with its talk-time share and the verbatim transcript at its longest", "segments — the same text, in the same `[HH:MM:SS] line` form, the model saw.", ); out.push(""); for (const r of results) { const video = priced.get(r.slug); const record = records.get(r.slug); out.push(`## ${r.slug}${r.bucket ? ` — ${r.bucket}` : ""}`); out.push(""); out.push(`**${r.title || "(untitled)"}**`); out.push(""); out.push( `${toHms(r.durationSeconds)} · ${r.expectedChunks} chunk(s) · outcome \`${r.outcome}\`` + (r.recordedChunks !== undefined ? ` · ${r.recordedChunksOk}/${r.recordedChunks} chunk(s) ok` : ""), ); out.push(""); if (!record || !video || !r.metrics) { out.push("_No record was written for this video._"); out.push(""); continue; } const m = r.metrics; for (let i = 0; i < record.speakers.length; i++) { const speaker = record.speakers[i]; const share = m.talkShares[i] ?? 0; const first = m.labelFirstChunk[i]; const present = m.labelChunkCount[i]; out.push( `### ${speaker.label} — ${pct(share)} of attributed time` + (speaker.confidence !== undefined ? `, confidence ${speaker.confidence.toFixed(2)}` : "") + (first !== null ? `, first seen in chunk ${first + 1}` : "") + (present !== null ? `, present in ${present}/${r.expectedChunks} chunk(s)` : ""), ); out.push(""); const mine = record.segments .filter((s) => s.speaker === speaker.index) .sort((a, b) => b.end - b.start - (a.end - a.start)) .slice(0, 3) .sort((a, b) => a.start - b.start); if (mine.length === 0) { out.push("_No segments._"); out.push(""); continue; } for (const seg of mine) { out.push(`_${toHms(seg.start)} – ${toHms(seg.end)}_`); out.push(""); out.push("```"); const lines = excerpt(video.cues, seg.start, Math.min(seg.end, seg.start + 90)); out.push(...(lines.length ? lines : ["(no cue text in range)"])); out.push("```"); out.push(""); } } if (m.nearDuplicatePairs.length) { out.push(`Near-duplicate label pairs: ${m.nearDuplicatePairs.map((p) => `\`${p}\``).join(", ")}`); out.push(""); } out.push("- [ ] labels are right"); out.push("- [ ] a label is wrong (which: ____)"); out.push("- [ ] a name appears that the transcript never states"); out.push("- [ ] one person is split across two labels"); out.push("- [ ] two people are merged into one label"); out.push(""); } return out.join("\n"); } function reportMarkdown(args: { label: string; method: AttributionMethod; sample: string; model: string; numCtx?: number; maxCues: number; results: VideoResult[]; cost: CostSummary; census: { chunks: number; videos: number; chunksPerAudioHour: number; estimated: number; statsSchemaStale: boolean } | null; boxStart: Record; boxEnd: Record; startedAt: string; }): string { const { results, cost } = args; const out: string[] = []; out.push(`# Attribution pilot — ${args.label}`); out.push(""); out.push( `Lane **${args.method}** · sample \`${args.sample}\` · ${results.length} video(s) · ` + `model \`${args.model}\` @ numCtx ${args.numCtx ?? "default"} (maxCues ${args.maxCues}) · ${args.startedAt}`, ); out.push(""); const outcomes: Record = {}; for (const r of results) outcomes[r.outcome] = (outcomes[r.outcome] ?? 0) + 1; out.push(`Outcomes: ${Object.entries(outcomes).map(([k, v]) => `${k} ${v}`).join(", ")}`); out.push(""); if (args.method === "text-only") { out.push("## Cross-chunk identity — the property the lane exists for"); out.push(""); out.push("| video | chunks | labels | new labels/chunk | singleton rate | top share | near-dup | generic |"); out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |"); for (const r of results) { const m = r.metrics; out.push( `| ${r.slug} | ${r.expectedChunks} | ${m?.labelsPerVideo ?? "—"} | ${ m ? fixed(m.newLabelsPerChunk) : "—" } | ${m ? (m.singletonLabelRate === null ? "n/a" : pct(m.singletonLabelRate)) : "—"} | ${ m ? pct(m.dominantSpeakerShare) : "—" } | ${m ? pct(m.nearDuplicateLabelRate) : "—"} | ${m ? pct(m.genericLabelRate) : "—"} |`, ); } out.push(""); const scored = results.map((r) => r.metrics).filter((m): m is VideoMetrics => m !== null); const withSeams = scored.filter((m) => m.newLabelsPerChunk !== null); if (withSeams.length) { out.push( `**Headline** — new labels per chunk after the first, over ${withSeams.length} multi-chunk video(s): ` + `mean ${fixed(sum(withSeams.map((m) => m.newLabelsPerChunk ?? 0)) / withSeams.length)}, ` + `max ${fixed(Math.max(...withSeams.map((m) => m.newLabelsPerChunk ?? 0)))}.`, ); out.push(""); out.push( "Read it WITH the top-share column: one label at ~100% scores a perfect 0 here", "and is degenerate, not good.", ); out.push(""); } const degenerate = scored.filter((m) => m.labelsPerVideo === 1); out.push( `Single-label (degenerate-risk) videos: ${degenerate.length} of ${scored.length}. ` + `Videos whose chunk reconstruction disagreed with the record: ${ scored.filter((m) => !m.chunkReconstructionOk).length }.`, ); out.push(""); out.push("## Coverage and label quality"); out.push(""); out.push("| video | segments | mark-free chunks | coverage | max gap | median gap | non-ascii labels |"); out.push("| --- | ---: | ---: | ---: | ---: | ---: | ---: |"); for (const r of results) { const m = r.metrics; out.push( `| ${r.slug} | ${m?.segments ?? "—"} | ${ m ? (m.markFreeChunks === null ? "n/a" : m.markFreeChunks) : "—" } | ${m ? pct(m.coverageRate) : "—"} | ${m ? toHms(Math.round(m.maxGapSeconds)) : "—"} | ${ m ? toHms(Math.round(m.medianGapSeconds)) : "—" } | ${m ? pct(m.nonAsciiLabelRate) : "—"} |`, ); } out.push(""); out.push( "Coverage is ~1.0 BY CONSTRUCTION for this lane — marks are tiled, each running", "to the next change — so it is a structural check, not a quality signal. The", "honest coverage metric here is mark-free chunks.", ); out.push(""); } else { out.push("## Diarized lane — the claim under test is ONE call per video"); out.push(""); out.push("| video | calls | clusters offered | clusters named | confidences |"); out.push("| --- | ---: | ---: | ---: | --- |"); for (const r of results) { out.push( `| ${r.slug} | ${r.calls.length} | ${r.clustersOffered ?? "—"} | ${r.clustersNamed ?? "—"} | ${ (r.confidences ?? []).map((c) => c.toFixed(2)).join(", ") || "—" } |`, ); } out.push(""); const calls = sum(results.map((r) => r.calls.length)); out.push( `**Calls per video: ${(calls / Math.max(1, results.length)).toFixed(2)}** ` + `(${calls} call(s) over ${results.length} video(s)). The claim holds iff this is 1.00.`, ); out.push(""); } const warnings: Record = {}; for (const r of results) { for (const [code, n] of Object.entries(r.warningsByCode)) { warnings[code] = (warnings[code] ?? 0) + n; } } out.push( `Warnings by code: ${ Object.keys(warnings).length ? Object.entries(warnings).map(([k, v]) => `${k} ${v}`).join(", ") : "none" }`, ); out.push(""); out.push("## Cost"); out.push(""); out.push( `Measured over the ${cost.videosMeasured} video(s) this run actually attributed ` + `(${cost.chunksOk}/${cost.chunks} chunk(s) ok, ${cost.calls} call(s), ` + `${cost.callsWithEngineTiming} with engine timing). Sample density ` + `**${cost.chunksPerAudioHour.toFixed(2)} chunks/audio-hour** — the corpus census reads ` + `${args.census ? args.census.chunksPerAudioHour.toFixed(2) : "2.46"}.`, ); out.push(""); out.push("| quantity | value |"); out.push("| --- | ---: |"); out.push(`| s/chunk, wall (video wall ÷ chunks) | ${fixed(cost.secondsPerChunkWall, 1)} |`); out.push(`| s/chunk, per-call wall | ${fixed(cost.secondsPerChunkCallWall, 1)} |`); out.push(`| s/chunk, engine | ${fixed(cost.secondsPerChunkEngine, 1)} |`); out.push(`| s/chunk, engine excl. model load | ${fixed(cost.secondsPerChunkEngineExLoad, 1)} |`); out.push(`| mean model-load s/call | ${fixed(cost.meanLoadSeconds, 2)} |`); out.push(`| decode tokens/s | ${fixed(cost.decodeTokensPerSecond, 1)} |`); out.push( `| digest chapters baseline, s/chunk engine | ${DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE.toFixed(1)} |`, ); out.push(""); if (args.method === "text-only" && args.census) { const idle = cost.secondsPerChunkEngineExLoad; const contended = cost.secondsPerChunkWall; out.push("## Corpus projection"); out.push(""); out.push( `Census: ${args.census.videos.toLocaleString()} transcribed video(s), ` + `**${args.census.chunks.toLocaleString()} chunks** at maxCues ${args.maxCues}` + (args.census.estimated ? `, ${args.census.estimated} estimated` : "") + (args.census.statsSchemaStale ? " — **STATS SCHEMA STALE, projection suspect**" : "") + ".", ); out.push(""); out.push( `**${idle === null ? "n/a" : sweepDays(args.census.chunks, idle).toFixed(1)} days idle ` + `to ${contended === null ? "n/a" : sweepDays(args.census.chunks, contended).toFixed(1)} days contended**` + ` — from n = ${cost.videosMeasured} video(s) / ${cost.chunks} chunk(s), one box, one hour.`, ); out.push(""); out.push( "This is a SECOND sweep on the same 8 GB card, additive to a digest sweep that", "has completed 0.17% of its own ~194k calls, with nothing arbitrating between", "them. One lane, no parallelism.", ); out.push(""); } out.push("## Box"); out.push(""); out.push(`Start: \`${JSON.stringify(args.boxStart)}\``); out.push(""); out.push(`End: \`${JSON.stringify(args.boxEnd)}\``); out.push(""); if (cost.unparsedLogLines > 0) { out.push( `**${cost.unparsedLogLines} unparsed engine log line(s) — every timing number above is SUSPECT.**`, ); out.push(""); } out.push("## Units"); out.push(""); out.push( "- The unit of text-only work is the **chunk** (one model call). Seconds-per-", " audio-hour is not a unit here: chunk density varies 4× across this corpus, and", " pricing in it is the error behind a retracted throughput headline. This report", " does not print one.", "- The unit of diarized work is **calls per video**.", "- Wall and engine are different quantities and are never substituted for each", " other. Wall is the realistic ceiling on this shared box; engine excl. load is", " the floor.", "- Every rate carries its denominator. At n ≈ 8 videos a bare percentage would be", " a lie of precision.", ); out.push(""); return out.join("\n"); } // --------------------------------------------------------------------------- async function main(): Promise { const flags = parseFlags(process.argv.slice(2)); const paths = getPaths(); const sampleSpec = flags.sample ?? "bakeoff"; const method: AttributionMethod = flags.method === "diarized" ? "diarized" : "text-only"; const dryRun = flags["dry-run"] === "true"; const force = flags.force === "true"; const asJson = flags.json === "true"; const label = flags.label ?? `${method}-${sampleSpec.replace(/[^a-zA-Z0-9]+/g, "-")}`; // THE LANE IS ENABLED IN MEMORY. settings.json has no `attribution` key at // all, so without this override every video returns "disabled" and the run // reports a flawless zero-cost hour. Nothing here is persisted. const pilotSettings: AttributionSettings = { enabled: true, appId: OLLAMA_DIGEST_APP_ID, model: "", diarizedEnabled: method === "diarized", textOnlyEnabled: method === "text-only", promptVersion: ATTRIBUTION_PROMPT_VERSION, }; const { app, config, modelRequested } = resolveAttributionTarget(method, pilotSettings); const maxCues = maxCuesForContext(config.numCtx); const videos = await resolveSample(sampleSpec, paths); const priced: PricedVideo[] = []; const skipped: string[] = []; for (const v of videos) { const p = await price(v, maxCues); if (p) priced.push(p); else skipped.push(v.slug); } const totalChunks = sum(priced.map((p) => p.chunks)); const totalAudio = sum(priced.map((p) => p.durationSeconds)); const census = await readChunkCensus(paths, maxCues); console.log( `Attribution pilot — lane ${method}, sample ${sampleSpec}, app ${app.id}, ` + `model ${modelRequested}, numCtx ${config.numCtx ?? "default"} → maxCues ${maxCues}`, ); console.log(`Settings override (in memory only): ${JSON.stringify(pilotSettings)}`); console.log(""); for (const p of priced) { console.log( ` ${p.slug}${p.bucket ? ` [${p.bucket}]` : ""} ${toHms(p.durationSeconds)} ` + `${p.cues.length} cues → ${p.chunks} chunk(s)`, ); } if (skipped.length) console.log(` skipped (no usable transcript): ${skipped.join(", ")}`); console.log(""); const projectedMinutes = (totalChunks * MEASURED_SECONDS_PER_CHUNK) / 60; console.log( `${priced.length} video(s), ${totalChunks} chunk(s), ${(totalAudio / 3600).toFixed(1)} audio-hour(s), ` + `${chunksPerAudioHour(totalChunks, totalAudio).toFixed(2)} chunks/audio-hour. ` + `At the digest lane's measured ${MEASURED_SECONDS_PER_CHUNK}s/chunk that is ` + `~${projectedMinutes.toFixed(0)} minute(s).`, ); if (census) { console.log( `Census: ${census.videos.toLocaleString()} video(s) / ${census.chunks.toLocaleString()} chunk(s) ` + `/ ${census.chunksPerAudioHour.toFixed(2)} chunks per audio-hour` + (census.statsSchemaStale ? " — STATS SCHEMA STALE" : ""), ); } console.log(""); if (dryRun) { console.log("--dry-run: priced only, no model call and no write."); return; } const boxStart = boxState(); const startedAt = new Date().toISOString(); const results: VideoResult[] = []; const pricedBySlug = new Map(priced.map((p) => [p.slug, p])); const records = new Map(); for (const video of priced) { const r = await runVideo(video, method, pilotSettings, paths, force, (m) => console.log(m)); results.push(r); const record = await loadAttribution(video.dir); if (record) records.set(video.slug, record); } const boxEnd = boxState(); const cost = summarizeCost(results); const report = { version: 1 as const, label, method, sample: sampleSpec, startedAt, finishedAt: new Date().toISOString(), engine: { appId: app.id, modelRequested, numCtx: config.numCtx, maxCues, overlapCues: DIGEST_OVERLAP_CUES, promptVersion: ATTRIBUTION_PROMPT_VERSION, }, settingsOverride: pilotSettings, skipped, box: { start: boxStart, end: boxEnd }, census, cost, baseline: { digestSecondsPerChunkEngine: DIGEST_BASELINE_SECONDS_PER_CHUNK_ENGINE }, videos: results, }; const outDir = path.join(paths.monorepoRoot, "plans", "attribution-pilot"); await mkdir(outDir, { recursive: true }); const jsonPath = path.join(outDir, `${label}.json`); const mdPath = path.join(outDir, `${label}.md`); const qualityPath = path.join(outDir, `${label}-quality.md`); await writeFile(jsonPath, JSON.stringify(report, null, 2) + "\n"); await writeFile( mdPath, reportMarkdown({ label, method, sample: sampleSpec, model: modelRequested, numCtx: config.numCtx, maxCues, results, cost, census, boxStart, boxEnd, startedAt, }), ); await writeFile(qualityPath, qualityMarkdown(label, method, results, pricedBySlug, records)); console.log(""); if (asJson) { console.log(JSON.stringify(report, null, 2)); } else { console.log( reportMarkdown({ label, method, sample: sampleSpec, model: modelRequested, numCtx: config.numCtx, maxCues, results, cost, census, boxStart, boxEnd, startedAt, }), ); } console.log(`Wrote ${jsonPath}`); console.log(`Wrote ${mdPath}`); console.log(`Wrote ${qualityPath}`); // A run in which NOTHING was attributed and nothing was already fresh is the // silent-no-op failure: the lane was off, or every video lacked a transcript. // It must not exit 0 looking like a clean measurement. const productive = results.filter( (r) => r.outcome === "attributed" || r.outcome === "already-exists", ).length; if (productive === 0) { console.error("No video was attributed and none was already fresh — nothing was measured."); process.exitCode = 1; } } main().catch((err) => { console.error(err); process.exit(1); });