// Client-safe types, constants and PURE merge helpers for the per-video AI // digest (chapters + topic tags). No node-only imports — the editor's video // panel is a "use client" file that pulls this in for the review UI. Server-only // disk I/O lives in digest-server.ts. Same split, for the same reason, as // doNotClean.ts / doNotClean-server.ts. // // TWO SIDECARS PER VIDEO DIR, and the split is the whole point: // // ai-digest.json machine-generated; freely overwritten by a re-run // ai-digest.overrides.json human-authored; regeneration NEVER writes it // // A sweep over 74k transcripts takes weeks, so a redo is unaffordable and human // corrections must survive one. Keeping the two in separate files means no merge // bug in the generator can destroy hand-written work: the generator only ever // opens the machine file. Readers compose the two with effectiveDigest(). // // NEVER rename these to `transcript..` — SUB_FILE_RE in videoStatus.ts // claims such a file as a subtitle track; sidecar() throws at declaration. import type { FieldDocs } from "./fieldDocs"; export const DIGEST_FILENAME = "ai-digest.json"; export const DIGEST_OVERRIDES_FILENAME = "ai-digest.overrides.json"; // Shape of ai-digest.json itself. Bumping this invalidates every digest on // disk, so it changes only when the FILE LAYOUT changes — not when a prompt // changes (that is promptVersion, which invalidates per section). export const DIGEST_SCHEMA_VERSION = 1; export const DIGEST_OVERRIDES_VERSION = 1; // Which engine lane produced a section. "local-gpu" is the default and carries // the corpus; "remote-api" is the opt-in metered overflow. // // The lane type, the app ids and the per-app config live HERE rather than in // digestApps.ts because settings.ts needs them and digestApps.ts imports execa — // a node-only dependency that must never be reachable from a client bundle. // digestApps.ts re-exports them so the registry still reads as one unit. export type DigestLane = "local-gpu" | "remote-api"; // Default ollama context, and the chunk size sized to it. Both live HERE rather // than in digestApps.ts / digestPrompt.ts so the pure prompt module can size a // chunk against them without reaching a module that imports execa, and so the // identity helper below can compare against the default without a cycle. // // 8192, ON MEASUREMENT, not on comfort. Round 2 of the bake-off // (plans/bakeoff/round2.md) ran qwen2.5:7b at both sizes over the same 6.96 // audio-hours, chunk-local in both cases: // // @16384 11.1% zero-yield, 9.34 chapters/h, worst gap 1:27:48, 25.1 days // @8192 11.8% zero-yield, 13.37 chapters/h, worst gap 0:24:13, 24.2 days // // Halving the window nearly halves the worst coverage gap and raises the // segmentation rate by 43% at no throughput cost — twice as many calls each // carry half the prompt, so the projected sweep is if anything shorter. 16k was // the cautious choice and it measured worse. // // A 7B model at 8k is also well within an 8 GB card (qwen2.5:7b KV cache // ~56 KB/token -> ~0.45 GB at 8k, atop 4.7 GB of weights). digestPrompt.ts // re-exports the cue count as DIGEST_MAX_CUES_PER_CHUNK and derives other sizes // from it; maxCuesForContext() scales the cue count with whatever numCtx an app // is actually configured for, so these two must stay in proportion (~10 // tokens/cue: 600 cues ~= 6k tokens of transcript inside an 8k window, leaving // room for the prompt and the response). export const DEFAULT_DIGEST_NUM_CTX = 8192; export const DEFAULT_DIGEST_MAX_CUES_PER_CHUNK = 600; // How the transcript markers inside ONE chunk are numbered, and therefore what // the model is asked to copy. // // absolute the chunk's cues carry their real video times, and the prompt // states the chunk's real range ("01:31:43 to 02:16:09"). // chunk-local the chunk is re-based to 00:00:00 and the prompt states // "00:00:00 to 00:44:26". The parser adds the offset back before // any guard runs, so the range clamp still checks the chunk's // REAL range and warnings still report real video times. // // This exists because of a measured failure, not a hunch. On a 2.3 h video // (community-notes/v2chrch, 3505 cues → 3 chunks) the third chunk — range // 01:31:43–02:16:09 — came back with nine starts of 00:00:00, 00:03:54, // 00:12:26 …: the model had reverted to counting from zero. The per-chunk clamp // caught all nine, which is exactly its job, but a caught error is still a lost // chunk, and >4 h videos are 8.2% of the corpus by count and 46% of its tokens. // chunk-local removes the large offset the model has to hold. Which mode is // actually better was a BAKE-OFF QUESTION (plans/tools/digest-bakeoff.ts), which // is why both were shipped rather than one being pre-applied as a fix. // // IT HAS BEEN ANSWERED. Round 2 (plans/bakeoff/round2.md), qwen2.5:7b@8192 over // the same 6.96 audio-hours: // // absolute 29.4% zero-yield chunks, 10.35 chapters/h, worst gap 1:05:16, // 47.5% of items rejected (65 of them out-of-range) // chunk-local 11.8% zero-yield chunks, 13.37 chapters/h, worst gap 0:24:13, // 19.1% rejected (13 out-of-range) // // gemma2:9b ranks the two the same way, so the effect is not model-specific. // chunk-local is now the DEFAULT. Changing it changes generated output, so it // was paired with a PROMPT_VERSION bump (digestPrompt.ts) rather than left to // digestPromptVariant's relative-to-default derivation — see the note there. export const DIGEST_TIMESTAMP_MODES = ["absolute", "chunk-local"] as const; export type DigestTimestampMode = (typeof DIGEST_TIMESTAMP_MODES)[number]; export const DEFAULT_DIGEST_TIMESTAMP_MODE: DigestTimestampMode = "chunk-local"; export function isDigestTimestampMode(v: unknown): v is DigestTimestampMode { return ( typeof v === "string" && (DIGEST_TIMESTAMP_MODES as readonly string[]).includes(v) ); } // The ONE place the recorded promptVariant string is derived, so every writer // (the controller, the bake-off harness) produces the same identity for the same // configuration. // // timestampMode is folded in rather than left to the operator to remember: a // knob that changes the output but not the identity would let a re-run under a // different mode skip every video as "fresh", which is precisely the failure // this identity exists to prevent. The default configuration maps to `undefined` // so pre-existing records stay fresh. export function digestPromptVariant(input: { promptVariant?: string; timestampMode?: DigestTimestampMode; // Cues per chunk. Folded in for the same reason as timestampMode, and on the // same evidence: halving it took qwen2.5:7b from 9.34 to 13.37 chapters/hour // and its worst coverage gap from 1:27:48 to 24:13 on a 3.5 h video. A knob // that changes the output that much cannot sit outside the identity, or a // re-run at a new chunk size would skip the whole corpus as "fresh". maxCues?: number; }): string | undefined { const parts: string[] = []; const named = input.promptVariant?.trim(); if (named) parts.push(named); if (input.timestampMode && input.timestampMode !== DEFAULT_DIGEST_TIMESTAMP_MODE) { parts.push(input.timestampMode); } // The default chunk size contributes NOTHING, so every record written before // this field existed still compares equal. if (input.maxCues && input.maxCues !== DEFAULT_DIGEST_MAX_CUES_PER_CHUNK) { parts.push(`c${Math.floor(input.maxCues)}`); } return parts.length > 0 ? parts.join("+") : undefined; } export const OLLAMA_DIGEST_APP_ID = "ollama-direct"; export const CLAUDE_DIGEST_APP_ID = "claude-code"; export const DEFAULT_DIGEST_APP_ID = OLLAMA_DIGEST_APP_ID; // Per-app configuration persisted under settings.digest.apps[id]. Every field is // optional; an app falls back to its own defaults. // Each field is documented in DIGEST_APP_CONFIG_FIELD_DOCS below (rendered into SETTINGS.md). export type DigestAppConfig = { bin?: string; baseUrl?: string; model?: string; numCtx?: number; temperature?: number; think?: boolean; timeoutMs?: number; }; export const DIGEST_APP_CONFIG_FIELD_DOCS: FieldDocs = { bin: "Binary path/name override (process-based apps only).", baseUrl: "Base URL override (HTTP apps only).", model: "Model id, e.g. \"qwen2.5:7b\" or \"haiku\".", numCtx: "Context window in tokens. MUST reach the engine explicitly for ollama:" + " its 4096 default silently truncates the input and the model then " + "summarizes whatever fragment survived — measured, and the single " + "easiest way to get quietly-wrong output at scale.", temperature: "Sampling temperature. 0 for a structured extraction task.", think: "Reasoning-model toggle (ollama's top-level `think`). Only sent when " + "set, so a model that does not support thinking is never handed a field" + " it rejects.\n\n" + "It matters for throughput, not correctness: measured on this box, " + "qwen3:8b with thinking on spends most of its output budget on a " + "`thinking` block before the JSON body the schema constrains. For an " + "extraction task with a pinned schema that reasoning buys little and " + "costs a multiple of the tokens, and tokens are what a multi-week sweep" + " is priced in.", timeoutMs: "Per-request wall-clock ceiling (ms). A wedged engine must not stall a " + "sweep.", }; // Whether an engine-reported model resolution looks like a DIFFERENT model // rather than a benign tag completion. "qwen2.5" resolving to "qwen2.5:7b" or // "qwen2.5:latest" is ollama filling in a tag; "qwen2.5:7b" coming back as // "llama3:8b" means the endpoint served the wrong weights. Freshness compares // the REQUESTED string on purpose (see AttributionProvenance.modelRequested), // so a wrong resolution would not invalidate anything — which is exactly why // the runners log it loudly instead. export function isModelResolutionSuspicious( requested: string, reported: string, ): boolean { if (!reported || !requested || reported === requested) return false; if (reported === `${requested}:latest`) return false; if (!requested.includes(":") && reported.startsWith(`${requested}:`)) { return false; } return true; } // The two generated sections. Per-SECTION provenance (not per-file) because the // controller does a read-modify-write merge, so one video can legitimately hold // local chapters and metered tags. export const DIGEST_SECTION_KINDS = ["chapters", "tags"] as const; export type DigestSectionKind = (typeof DIGEST_SECTION_KINDS)[number]; export function isDigestSectionKind(v: unknown): v is DigestSectionKind { return ( typeof v === "string" && (DIGEST_SECTION_KINDS as readonly string[]).includes(v) ); } // Who decided this item. Every AI decision is marked as such and is // human-overridable — the same mechanism Phase 9 attribution will reuse rather // than inventing a second one. export type DecidedBy = "ai" | "human"; // A chapter: a titled moment. `start` is SECONDS (the cue unit — see vtt.ts), // snapped to a real cue boundary by the parser. `clock` is the HH:MM:SS form the // model emitted, kept for auditing what the model actually said. export type DigestChapter = { // Stable id so an override can shadow exactly one generated item. Derived from // the snapped start (see chapterId) — deterministic, no Date.now/random. id: string; start: number; clock: string; title: string; decidedBy: DecidedBy; // Defaults true. An override sets false to SUPPRESS a generated item without // deleting it (the searchAliases.ts `enabled` idiom), so a regeneration that // re-emits the same item does not resurrect something a human rejected. enabled?: boolean; }; export type DigestTag = { id: string; tag: string; decidedBy: DecidedBy; enabled?: boolean; }; export type DigestItem = DigestChapter | DigestTag; // Why an item or a whole chunk was rejected. Recorded, never silently dropped — // a multi-week sweep is only tunable if its failures are inspectable, and the // Phase 11a review queue is built entirely on this array. export type DigestWarningCode = | "malformed-timestamp" | "out-of-range" | "language-drift" | "non-monotonic" | "seam-duplicate" | "empty-title" | "empty-output" | "parse-failed" | "chunk-failed"; export type DigestWarning = { code: DigestWarningCode; // Which section's generation produced it, so warnings stay attributable after // a read-modify-write merge of two separately-generated sections. section: DigestSectionKind; // Zero-based index of the transcript chunk that produced it, when applicable. chunk?: number; // The offending value, verbatim, so a prompt regression is diagnosable from // the artifact alone. value?: string; detail?: string; }; // What a regeneration compares against to decide "skip". This is what makes a // re-run targeted instead of a second multi-week sweep. export type DigestProvenance = { appId: string; // The model that ACTUALLY ran, as the engine reported it (e.g. "qwen2.5:7b"). // The audit record. model: string; // What the config ASKED for (e.g. "qwen2.5"). Freshness compares this, not // `model`: an alias resolving to a full tag is not a model change, and treating // it as one would re-run the entire corpus. Older records lack it, so readers // fall back to `model`. modelRequested?: string; lane: DigestLane; generatedAt: string; // Bumped when the prompt/schema changes. Invalidates this section only. promptVersion: number; // Which prompt SHAPE produced this section, when it was not the default one. // promptVersion answers "has the prompt changed since?"; this answers "which // of several concurrently-supported shapes was used?" — the two are different // questions and a bake-off needs both. Absent means the default shape // (timestampMode "absolute", no named variant), so every record written before // this field existed stays valid: isSectionFresh treats absent-on-both as // equal. Same "one identity field per thing that can change the output" rule // the rest of this record already follows. promptVariant?: string; // Hash of the channel-context inputs. Plumbed from the start even while // context files are empty, so adding them later doesn't invalidate the corpus. contextHash: string; // Number of transcript chunks the engine was asked to process, and how many // came back usable. A section generated from 3 of 5 chunks is suspect. chunks?: number; chunksOk?: number; // Metered lanes only: what this section cost. costUsd?: number; }; export type DigestSection = { provenance: DigestProvenance; items: T[]; }; export type DigestSections = { chapters?: DigestSection; tags?: DigestSection; }; // One entry per generation pass, appended on change so a re-run is auditable // (the availability-server.ts history idiom). export type DigestHistoryEntry = { section: DigestSectionKind; generatedAt: string; appId: string; model: string; promptVersion: number; contextHash: string; itemCount: number; warningCount: number; }; // A generation pass that produced NOTHING usable, recorded so the failure // survives. // // The two total-failure paths in digestVideo deliberately do not write a // section: an empty section carrying current provenance would read as FRESH and // the video would never be retried. The consequence was that the WORST failures // persisted zero evidence — they existed only in a job log that rotates at 500 // records / 30 days — so a warnings-driven review queue would have been // systematically blind to exactly the videos it exists to catch. // // This is a sibling of `sections`, not a member of it, which is what keeps the // retry behaviour intact: isSectionFresh reads `sections[section].provenance` // and nothing else, so nothing here can make a failed video look done. export type DigestSectionFailure = { section: DigestSectionKind; at: string; // ISO appId: string; model: string; promptVersion: number; // WHY nothing survived, and the distinction is the whole point: // "no-output" — every chunk failed or came back empty. The model never // proposed anything: an engine, prompt or context problem. // "all-rejected" — the model proposed items and every one failed a guard. // The content exists and the guard threw it away. // The validation run's worst video was the second kind — 13 chapters clamped // away as out-of-range — and it looked identical to the first from outside. // They need different fixes, and only a recorded artifact tells them apart. reason: "no-output" | "all-rejected"; chunks: number; chunksOk: number; warnings: DigestWarning[]; }; export type DigestRecord = { digestSchemaVersion: number; // The most recent generation pass's prompt/context identity. Per-section // provenance is authoritative for freshness (a video can hold sections made // at different versions); these mirror whichever pass wrote last, so a // corpus-wide sweep can be surveyed without opening every section. promptVersion: number; contextHash: string; warnings: DigestWarning[]; sections: DigestSections; // Latest total failure per section, if any. At most one entry per section — // an 81-day sweep retries, and an append-only list would grow without bound // on a video that fails every time. Cleared for a section that later // succeeds. See DigestSectionFailure. failures?: DigestSectionFailure[]; history?: DigestHistoryEntry[]; // Set when this digest was SHARED from another video (a duplicate cluster's // canonical member) rather than generated for this one. Keeps the sharing // honest and lets a later correction propagate to the whole cluster. derivedFrom?: DigestDerivedFrom; }; export type DigestDerivedFrom = { // `${channelSlug}/${id}` of the canonical member the digest was generated for. slug: string; clusterId: string; sharedAt: string; // Measured max cue-timing offset (seconds) between the two videos at the // sampled anchors. Sharing only happens at near-zero offset, so this records // WHY it was considered safe. offsetSeconds: number; }; // --------------------------------------------------------------------------- // Overrides // --------------------------------------------------------------------------- // Human-authored shadow of the machine file, id-keyed exactly like a per-site // search-alias list shadows the global one (mergeAliases: later source wins). // An override entry with an id that matches a generated item REPLACES it; an id // with no match APPENDS. `enabled: false` suppresses without deleting. export type DigestOverrides = { version: number; chapters?: DigestChapter[]; tags?: DigestTag[]; // Free-text operator note (why this was corrected). Never read by code. note?: string; updatedAt?: string; }; // --------------------------------------------------------------------------- // Pure helpers // --------------------------------------------------------------------------- // Deterministic per-item ids. A chapter's identity is its (snapped) start // second: that is what a human is correcting when they retitle a chapter, and it // stays stable across a regeneration that produces the same segmentation. export function chapterId(startSeconds: number): string { return `c${Math.max(0, Math.floor(startSeconds))}`; } // A tag's identity is its normalized text, so re-generating the same tag lands // on the same id (and thus keeps honoring a human's `enabled: false`). export function tagId(tag: string): string { const slug = tag .toLowerCase() .replace(/[^\p{L}\p{N}]+/gu, "-") .replace(/^-+|-+$/g, ""); return `t${slug || "tag"}`; } // Merge a machine-generated item list with its human shadow. Later source wins // per id (mergeAliases), an overridden item reports decidedBy: "human", and // suppressed items (enabled === false) are dropped from the effective list. // `sort` keeps the composed list in the caller's canonical order. function mergeItems( machine: T[], overrides: T[] | undefined, sort: (a: T, b: T) => number, ): T[] { const byId = new Map(); for (const item of machine) byId.set(item.id, item); for (const item of overrides ?? []) { const existing = byId.get(item.id); // A human edit is authoritative for the fields it names, but an override // authored as a patch (id + title only) must not blank the machine's start. byId.set(item.id, { ...(existing ?? {}), ...item, decidedBy: "human" } as T); } return Array.from(byId.values()) .filter((item) => item.enabled !== false) .sort(sort); } const byStart = (a: DigestChapter, b: DigestChapter): number => a.start - b.start || a.title.localeCompare(b.title); const byTag = (a: DigestTag, b: DigestTag): number => a.tag.localeCompare(b.tag); export type EffectiveDigest = { chapters: DigestChapter[]; tags: DigestTag[]; // True when any effective item came from the override file — the signal the // editor uses to show "edited by hand". hasOverrides: boolean; warnings: DigestWarning[]; sections: DigestSections; derivedFrom?: DigestDerivedFrom; }; // Compose the machine file and the human shadow into what a reader should see. // Either side may be null (no digest yet / no corrections yet). export function effectiveDigest( machine: DigestRecord | null, overrides: DigestOverrides | null, ): EffectiveDigest { const chapters = mergeItems( machine?.sections.chapters?.items ?? [], overrides?.chapters, byStart, ); const tags = mergeItems( machine?.sections.tags?.items ?? [], overrides?.tags, byTag, ); return { chapters, tags, hasOverrides: (overrides?.chapters?.length ?? 0) > 0 || (overrides?.tags?.length ?? 0) > 0, warnings: machine?.warnings ?? [], sections: machine?.sections ?? {}, ...(machine?.derivedFrom ? { derivedFrom: machine.derivedFrom } : {}), }; } // The freshness test that keeps a re-run from becoming a second sweep: a section // is regenerated ONLY when its recorded identity differs from what we would // produce now. Missing section → stale (never generated). export type DigestFreshnessTarget = { appId: string; model: string; promptVersion: number; contextHash: string; // Absent (or empty) means the default prompt shape — see DigestProvenance. promptVariant?: string; schemaVersion?: number; }; // Absent and "" are the same thing (the default shape), so that a record written // before promptVariant existed compares equal to one written today with the // default settings. Without this normalization, adding the field would have // invalidated every digest on disk. function sameVariant(a: string | undefined, b: string | undefined): boolean { return (a ?? "") === (b ?? ""); } export function isSectionFresh( record: DigestRecord | null, section: DigestSectionKind, target: DigestFreshnessTarget, ): boolean { if (!record) return false; if ( record.digestSchemaVersion !== (target.schemaVersion ?? DIGEST_SCHEMA_VERSION) ) { return false; } const p = record.sections[section]?.provenance; if (!p) return false; return ( p.appId === target.appId && (p.modelRequested ?? p.model) === target.model && p.promptVersion === target.promptVersion && p.contextHash === target.contextHash && sameVariant(p.promptVariant, target.promptVariant) ); } // A shared digest is fresh for the receiving video as long as it still points at // the same canonical member; the canonical member's own freshness is what drives // regeneration, and the share is re-applied from it. export function isSharedFrom( record: DigestRecord | null, canonicalSlug: string, ): boolean { return record?.derivedFrom?.slug === canonicalSlug; } // One row per section kind the current settings ask for. THE SAME FOLD the // digest operation's state() counts — a section is fresh here iff state() // counts it fresh — so the per-video panel and the channel's work list cannot // disagree about one section. Before this the video page folded its own copy // of the rule beside the registry's, which is the way two counters end up // describing the same disk differently. // // A SHARED digest (derivedFrom) is fresh for the receiving video as long as it // still points at its canonical member (isSharedFrom's contract); state() // returns `present` on the same condition, before it ever reaches this fold. export type DigestSectionState = { kind: DigestSectionKind; present: boolean; fresh: boolean; provenance?: DigestProvenance; }; export function digestSectionStates( record: DigestRecord | null, sections: readonly DigestSectionKind[], target: DigestFreshnessTarget, ): DigestSectionState[] { return sections.map((kind) => { const section = record?.sections[kind]; return { kind, present: section !== undefined, fresh: record?.derivedFrom != null || isSectionFresh(record, kind, target), ...(section?.provenance ? { provenance: section.provenance } : {}), }; }); }