Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a9e15e7b204851f9deb7b8794d87d96bce4d68ef
parent f21748d52c92bf9d9b256ffabb5acd32677b3d50
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Fri, 28 Aug 2026 15:39:49 -0400

common: a video's operations are read once, from the registry

The per-video page knew about exactly one derived-data feature: it re-derived
the digest's freshness itself, calling resolveDigestTarget with its own lane
and folding the per-section rule a second time beside the registry's. The
speaker operations had no per-video reader at all.

controller/videoOperations.ts is that reader. It walks OPERATIONS in registry
order, resolves each entry's target once and asks the entry itself for the
video's state — never re-deriving a classification — off ONE readVideoFiles.
`shownOnVideoPage` is kept separate from `enabled` on purpose: a sidecar
captured before its feature was switched off is still something an operator
needs to see. The external operations are deliberately absent; their per-video
surface is the stage cards inside VideoPanel.

digestSectionStates() in lib/digest.ts is now the ONE fold. The digest
operation's state() counts fresh sections through it, and the per-video panel
will read its rows from it in the next commit, so a panel and a work list
cannot disagree about one section.

hasDigest() had one caller — countDataFiles, the channel count's
ground-truth walk — and being exported made it read as a general "is this
video digested", which is the second definition of digested the registry's
state() exists to be the only one of. It moves into channels.ts as the private
hasDigestWithItems, identity-blind comment and body verbatim.

hasAttribution() had zero callers and is deleted. hasDiarization stays: it is
the live audio-deletion guard.

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>

Diffstat:
Mcommon/controller/channelProjection.test.ts | 3++-
Mcommon/controller/channels.ts | 25+++++++++++++++++++++++--
Acommon/controller/videoOperations.test.ts | 242+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Acommon/controller/videoOperations.ts | 133+++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mcommon/lib/attribution-server.ts | 4----
Mcommon/lib/digest-server.ts | 15---------------
Mcommon/lib/digest.ts | 34++++++++++++++++++++++++++++++++++
Mcommon/lib/operations.ts | 9++++++---
8 files changed, 440 insertions(+), 25 deletions(-)

diff --git a/common/controller/channelProjection.test.ts b/common/controller/channelProjection.test.ts @@ -64,7 +64,8 @@ async function seedChannel( if (v.audio) await writeFile(path.join(dir, "audio.mp3"), "x"); if (v.digest) { // digestSchemaVersion is required (loadDigest rejects the record without - // it), and hasDigest only counts a record with a non-empty section. + // it), and the channel count only counts a record with a non-empty + // section. await writeFile( path.join(dir, "ai-digest.json"), JSON.stringify({ diff --git a/common/controller/channels.ts b/common/controller/channels.ts @@ -12,7 +12,7 @@ import { isVideoTranscribed, readVideoFiles, } from "../lib/videoStatus"; -import { hasDigest } from "../lib/digest-server"; +import { loadDigest } from "../lib/digest-server"; // TYPE-ONLY, and it must stay that way: ./channelSnapshot imports // readChannelConfig from this module, and it drags in the snapshot generator's // whole dependency graph (lmdb, the archive reader, the digest layer). A value @@ -64,6 +64,25 @@ async function exists(p: string): Promise<boolean> { } } +// Cheap existence/coverage check for the channel digestCount below, which runs +// over every video dir in a channel. It is identity-BLIND on purpose — it +// answers "is there a digest at all?", never "is it current"; the work list is +// the digest operation's `state()`. Reads the file (a few KB) rather than +// statting, because a digest whose sections are all empty is not coverage. +// +// PRIVATE, and it lives here rather than in lib/digest-server.ts because this +// is its only caller: exported, it read as a general "does this video have a +// digest" helper, which is exactly the second definition of "digested" the +// registry's state() exists to be the only one of. +async function hasDigestWithItems(videoDir: string): Promise<boolean> { + const record = await loadDigest(videoDir); + if (!record) return false; + return ( + (record.sections.chapters?.items.length ?? 0) > 0 || + (record.sections.tags?.items.length ?? 0) > 0 + ); +} + async function countDataFiles(dataDir: string): Promise<{ videos: number; transcripts: number; @@ -91,7 +110,9 @@ async function countDataFiles(dataDir: string): Promise<{ // Only transcribed videos can carry a digest, so the sidecar read is // skipped for the rest — the same conditional per-video sidecar-read // pattern channelSnapshot.ts uses for coverage and VTT provenance. - digest: isVideoTranscribed(files) ? await hasDigest(dir) : false, + digest: isVideoTranscribed(files) + ? await hasDigestWithItems(dir) + : false, }; }, ); diff --git a/common/controller/videoOperations.test.ts b/common/controller/videoOperations.test.ts @@ -0,0 +1,242 @@ +// The per-video registry reader. +// +// Run with: node_modules/.bin/tsx --test common/controller/videoOperations.test.ts +// +// Its own file because it needs a SETTINGS SEAM: the digest entry's +// resolveTarget reads settings and the channel's context note from disk (via +// resolveDigestTarget), and getPaths() memoizes its first answer at module +// scope — so TRANSCRIPTS_DIR and SETTINGS_FILE must be set before anything can +// import the module under test. Same arrangement as laneForOperation.test.ts. + +import { mkdtempSync, writeFileSync } from "node:fs"; +import { mkdir, rm, writeFile } from "node:fs/promises"; +import os from "node:os"; +import path from "node:path"; +import { test, after } from "node:test"; +import assert from "node:assert/strict"; + +// Set BEFORE anything can call getPaths(). node:test runs each file in its own +// process, so this is scoped to this file alone. +const ROOT = mkdtempSync(path.join(os.tmpdir(), "video-operations-")); +process.env.TRANSCRIPTS_DIR = ROOT; +const SETTINGS_FILE = path.join(ROOT, "settings.json"); +process.env.SETTINGS_FILE = SETTINGS_FILE; + +// getSettings() re-reads the file on every call, so a case can flip a feature +// between assertions — but the FIRST write has to land before any import that +// might read it. +function writeSettings(settings: Record<string, unknown>): void { + writeFileSync(SETTINGS_FILE, JSON.stringify(settings), "utf8"); +} + +// Both speaker features armed and the digest lane pointed at the default app. +// The diarization model paths are full paths on purpose: the freshness target +// compares BASENAMES, and a sidecar recording "seg-1.onnx" has to read fresh +// against a setting of "/models/seg-1.onnx". +function settingsOn(over: Record<string, unknown> = {}): Record<string, unknown> { + return { + diarization: { + enabled: true, + segModel: "/models/seg-1.onnx", + embModel: "/models/emb-1.onnx", + }, + attribution: { + enabled: true, + diarizedEnabled: true, + textOnlyEnabled: true, + }, + ...over, + }; +} + +writeSettings(settingsOn()); + +const { inspectVideoOperations, orderForVideoPage, shownOnVideoPage } = + await import("./videoOperations"); +const { getPaths } = await import("../lib/paths"); +const { getSettings } = await import("../lib/settings"); + +after(() => rm(ROOT, { recursive: true, force: true })); + +const SLUG = "chan"; + +// A video dir described by what is on disk. Written in mtime order — +// metadata and the raw transcript first, the cues sidecar LAST — because +// isCuesJsonFresh compares mtimes and a cues file older than its raw +// transcript reads as superseded (which is the `deferred` case below). +async function seed( + videoId: string, + opts: { + transcript?: boolean; + cues?: boolean; + diarization?: boolean; + attribution?: string; + } = {}, +): Promise<string> { + const dir = path.join(ROOT, "channels", SLUG, "data", videoId); + await mkdir(dir, { recursive: true }); + await writeFile( + path.join(dir, "metadata.info.json"), + JSON.stringify({ id: videoId, duration: 600 }), + ); + if (opts.transcript !== false) { + await writeFile( + path.join(dir, "transcript.json"), + JSON.stringify({ transcription: [{ text: "hi" }] }), + ); + } + if (opts.diarization) { + await writeFile( + path.join(dir, "diarization.json"), + JSON.stringify({ + videoId, + generatedAt: "2026-08-07T00:00:00.000Z", + speakers: 2, + turns: [{ start: 0, end: 4, speaker: 0 }], + engine: { + engine: "sherpa-onnx", + segmentationModel: "seg-1.onnx", + embeddingModel: "emb-1.onnx", + threshold: 0.9, + }, + }), + ); + } + if (opts.attribution !== undefined) { + await writeFile(path.join(dir, "attribution.json"), opts.attribution); + } + if (opts.cues !== false) { + await writeFile( + path.join(dir, "transcript.cues.json"), + JSON.stringify({ cues: [] }), + ); + } + return dir; +} + +function inspect(videoId: string) { + return inspectVideoOperations({ + paths: getPaths(), + channelSlug: SLUG, + videoId, + settings: getSettings(), + }); +} + +function stateOf( + views: Awaited<ReturnType<typeof inspect>>, + id: string, +): string | undefined { + return views.find((v) => v.id === id)?.state; +} + +test("one view per registry entry, in OPERATIONS order", async () => { + writeSettings(settingsOn()); + await seed("vid-order"); + const views = await inspect("vid-order"); + assert.deepEqual( + views.map((v) => v.id), + ["diarization", "attribution-diarized", "attribution-text", "digest"], + ); + // The label and the settings block come off the entry, not off a table kept + // beside it — that is what makes a fifth operation need no page. + assert.equal( + views.find((v) => v.id === "diarization")?.label, + "Speaker diarization", + ); + assert.equal( + views.find((v) => v.id === "attribution-text")?.settingsBlock, + "attribution", + ); +}); + +test("a diarized video: the capture is present, both naming lanes are missing", async () => { + writeSettings(settingsOn()); + await seed("vid-diarized", { diarization: true }); + const views = await inspect("vid-diarized"); + assert.equal(stateOf(views, "diarization"), "present"); + // `missing`, not `stale`: the record this lane is responsible for was never + // made. Both are reachable work and they mean different things. + assert.equal(stateOf(views, "attribution-diarized"), "missing"); + assert.equal(stateOf(views, "attribution-text"), "missing"); + // outputs track the listing this page already read. + const diarizationView = views.find((v) => v.id === "diarization"); + assert.equal(diarizationView?.outputs[0]?.name, "diarization.json"); + assert.equal(diarizationView?.outputs[0]?.present, true); + assert.equal( + views.find((v) => v.id === "attribution-text")?.outputs[0]?.present, + false, + ); +}); + +test("an undiarized video: the diarized naming lane is blocked, and says on what", async () => { + writeSettings(settingsOn()); + await seed("vid-undiarized"); + const views = await inspect("vid-undiarized"); + const view = views.find((v) => v.id === "attribution-diarized"); + assert.equal(view?.state, "blocked"); + // BLOCKED names a thing this system produces — the surface can say what the + // video is waiting FOR rather than just that it is stuck. + assert.equal(view?.dependsOn[0]?.id, "diarization"); + assert.equal(view?.dependsOn[0]?.label, "Speaker diarization"); +}); + +test("an untranscribed video is not-applicable to the speaker lanes and blocks the digest", async () => { + writeSettings(settingsOn()); + await seed("vid-raw", { transcript: false, cues: false }); + const views = await inspect("vid-raw"); + assert.equal(stateOf(views, "diarization"), "not-applicable"); + assert.equal(stateOf(views, "attribution-diarized"), "not-applicable"); + assert.equal(stateOf(views, "attribution-text"), "not-applicable"); + // Blocked, not not-applicable: it is waiting on the transcription operation, + // which is a thing this system produces. + assert.equal(stateOf(views, "digest"), "blocked"); +}); + +test("no normalized transcript: the digest is deferred, and carries its own reason", async () => { + writeSettings(settingsOn()); + await seed("vid-nocues", { cues: false }); + const views = await inspect("vid-nocues"); + const digest = views.find((v) => v.id === "digest"); + assert.equal(digest?.state, "deferred"); + // The kind's own words, verbatim — a plural fragment completing "N videos + // are …". No singular twin: one copy of a sentence cannot drift from itself. + assert.match(digest?.deferredHint ?? "", /normalized transcript/); +}); + +test("a digested-able video with no record reads missing", async () => { + writeSettings(settingsOn()); + await seed("vid-nodigest"); + assert.equal(stateOf(await inspect("vid-nodigest"), "digest"), "missing"); +}); + +test("shownOnVideoPage: off with nothing on disk hides; off with a sidecar shows", async () => { + // Capture switched off. The video with no sidecar has nothing to show; the + // one with a sidecar has a record that cost audio nobody has any more, and + // hiding it is how that becomes invisible. + writeSettings(settingsOn({ diarization: { enabled: false } })); + await seed("vid-off-empty"); + await seed("vid-off-record", { diarization: true }); + + const empty = (await inspect("vid-off-empty")).find( + (v) => v.id === "diarization", + ); + assert.equal(empty?.enabled, false); + assert.equal(shownOnVideoPage(empty!), false); + + const withRecord = (await inspect("vid-off-record")).find( + (v) => v.id === "diarization", + ); + assert.equal(withRecord?.enabled, false); + assert.equal(shownOnVideoPage(withRecord!), true); +}); + +test("orderForVideoPage puts the digest group ahead of the speakers group", async () => { + writeSettings(settingsOn()); + await seed("vid-sort", { diarization: true }); + const ordered = orderForVideoPage(await inspect("vid-sort")); + assert.deepEqual( + ordered.map((v) => v.id), + ["digest", "diarization", "attribution-diarized", "attribution-text"], + ); +}); diff --git a/common/controller/videoOperations.ts b/common/controller/videoOperations.ts @@ -0,0 +1,133 @@ +// WHAT ONE VIDEO'S OPERATIONS SAY ABOUT IT, read once, from the registry. +// +// The per-video page used to know about exactly one derived-data feature: it +// re-derived the digest's freshness itself, from a resolver it called with its +// own lane argument, and folded the per-section rule a second time beside the +// registry's own. Diarization and attribution had no per-video surface at all +// — their records existed on disk with nowhere for a human to look at them. +// +// So this is the reader that makes a fifth operation need no page: it walks +// OPERATIONS in registry order, resolves each entry's target ONCE and asks the +// entry itself for the video's state. Nothing here re-derives a classification, +// and nothing here filters on `enabled` — shownOnVideoPage is that rule, and it +// is deliberately separate so a sidecar captured before its feature was +// switched off is still something the operator can see. +// +// THE EXTERNAL OPERATIONS ARE DELIBERATELY ABSENT. Download, transcode and +// transcription already have a per-video surface — the pipeline stage cards +// inside the editor's VideoPanel, which carry their own per-video actions. This +// reader is over OPERATIONS, the backfill/derived-data registry, and that +// asymmetry is intentional rather than an omission. +// +// ONE readVideoFiles PER PAGE. It is a readdir, and the registry's state() +// takes the listing as its probe rather than re-reading it, so the whole page +// costs one listing plus whatever sidecar reads the individual entries decide +// are worth paying for. + +import path from "node:path"; +import type { Paths } from "../lib/paths"; +import type { SiteSettings } from "../lib/settings"; +import { readVideoFiles } from "../lib/videoStatus"; +import { + OPERATIONS, + OPERATION_GROUP_ORDER, + operationLabel, + type OperationClassification, + type OperationGroup, + type OperationSettingsBlock, +} from "../lib/operations"; + +export type VideoOperationView = { + id: string; + label: string; + group: OperationGroup; + settingsBlock?: OperationSettingsBlock; + // Operation.state(probe), NEVER re-derived. The whole point of the reader. + state: OperationClassification; + // op.enabled(settings) — the feature's own gate, kept beside the state + // rather than folded into it: a switched-off operation still has a state. + enabled: boolean; + // op.outputs against the listing this page already read. + outputs: { name: string; present: boolean }[]; + // The kind's own words for what a `deferred` video is waiting for. A plural + // fragment completing "N videos are …" — kept VERBATIM rather than given a + // singular twin, because one copy of a sentence cannot drift from itself and + // operations.test.ts pins the digest one. + deferredHint?: string; + dependsOn: { id: string; label: string }[]; + // The resolved target, erased exactly as Operation.resolveTarget erases it. + // The one consumer that narrows it is the digest body (it needs + // {target, sections} for its per-section rows). Never crosses to a client. + target: unknown; +}; + +export async function inspectVideoOperations(opts: { + paths: Paths; + channelSlug: string; + videoId: string; + settings: SiteSettings; +}): Promise<VideoOperationView[]> { + const { paths, channelSlug, videoId, settings } = opts; + const videoDir = path.join(paths.channelsDir, channelSlug, "data", videoId); + // checkUntranscribable, because several entries classify an untranscribable + // video as not-applicable and would otherwise report it as work. + const files = await readVideoFiles(videoDir, { checkUntranscribable: true }); + const views: VideoOperationView[] = []; + for (const op of OPERATIONS) { + const target = await op.resolveTarget({ settings, paths, channelSlug }); + const state = await op.state({ + videoDir, + videoId, + files, + target, + settings, + }); + views.push({ + id: op.id, + label: op.label, + group: op.group, + ...(op.settingsBlock ? { settingsBlock: op.settingsBlock } : {}), + state, + enabled: op.enabled(settings), + outputs: op.outputs.map((name) => ({ + name, + present: files.entries.includes(name), + })), + ...(op.deferredHint ? { deferredHint: op.deferredHint } : {}), + dependsOn: (op.dependsOn ?? []).map((id) => ({ + id, + label: operationLabel(id), + })), + target, + }); + } + return views; +} + +// Shown when the feature is on OR its output is on disk: a sidecar captured +// before the feature was switched off is still a thing the operator needs to +// see, and hiding it is how a record that cost audio nobody has any more +// becomes invisible. +export function shownOnVideoPage(view: VideoOperationView): boolean { + return view.enabled || view.outputs.some((o) => o.present); +} + +// Group order first (OPERATION_GROUP_ORDER — the order the channel stages and +// the /channels columns already use, so a reader moving between pages sees one +// sequence), registry order within a group. Pure, and a stable sort: the +// registry's own order is what decides the three speaker operations. +export function orderForVideoPage( + views: VideoOperationView[], +): VideoOperationView[] { + const groupRank = (g: OperationGroup): number => { + const i = OPERATION_GROUP_ORDER.indexOf(g); + return i === -1 ? OPERATION_GROUP_ORDER.length : i; + }; + return views + .map((view, index) => ({ view, index })) + .sort( + (a, b) => + groupRank(a.view.group) - groupRank(b.view.group) || a.index - b.index, + ) + .map((e) => e.view); +} diff --git a/common/lib/attribution-server.ts b/common/lib/attribution-server.ts @@ -33,10 +33,6 @@ export async function loadAttribution( } } -export async function hasAttribution(videoDir: string): Promise<boolean> { - return (await loadAttribution(videoDir)) !== null; -} - // tmp + rename, so a crash mid-write leaves the previous record rather than a // truncated one. Identical to writeDiarization; the atomicity is what lets // loadAttribution treat "parsed but wrong shape" as a real anomaly rather than diff --git a/common/lib/digest-server.ts b/common/lib/digest-server.ts @@ -82,21 +82,6 @@ export async function loadDigest( } } -// Cheap existence/coverage check for the snapshot's `digestEngines` split and -// the channel digestCount, which run over every video dir in a channel. It is -// identity-BLIND on purpose — it answers "is there a digest at all?", never -// "is it current"; the work list is the digest operation's `state()`. Reads the -// file (a few KB) rather than statting, because a digest whose sections are all -// empty is not coverage. -export async function hasDigest(videoDir: string): Promise<boolean> { - const record = await loadDigest(videoDir); - if (!record) return false; - return ( - (record.sections.chapters?.items.length ?? 0) > 0 || - (record.sections.tags?.items.length ?? 0) > 0 - ); -} - export async function loadDigestOverrides( videoDir: string, ): Promise<DigestOverrides | null> { diff --git a/common/lib/digest.ts b/common/lib/digest.ts @@ -542,3 +542,37 @@ export function isSharedFrom( ): boolean { return record?.derivedFrom?.slug === canonicalSlug; } + +// One row per section kind the current settings ask for. THE SAME FOLD the +// digest operation's state() counts — a section is fresh here iff state() +// counts it fresh — so the per-video panel and the channel's work list cannot +// disagree about one section. Before this the video page folded its own copy +// of the rule beside the registry's, which is the way two counters end up +// describing the same disk differently. +// +// A SHARED digest (derivedFrom) is fresh for the receiving video as long as it +// still points at its canonical member (isSharedFrom's contract); state() +// returns `present` on the same condition, before it ever reaches this fold. +export type DigestSectionState = { + kind: DigestSectionKind; + present: boolean; + fresh: boolean; + provenance?: DigestProvenance; +}; + +export function digestSectionStates( + record: DigestRecord | null, + sections: readonly DigestSectionKind[], + target: DigestFreshnessTarget, +): DigestSectionState[] { + return sections.map((kind) => { + const section = record?.sections[kind]; + return { + kind, + present: section !== undefined, + fresh: + record?.derivedFrom != null || isSectionFresh(record, kind, target), + ...(section?.provenance ? { provenance: section.provenance } : {}), + }; + }); +} diff --git a/common/lib/operations.ts b/common/lib/operations.ts @@ -92,8 +92,8 @@ import { import type { AutoQueueKind } from "../jobs/autoQueueState"; import { DIGEST_FILENAME, + digestSectionStates, isDigestSectionKind, - isSectionFresh, type DigestAppConfig, type DigestFreshnessTarget, type DigestItem, @@ -1173,8 +1173,11 @@ const digest: Operation = { // The empty-sections case still reads `present` (0 of 0 fresh), exactly as // the `every()` this replaces did, so a caller passing no sections is // unchanged. - const fresh = sections.filter((section) => - isSectionFresh(record, section, freshness), + // digestSectionStates is the fold, shared with the per-video digest panel: + // one definition of "this section is fresh", so a panel and a work list + // cannot disagree about the same section. + const fresh = digestSectionStates(record, sections, freshness).filter( + (s) => s.fresh, ).length; if (fresh === sections.length) return "present"; // No record at all cannot be part-done, and it is the one branch that must