import { rm, writeFile } from "node:fs/promises"; import { join } from "node:path"; import { test, expect } from "@playwright/test"; import type { Page } from "@playwright/test"; import { channelStage, generateReport, pathExists, readJson, resetData, resolvePath, writeChannelConfig, writeDigestVideo, writeSettings, } from "./helpers"; import type { AttributionRecord } from "../../common/lib/attribution"; // Speaker attribution: the SECOND and THIRD backfill kinds, and the pair that // share one file. // // What these specs are for, in order of how much they would cost to get wrong: // // 1. THE TEXT LANE MUST NEVER OVERWRITE A DIARIZED RECORD. Both kinds write // attribution.json. The diarized lane costs about one model call per video // and names clusters an audio diarizer produced; the text lane costs ~30 // calls and guesses at identity across chunk seams. A downgrade silently // replaces the first with the second, and nothing on disk would say so. // 2. THE UPGRADE MUST HAPPEN. A text-only record where diarization exists is // the upgrade queue — the whole of PLAN.md's bespoke "upgrade job", falling // out of the registry rather than needing new machinery. // 3. THE INDICATORS MUST STAY GENERIC. Registering these kinds is supposed to // light every surface with no UI change; the last spec is what makes that a // claim rather than an assumption, and it also pins that the two numbers // are still not summed for a kind whose missing-input population is ~73,000 // videos on the real corpus. // // The engine is the ollama HTTP stub the digest specs already use // (e2e/fixtures/ollama-stub.mjs, wired in via OLLAMA_URL in dev:test). It // answers from the SCHEMA — the diarized lane's cluster enum is read back out of // the request — so a spec cannot pass on a cluster the video never had. const CHANNEL = "attribution-channel"; // The video that HAS diarization: the cheap, better lane. const DIARIZED = "attrvid0001"; // The video that does not: the text-only lane's territory. const TEXTONLY = "attrvid0002"; const SLOW = 120_000; function dataRel(videoId: string, file: string): string { return join("test-transcripts", "channels", CHANNEL, "data", videoId, file); } const DIARIZED_AT = "2026-08-06T00:00:00.000Z"; // Two speakers, half the video each — so the cluster ranking is a tie broken by // index, and the stub's first name lands on cluster 0 deterministically. function diarizationRecord(generatedAt = DIARIZED_AT) { return { videoId: DIARIZED, generatedAt, speakers: 2, turns: [ { start: 0, end: 300, speaker: 0 }, { start: 300, end: 600, speaker: 1 }, ], engine: { engine: "fake-diarize", segmentationModel: "null", embeddingModel: "null", threshold: 0.5, }, }; } function attributionRecord(over: Record = {}) { return { videoId: DIARIZED, generatedAt: "2020-01-01T00:00:00.000Z", speakers: [{ index: 0, label: "An Earlier Guess" }], segments: [{ start: 0, end: 10, speaker: 0 }], provenance: { method: "text-only", appId: "ollama-direct", model: "qwen2.5:7b", modelRequested: "qwen2.5:7b", promptVersion: 1, generatedAt: "2020-01-01T00:00:00.000Z", ...(over.provenance as Record), }, }; } // Settings with both attribution lanes armed and the backfill lane switched on. // // Nothing here has to force a share any more: the `weight` scalar retired in // slice 1.3, and the idle-only rule is now the operation's declared // `contendsFor` — attribution contends for the network, so it keeps its slots // whatever transcription is doing. That is exactly what these specs need, and // the GPU carve-out has a pure unit test (operationBatch.test.ts), which is the // right place for it — no pool, no GPU, no timing. function attributionSettings(over: { attribution?: Record; } = {}) { return { adminTitle: "Test Admin", maxTranscriptPageBytes: 8388608, sleepBetweenDownloadsSeconds: 0, minFreeDiskGB: 0, verifyAvailabilityBeforeClean: false, syncScheduler: { fullSweepIntervalMinutes: 0 }, digest: { localAppId: "ollama-direct", remoteAppId: "claude-code" }, // Diarization CAPTURE stays off throughout. Naming clusters that were // already captured must not require the capture lane to still be armed — // otherwise switching capture off would strand exactly the work it exists to // protect. backfill: { concurrency: 1, allowRedownload: false, }, // THE LANE'S GATE, SPELLED. It used to be spelled by `backfill.enabled: // true` in the block above — the inverted retired field, where `true` meant // NOT held. S0-pause deleted it, and the lane's gate defaults SHUT // (`defaultHeldFor`), so a fixture that wants the lane to run says so on // the lane. autoQueue: { backfill: { held: false } }, attribution: { enabled: true, appId: "ollama-direct", model: "qwen2.5:7b", diarizedEnabled: true, textOnlyEnabled: true, promptVersion: 1, ...over.attribution, }, }; } async function seedChannel(opts: { diarization?: boolean } = {}) { await resetData(null); await writeChannelConfig(CHANNEL); await writeDigestVideo({ channelSlug: CHANNEL, videoId: DIARIZED }); await writeDigestVideo({ channelSlug: CHANNEL, videoId: TEXTONLY }); if (opts.diarization !== false) { await writeFile( resolvePath(dataRel(DIARIZED, "diarization.json")), JSON.stringify(diarizationRecord()) + "\n", ); } } async function runBackfill(page: Page): Promise { await page.goto(channelStage(CHANNEL, "speakers")); await page .getByRole("button", { name: "Run speaker work", exact: true }) .click(); // The batch's closing summary line — emitted after the last write, so it is // the happens-before edge for the sidecar reads below. await expect(page.getByLabel("Run speaker work output")).toContainText( "already current", { timeout: 90_000 }, ); } function readAttribution(videoId: string) { return readJson(dataRel(videoId, "attribution.json")); } // --------------------------------------------------------------------------- // 1. Both lanes produce a sidecar, and each one produces its OWN kind of record // --------------------------------------------------------------------------- test("each lane writes the record it is responsible for", async ({ page }) => { test.setTimeout(SLOW); await seedChannel(); await writeSettings(attributionSettings()); await generateReport(page, CHANNEL); await runBackfill(page); // The diarized lane: names read back onto the clusters the diarizer found. const diarized = await readAttribution(DIARIZED); expect(diarized.provenance.method).toBe("diarized"); // The identity of the diarization run these names point at. Without it a // re-diarization would leave the names pointing at different clusters, and // nothing on disk could tell. expect(diarized.provenance.diarizationGeneratedAt).toBe(DIARIZED_AT); // The stub answers from the schema's cluster ENUM, so a cluster here proves // the runner sent the clusters the sidecar actually contains. expect(diarized.speakers.map((s) => s.cluster).sort()).toEqual([0, 1]); expect(diarized.speakers[0].label).toBe("Marla Vance"); expect(diarized.speakers[0].confidence).toBeGreaterThan(0.5); // Segments come from the diarizer's own acoustic boundaries, not from a model // reading text. expect(diarized.segments.length).toBeGreaterThan(0); expect(diarized.segments[0].start).toBe(0); // ONE call: no chunking happened at all, which is the whole cost argument for // this lane. expect(diarized.provenance.chunks).toBeUndefined(); // The text-only lane: no clusters to point at, and no confidence claimed. const textOnly = await readAttribution(TEXTONLY); expect(textOnly.provenance.method).toBe("text-only"); expect(textOnly.provenance.chunksOk).toBe(textOnly.provenance.chunks); expect(textOnly.speakers.length).toBeGreaterThan(0); expect(textOnly.speakers[0].cluster).toBeUndefined(); // Deliberately absent: the model was asked where the speaker changes, not how // sure it is who anyone is. Inventing a number here would be the overclaiming // the plan warns against. expect(textOnly.speakers[0].confidence).toBeUndefined(); expect(textOnly.segments.length).toBeGreaterThan(0); }); // --------------------------------------------------------------------------- // 2. THE UPGRADE — a text-only record where diarization exists is work // --------------------------------------------------------------------------- test("the diarized lane upgrades a text-only record", async ({ page }) => { test.setTimeout(SLOW); await seedChannel(); await writeFile( resolvePath(dataRel(DIARIZED, "attribution.json")), JSON.stringify(attributionRecord()) + "\n", ); await writeSettings(attributionSettings()); await generateReport(page, CHANNEL); // The card counts it as work, not as done. `missing` rather than `stale`: the // diarized record genuinely was never made. One panel is open at a time now, // so the Backfill card has to be asked for — it used to be on screen because // every stage rendered expanded. await page.goto(channelStage(CHANNEL, "speakers")); await expect( page.getByLabel("video needing speaker work attrvid0001"), ).toBeVisible(); await runBackfill(page); const record = await readAttribution(DIARIZED); expect(record.provenance.method).toBe("diarized"); expect(record.speakers.map((s) => s.label)).not.toContain("An Earlier Guess"); expect(record.provenance.diarizationGeneratedAt).toBe(DIARIZED_AT); }); // --------------------------------------------------------------------------- // 3. THE DOWNGRADE — and it must not happen, ever // --------------------------------------------------------------------------- test("the text lane will not overwrite a record made from the audio", async ({ page, }) => { test.setTimeout(SLOW); // No diarization.json at all: the audio was captured, named, and the sidecar // has since been cleaned away. The diarized record is now the ONLY thing that // knows who was speaking, and it is unreproducible. await seedChannel({ diarization: false }); const original = attributionRecord({ provenance: { method: "diarized", appId: "ollama-direct", model: "qwen2.5:7b", modelRequested: "qwen2.5:7b", promptVersion: 1, generatedAt: "2020-01-01T00:00:00.000Z", diarizationGeneratedAt: DIARIZED_AT, }, }); await writeFile( resolvePath(dataRel(DIARIZED, "attribution.json")), JSON.stringify(original) + "\n", ); // Only the text lane is armed, so nothing else can be what preserved the file. await writeSettings( attributionSettings({ attribution: { diarizedEnabled: false } }), ); await generateReport(page, CHANNEL); await runBackfill(page); const after = await readAttribution(DIARIZED); // Byte-for-byte the record that was there: same method, same names, same // timestamp. Not merely "still diarized" — a rewrite that happened to keep the // method would still have destroyed the labels. expect(after.provenance.method).toBe("diarized"); expect(after.generatedAt).toBe("2020-01-01T00:00:00.000Z"); expect(after.speakers[0].label).toBe("An Earlier Guess"); // The lane was not idle, though — the OTHER video had no record and got one. expect(await pathExists(dataRel(TEXTONLY, "attribution.json"))).toBe(true); }); // --------------------------------------------------------------------------- // 3b. The video page — one panel per operation, and a Run for one video. // // This spec owns them because its seeding IS the interesting case: capture is // switched OFF throughout while attrvid0001 carries a diarization.json. That is // exactly "shown because its output is on disk", and it is not hypothetical — // a sidecar produced from audio the cleanup sweep has since deleted is the only // surviving record of what was heard. // --------------------------------------------------------------------------- test("the video page draws one panel per operation, with the registry's own state", async ({ page, }) => { await seedChannel(); await writeSettings(attributionSettings()); await page.goto(`/channels/${CHANNEL}/videos/${DIARIZED}`); // Shown even though capture is off, and SAID to be off — otherwise its pill // would offer a regeneration nothing can perform. await expect(page.getByLabel("diarization panel")).toBeVisible(); await expect(page.getByLabel("diarization off")).toBeVisible(); // WHICH configuration produced it. After the audio is gone this is the only // thing that can say. await expect(page.getByLabel("diarization provenance")).toContainText( "fake-diarize", ); await expect(page.getByLabel("diarization summary")).toContainText( "2 speakers", ); // Both naming lanes: the record each is responsible for was never made. await expect(page.getByLabel("attribution-diarized freshness")).toHaveText( "not generated", ); await expect(page.getByLabel("attribution-text freshness")).toHaveText( "not generated", ); // ONE attribution.json, two panels reading it — the empty state appears // twice, and that is the two-kinds-one-file rule made visible. expect(await page.getByLabel("attribution empty").count()).toBe(2); // writeDigestVideo lays down a fresh transcript.cues.json, so the digest is // `missing` rather than `deferred`. await expect(page.getByLabel("digest freshness")).toHaveText("not generated"); // The heading is the registry entry's label, not a string kept beside it. await expect( page.getByRole("heading", { name: "Speaker names (from the audio)" }), ).toBeVisible(); await page.goto(`/channels/${CHANNEL}/videos/${TEXTONLY}`); // Off AND nothing on disk: no panel at all. expect(await page.getByLabel("diarization panel").count()).toBe(0); // And the blocked lane names what it is waiting FOR rather than just saying // stuck — the same sentence the stage card uses. await expect(page.getByLabel("attribution-diarized freshness")).toHaveText( "waiting on Speaker diarization", ); }); test("Run from the video page runs that video and no other", async ({ page, }) => { test.setTimeout(SLOW); await seedChannel(); await writeSettings(attributionSettings()); await page.goto(`/channels/${CHANNEL}/videos/${TEXTONLY}`); await page .getByRole("button", { name: "Run Speaker names (from the transcript)", exact: true, }) .click(); // The batch's closing summary line, which is emitted after the last write — // the happens-before edge for the sidecar reads below. await expect( page.getByLabel("Run Speaker names (from the transcript) output"), ).toContainText("1 done", { timeout: 90_000 }); expect((await readAttribution(TEXTONLY)).provenance.method).toBe("text-only"); // THE POINT OF THE ids SCOPE. attrvid0001 is reachable for this same // operation and sits in the same channel; only the id scope kept it // untouched, and without it this run would have been a channel run. expect(await pathExists(dataRel(DIARIZED, "attribution.json"))).toBe(false); // StreamActionLog calls router.refresh() when a run finishes, so the pill // re-renders from the server without a manual reload. await expect(page.getByLabel("attribution-text freshness")).toHaveText( "current", ); // AND THE LOG SURVIVES THAT REFRESH. The pill above is the proof the refresh // landed: the record now exists, so the panel draws its record body instead // of the empty state. When that swap MOVED the run button down the fragment's // child list, React remounted StreamActionLog and the log the operator was // reading disappeared — the whole log element with it. That is also what made // the wait above flaky (1-2 runs in 10): it could only pass in the window // between the summary line arriving and the refresh landing, and on a fast // box the two are ~200 ms apart. Asserted AFTER the pill so the refresh has // definitely happened, which is what makes this deterministic rather than a // second race. await expect( page.getByLabel("Run Speaker names (from the transcript) output"), ).toContainText("1 done"); }); // --------------------------------------------------------------------------- // 4. The indicators, which were supposed to need no UI work at all // --------------------------------------------------------------------------- test("the stage card and the operation pages show attribution beside diarization, with the two numbers still apart", async ({ page, }) => { test.setTimeout(SLOW); await seedChannel(); await writeSettings(attributionSettings()); await rm(resolvePath(`test-transcripts/channels/${CHANNEL}/snapshot.json`), { force: true, }); await generateReport(page, CHANNEL); await page.goto(channelStage(CHANNEL, "speakers")); const section = page.getByLabel("speakers section"); // Per-kind lines appear because there is now more than one kind — the same // component, unmodified, driven by a bigger registry. // // The diarized lane can reach ONE video (the one with diarization.json) and // reports the other as BLOCKED on diarization. That 1-vs-1 split is the // corpus's handful-vs-73,000 in miniature, and the card must never show "2". // // This used to assert "1 needing media", and that was the defect: the second // video is not waiting for media, it is waiting for the diarization lane — // and re-acquiring audio for it could never have helped. await expect( section.getByLabel("speakers operation attribution-diarized"), ).toContainText("1 reachable"); await expect( section.getByLabel("speakers operation attribution-diarized"), ).toContainText("0 needing media"); await expect( section.getByLabel("speakers operation attribution-diarized"), ).toContainText("1 blocked"); // And the card names what it is waiting for, rather than just saying stuck. await expect(section.getByLabel("speakers blocked")).toContainText( "Speaker diarization", ); // The text lane reaches BOTH: its input is the cue stream, which every // transcribed video has. That is exactly why running it corpus-wide is the // expensive option. await expect( section.getByLabel("speakers operation attribution-text"), ).toContainText("2 reachable"); await expect( section.getByLabel("speakers operation attribution-text"), ).toContainText("0 needing media"); // Reachable work across kinds: 1 + 2. The needs-re-acquiring figure is on its // own line and is never folded into it. await expect(section.getByLabel("speakers reachable")).toContainText( "3 videos can be worked on now", ); // The card names the WORK, not the queue: three operations were hiding behind // the word "backfill", and the heading now says which group they are and how // many of them there are. await expect(section.getByRole("heading")).toContainText("Speaker work"); // AND THE RE-ACQUIRE LINE IS GONE, which is the point of the whole change. // Nothing in this fixture needs media fetched: the one video that cannot be // attributed from audio is waiting for the diarization lane, and no download // would have helped it. Before this, that video was reported here and would // have cost a download-and-delete under allowRedownload. await expect(section.getByLabel("speakers needs re-acquiring")).toHaveCount(0); await expect(section.getByLabel("speakers blocked")).toContainText("1"); // The operation pages, which split the same corpus by construction: one page // per attribution kind, each with its own band. The cross-kind sum the old // /actionable row carried ("3" = 1 diarized + 2 text) has no home here and is // not re-asserted — it still lives on the channel page's speakers stage, // asserted a few lines above. await page.goto("/operations/attribution-diarized"); // The lane's console is where the channel appears — as the video it would // dispatch next, which is the sweep itinerary's job done by the dispatcher. const lane = page.locator('section[data-lane="backfill"]'); await expect(lane.getByText(new RegExp(`${CHANNEL}/`))).toBeVisible({ timeout: 30_000, }); // The per-operation figures are the RAIL'S row for this operation: one band // per operation, its populations stated separately and never summed. const diarized = page.locator('li[data-operation="attribution-diarized"]'); await expect(diarized.getByText("1 reachable")).toBeVisible(); await expect(diarized.getByText("1 blocked")).toBeVisible(); await page.goto("/operations/attribution-text"); await expect( page.locator('li[data-operation="attribution-text"]').getByText("2 reachable"), ).toBeVisible(); });