commit 88f045ef11a3edfa85e46538f5d50d8a22150b77
parent dcf1e830a220fa1d05d272fc015803abe7256b60
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Fri, 28 Aug 2026 00:17:14 -0400
attribution: freshness knows which transcript the names were made from
Speaker names are an assertion ABOUT a text. Until now nothing on disk recorded
WHICH text — so replacing a video's YouTube auto-captions with our own whisper
transcript left the old names looking fresh forever, over different wording,
different timings and possibly different speakers.
That replacement used to be rare. The previous commit's hand-off makes it a
routine outcome of a backfill sweep, which is what turns this from a latent
inconsistency into something worth a field.
`AttributionProvenance.transcriptSource` is that field, and it is the same kind
of field as `diarizationGeneratedAt` beside it, written in the same voice: the
thing the record points at, recorded so its replacement is detectable.
`AttributionTranscriptSource` is narrower than `NormalizedTranscript.source`
because cues.json's `source` is pickIndexTranscript's `kind`
(normalizeTranscript.ts:133) and is never "live_chat"; `transcriptSourceOf`
maps anything else to undefined — UNKNOWN MEANS DO NOT ASSERT.
The comparison in isAttributionFresh carries one condition that is the INVERSE
of the diarization guard directly above it: the RECORD must carry the field too,
not just the target. That is sameVersion's absent-means-default rule applied to
an optional string. Without it, shipping the field would invalidate every
attribution.json in the corpus at once — ~194,000 model calls on one channel
alone. Records written from now on carry it and are compared; records already on
disk are left alone.
Both write sites fill it from what they already hold, at zero I/O cost:
attributeOne from the normalized transcript it just read, and operations.ts's
two state() branches from pickIndexTranscript over the VideoFiles the snapshot
already loaded — the cost bar there is the channel snapshot's own per-video
classification, and this adds nothing to it. attributionTarget() is unchanged;
the per-video half is spliced by the caller exactly as the diarized lane's
already is.
The digest lane is equally blind to a transcript replacement (contextHash hashes
the channel's digest-context.md, and DigestProvenance records no transcript
source). Deliberately NOT fixed here and recorded as a follow-up in FACTS.md
instead: the same field under the same rule would re-queue a local LLM digest
for every ASR->whisper replacement, which is the operator's cost call to make.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
Diffstat:
4 files changed, 141 insertions(+), 1 deletion(-)
diff --git a/common/controller/attributeOne.ts b/common/controller/attributeOne.ts
@@ -30,6 +30,7 @@ import {
ATTRIBUTION_PROMPT_VERSION,
isAttributionDowngrade,
isAttributionFresh,
+ transcriptSourceOf,
type AttributionFreshnessTarget,
type AttributionMethod,
type AttributionRecord,
@@ -163,11 +164,17 @@ export async function attributeOneVideo(
// The per-video half of the diarized identity. See
// AttributionProvenance.diarizationGeneratedAt: a cluster index means nothing
// except relative to the run that produced it.
+ //
+ // The transcript half of the identity applies to BOTH lanes: the names were
+ // found in this text, and if it is replaced they are unverified. Free — the
+ // normalized transcript is already in hand.
+ const src = transcriptSourceOf(transcript.source);
const fullTarget: AttributionFreshnessTarget = {
...target,
...(diarization?.generatedAt
? { diarizationGeneratedAt: diarization.generatedAt }
: {}),
+ ...(src ? { transcriptSource: src } : {}),
};
if (!opts.force && isAttributionFresh(existing, fullTarget)) {
return "already-exists";
@@ -290,6 +297,7 @@ export async function attributeOneVideo(
...(diarization?.generatedAt
? { diarizationGeneratedAt: diarization.generatedAt }
: {}),
+ ...(src ? { transcriptSource: src } : {}),
...(chunks !== undefined ? { chunks, chunksOk } : {}),
...(app.metered && costUsd > 0 ? { costUsd } : {}),
durationMs: Date.now() - started,
diff --git a/common/lib/attribution.test.ts b/common/lib/attribution.test.ts
@@ -6,6 +6,7 @@ import {
isAttributionDowngrade,
isAttributionFresh,
attributionSpeechSeconds,
+ transcriptSourceOf,
type AttributionRecord,
} from "./attribution";
import { attributionStatus } from "./attributionStatus";
@@ -164,6 +165,70 @@ test("re-diarizing invalidates the names that pointed at the old clusters", () =
);
});
+// The names are an assertion ABOUT a text. Replace the text — our own whisper
+// transcript over YouTube's auto-captions, which the backfill's auto-transcribe
+// hand-off now makes routine — and the assertion is unverified: different
+// wording, different timings, possibly different speakers named.
+test("our own transcript replacing the auto-captions invalidates the names made from them", () => {
+ const fromVtt = record({ transcriptSource: "vtt" });
+ assert.equal(
+ isAttributionFresh(fromVtt, {
+ ...attributionTarget(CURRENT, "diarized"),
+ transcriptSource: "whisper",
+ }),
+ false,
+ );
+ // Same source: nothing changed, no regeneration.
+ assert.equal(
+ isAttributionFresh(record({ transcriptSource: "whisper" }), {
+ ...attributionTarget(CURRENT, "diarized"),
+ transcriptSource: "whisper",
+ }),
+ true,
+ );
+});
+
+test("a record written before transcriptSource existed is not invalidated by it", () => {
+ // The INVERSE of the diarization guard, and deliberately so: that one fires on
+ // the target alone, this one requires the record to carry the field too. A
+ // record on disk knows nothing about its transcript's source, and reading that
+ // silence as "different" would invalidate every attribution in the corpus the
+ // moment the field shipped. Same rule as sameVersion's absent-means-default.
+ assert.equal(
+ isAttributionFresh(record(), {
+ ...attributionTarget(CURRENT, "diarized"),
+ transcriptSource: "whisper",
+ }),
+ true,
+ );
+});
+
+test("a target that does not know the source does not compare it", () => {
+ // A caller that has not read the video's files cannot assert anything.
+ assert.equal(
+ attributionTarget(CURRENT, "text-only").transcriptSource,
+ undefined,
+ );
+ assert.equal(
+ isAttributionFresh(
+ record({ transcriptSource: "vtt" }),
+ attributionTarget(CURRENT, "diarized"),
+ ),
+ true,
+ );
+});
+
+test("transcriptSourceOf asserts nothing about a source it does not know", () => {
+ assert.equal(transcriptSourceOf("whisper"), "whisper");
+ assert.equal(transcriptSourceOf("vtt"), "vtt");
+ // cues.json's source is pickIndexTranscript's kind, so live_chat never
+ // reaches attribution — but "unknown means do not assert" is the rule, not an
+ // accident of which values happen to occur.
+ assert.equal(transcriptSourceOf("live_chat"), undefined);
+ assert.equal(transcriptSourceOf(null), undefined);
+ assert.equal(transcriptSourceOf(undefined), undefined);
+});
+
test("no record is never fresh", () => {
assert.equal(
isAttributionFresh(null, attributionTarget(CURRENT, "diarized")),
diff --git a/common/lib/attribution.ts b/common/lib/attribution.ts
@@ -109,6 +109,17 @@ export type AttributionProvenance = {
// and nothing on disk could tell. With it, the naming goes stale exactly when
// the thing it names is replaced.
diarizationGeneratedAt?: string;
+ // Which transcript the names were made from — "whisper" (ours) or "vtt"
+ // (YouTube's, usually its ASR).
+ //
+ // Same kind of field as diarizationGeneratedAt above, for the same kind of
+ // reason: the names are an assertion ABOUT a text, and replacing that text
+ // makes the assertion unverified. The backfill's auto-transcribe hand-off makes
+ // this a routine event rather than a rarity — a video whose only transcript was
+ // YouTube ASR gets our own, with different wording, different timings and
+ // possibly different speakers named. Without this field nothing on disk could
+ // tell that had happened.
+ transcriptSource?: AttributionTranscriptSource;
// Text-only lane: how the video was sliced and how much of it survived. Same
// pair digestVideo records, and for the same reason — one 500 from ollama
// costs that chunk, not the 30-call video, and a partial result has to be
@@ -120,6 +131,22 @@ export type AttributionProvenance = {
durationMs?: number;
};
+// Which transcript the names were found in.
+//
+// Narrower than NormalizedTranscript.source on purpose: cues.json's `source` is
+// pickIndexTranscript's `kind` (normalizeTranscript.ts:133), which is only ever
+// "whisper" or "vtt" — never "live_chat".
+export type AttributionTranscriptSource = "whisper" | "vtt";
+
+// Map a recorded/normalized source string onto the two values freshness knows,
+// or undefined for anything else. UNKNOWN MEANS DO NOT ASSERT: an unrecognised
+// value must not be turned into a comparison that invalidates a record.
+export function transcriptSourceOf(
+ source: string | null | undefined,
+): AttributionTranscriptSource | undefined {
+ return source === "whisper" || source === "vtt" ? source : undefined;
+}
+
export type AttributionRecord = {
videoId: string;
// ISO 8601, set when the sidecar is finalized.
@@ -183,6 +210,9 @@ export type AttributionFreshnessTarget = {
// now. Absent means "do not compare" — which is what the text-only lane wants,
// since it never reads diarization at all.
diarizationGeneratedAt?: string;
+ // The source of the transcript on disk right now. Absent means "do not
+ // compare" — a caller that has not read the video's files cannot assert it.
+ transcriptSource?: AttributionTranscriptSource;
};
// The identity the current configuration would produce, from a structurally
@@ -248,6 +278,23 @@ export function isAttributionFresh(
// AttributionProvenance.diarizationGeneratedAt.
return false;
}
+ if (
+ target.transcriptSource !== undefined &&
+ p.transcriptSource !== undefined &&
+ p.transcriptSource !== target.transcriptSource
+ ) {
+ // The text these names were found in has been replaced — our own transcript
+ // over YouTube's auto-captions, typically.
+ //
+ // Note the extra condition, which is the INVERSE of the diarization guard
+ // just above: the RECORD must carry the field too. That is sameVersion's
+ // rationale applied to an optional string rather than a number — a record
+ // written before this field existed knows nothing about its transcript's
+ // source, and reading that silence as "different" would invalidate every
+ // attribution on disk the moment this shipped. Records written from now on
+ // carry it and are compared.
+ return false;
+ }
return (
p.appId === target.appId &&
(p.modelRequested ?? p.model) === target.model &&
diff --git a/common/lib/operations.ts b/common/lib/operations.ts
@@ -119,9 +119,11 @@ import {
ATTRIBUTION_FILENAME,
isAttributionDowngrade,
isAttributionFresh,
+ transcriptSourceOf,
type AttributionFreshnessTarget,
type AttributionMethod,
type AttributionRecord,
+ type AttributionTranscriptSource,
} from "./attribution";
import { loadAttribution, writeAttribution } from "./attribution-server";
import {
@@ -131,6 +133,7 @@ import {
WHISPER_FILENAME,
findSourceMedia,
isVideoTranscribed,
+ pickIndexTranscript,
readVideoDurationSec,
type VideoFiles,
} from "./videoStatus";
@@ -820,6 +823,19 @@ function attributionApplies(files: VideoFiles): boolean {
// includes the generatedAt of the diarization.json it names clusters from, and
// that is a disk read — so it is added inside the one state() branch that has
// already paid for the read. See AttributionProvenance.diarizationGeneratedAt.
+// The transcript half of the per-video identity, for both lanes: which text the
+// names would be made from now. ZERO I/O — pickIndexTranscript reads the
+// already-loaded VideoFiles, and this runs per video per job start, where the
+// cost bar is the channel snapshot's own per-video classification. `{}` when
+// there is no transcript at all, which state() has already ruled out but which
+// must not be turned into a false assertion here.
+function transcriptSourceTarget(
+ files: VideoFiles,
+): { transcriptSource?: AttributionTranscriptSource } {
+ const source = transcriptSourceOf(pickIndexTranscript(files)?.kind);
+ return source ? { transcriptSource: source } : {};
+}
+
function attributionTargetFor(
settings: SiteSettings,
method: AttributionMethod,
@@ -859,7 +875,10 @@ const attributionText: Operation = {
// something better here. Reporting it as work would put this lane in a loop
// of "attempt, refuse, still outstanding" across every pass of a sweep.
if (isAttributionDowngrade(record, "text-only")) return "present";
- return isAttributionFresh(record, target as AttributionFreshnessTarget)
+ return isAttributionFresh(record, {
+ ...(target as AttributionFreshnessTarget),
+ ...transcriptSourceTarget(files),
+ })
? "present"
: "stale";
},
@@ -944,6 +963,7 @@ const attributionDiarized: Operation = {
return isAttributionFresh(record, {
...(target as AttributionFreshnessTarget),
diarizationGeneratedAt: diarization.generatedAt,
+ ...transcriptSourceTarget(files),
})
? "present"
: "stale";