commit a7aa7b1e9a629bad6382fedaa50dc5e72955c7f7
parent b975a7ba447882eef0c465da20cdc865e97fbfc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 29 Jul 2026 20:21:27 -0400
Make tags real: generate them, score them, cover them
Tags had never been generated. Not once. parseTags, a tags schema, a
sections branch in digestVideo and a settings checkbox had all shipped,
and 0 of 109 sidecars contained a tags section — the fourth
built-but-never-run defect this week.
They work, and they are good. 7/7 videos on teamrcn and 4/4 on
steven-crowder+nuxanor-kick came back with 0 warnings: parseTags accepted
every live output. The tags are specific where the chapters are generic —
an RC-car channel produced "losi comp crawler", "savage flux xs",
"mbx6e", against chapters titled "Introduction" three times.
THE SURCHARGE IS 5-15%, NOT ~100%. Both the settings comment and my own
prediction of ~55% were wrong, and the mechanism is worth knowing: a tag
call re-sends the same transcript as the chapter call before it, so it
hits the engine's cached prefix and pays +0.4s of prefill across 4 extra
calls against 22.4s for the first 4. All it pays is decode, and a tag
list is ~30 output tokens where a chapter list is ~250.
The corollary decides Stage 4: run tags in the SAME pass. Measured on the
same video with the model unloaded between runs, tags alone pay full
prefill (28.0s, identical to chapters) — 44% of a whole chapters pass, or
3-9x the marginal cost of just including them now.
Also: digest-validate.ts said the bake-off "generates into a scratch
directory and scores what it generated". It scores in memory and writes
no digests at all.
Quality finding not fixed here: 31.5 tags/video on multi-hour videos but
only 2.4% cross-video reuse, and "antisemitism"/"anti-semitism" land as
two tags. Specific, but the vocabulary does not converge — that is Phase
7's problem, and it is now measured rather than assumed.
editor e2e digest.spec: 14/14 (was 12).
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Diffstat:
5 files changed, 244 insertions(+), 9 deletions(-)
diff --git a/common/bin/digest-validate.ts b/common/bin/digest-validate.ts
@@ -2,10 +2,13 @@
// Score digests that ALREADY EXIST on disk, using the same metric definitions as
// the bake-off harness (common/bin/digest-bakeoff.ts).
//
-// The bake-off generates into a scratch directory and scores what it generated;
-// this reads what a real sweep wrote. That is the difference between "how does
-// this configuration behave on a hand-picked sample" and "how did it behave on
-// the corpus", and the second is what a validation run is for.
+// The bake-off scores IN MEMORY and writes no digests at all; this reads what a
+// real sweep wrote to disk. That is the difference between "how does this
+// configuration behave on a hand-picked sample" and "how did it behave on the
+// corpus", and the second is what a validation run is for.
+//
+// (An earlier version of this comment said the bake-off "generates into a scratch
+// directory and scores what it generated". It does not, and never did.)
//
// Usage: pnpm exec tsx bin/digest-validate.ts <channel> [<channel> ...]
@@ -33,6 +36,27 @@ function normalizeTitle(title: string): string {
return title.trim().toLowerCase().replace(/\s+/g, " ");
}
+// The tag equivalent of isGenericTitle, and deliberately a SEPARATE regex rather
+// than a reuse of it: a generic chapter title and a useless tag are different
+// failures. "Introduction" is a bad chapter name; the tags that make a tag corpus
+// worthless are the ones that describe the MEDIUM or the FORMAT rather than the
+// subject ("video", "podcast", "commentary", "discussion"), because they match
+// every video and so partition nothing.
+//
+// Anchored whole-string, unlike the chapter version's prefix match: a tag is a
+// short noun phrase, so "video production" is a real topic and must not be
+// condemned by the leading word — the first live run produced exactly that tag.
+const GENERIC_TAG_RE =
+ /^(the\s+)?(video|videos|vid|clip|clips|stream|streams|livestream|vod|podcast|episode|show|channel|content|media|footage|recording|broadcast|commentary|discussion|conversation|talk|talking|chat|chatting|interview|review|reaction|reacting|opinion|opinions|thoughts|rant|update|updates|news|misc|miscellaneous|other|general|various|topics?|entertainment|social media|youtube|twitch)$/i;
+
+function isGenericTag(tag: string): boolean {
+ return GENERIC_TAG_RE.test(tag.trim());
+}
+
+function normalizeTag(tag: string): string {
+ return tag.trim().toLowerCase().replace(/\s+/g, " ");
+}
+
type VideoScore = {
slug: string;
durationSeconds: number;
@@ -44,6 +68,20 @@ type VideoScore = {
duplicate: number;
warnings: number;
derived: boolean;
+ // Tags. Kept on the same row as the chapter metrics so one pass over the corpus
+ // scores both, and so `derived` excludes shared digests from both alike.
+ //
+ // `hasTags` distinguishes "this video has no tags section" from "it has one that
+ // came back empty" — before tags were ever generated the whole corpus was the
+ // first case, and a metric that could not tell them apart reported a 0% failure
+ // rate on a capability that had never run once.
+ hasTags: boolean;
+ tagChunks: number;
+ tagChunksOk: number;
+ tags: number;
+ tagGeneric: number;
+ tagDuplicate: number;
+ tagWarnings: number;
};
async function readDuration(videoDir: string): Promise<number> {
@@ -90,6 +128,22 @@ function scoreVideo(
seen.add(n);
}
+ // Tags, same accumulate-then-sum shape. parseTags already de-dups across the
+ // chunk overlap, so a duplicate surviving to disk is a PARSER failure rather
+ // than the expected overlap noise — which is why it is counted at all.
+ const tagSection = record.sections.tags;
+ const tagItems = tagSection?.items ?? [];
+ const tp = tagSection?.provenance;
+ let tagGeneric = 0;
+ let tagDuplicate = 0;
+ const tagsSeen = new Set<string>();
+ for (const t of tagItems) {
+ if (isGenericTag(t.tag)) tagGeneric++;
+ const n = normalizeTag(t.tag);
+ if (tagsSeen.has(n)) tagDuplicate++;
+ tagsSeen.add(n);
+ }
+
return {
slug,
durationSeconds,
@@ -101,6 +155,13 @@ function scoreVideo(
duplicate,
warnings: record.warnings.length,
derived: record.derivedFrom != null,
+ hasTags: tagSection != null,
+ tagChunks: tp?.chunks ?? 0,
+ tagChunksOk: tp?.chunksOk ?? 0,
+ tags: tagItems.length,
+ tagGeneric,
+ tagDuplicate,
+ tagWarnings: record.warnings.filter((w) => w.section === "tags").length,
};
}
@@ -114,6 +175,9 @@ async function main(): Promise<void> {
const videos: VideoScore[] = [];
const byCode = new Map<DigestWarningCode, number>();
const provenanceSeen = new Map<string, number>();
+ // Tag vocabulary, keyed the same way VideoScore.slug is, so the cross-video
+ // reuse metric can be computed after the shared-digest filter has been applied.
+ const tagsBySlug = new Map<string, string[]>();
for (const channelSlug of channels) {
const dataDir = path.join(paths.channelsDir, channelSlug, "data");
@@ -123,7 +187,12 @@ async function main(): Promise<void> {
const record = await loadDigest(videoDir);
if (!record) continue;
const duration = await readDuration(videoDir);
- videos.push(scoreVideo(`${channelSlug}/${id}`, record, duration));
+ const slug = `${channelSlug}/${id}`;
+ videos.push(scoreVideo(slug, record, duration));
+ tagsBySlug.set(
+ slug,
+ (record.sections.tags?.items ?? []).map((t) => normalizeTag(t.tag)),
+ );
for (const w of record.warnings) {
byCode.set(w.code, (byCode.get(w.code) ?? 0) + 1);
}
@@ -176,6 +245,75 @@ async function main(): Promise<void> {
const medianGap = gaps.length > 0 ? gaps[Math.floor(gaps.length / 2)].maxGapSeconds : 0;
console.log(`Median coverage gap: ${toHms(medianGap)}`);
+ // ---- Tags -------------------------------------------------------------
+ //
+ // Reported unconditionally, INCLUDING the "not generated" case. Tags were a
+ // shipped capability for months with zero coverage — parseTags, a schema, a
+ // digestVideo branch and a settings checkbox all existed and 0 of 109 sidecars
+ // had ever contained a tags section. A metrics block that appears only when
+ // there is something to measure is exactly how that stays invisible.
+ console.log("");
+ const withTags = native.filter((v) => v.hasTags);
+ if (withTags.length === 0) {
+ console.log(
+ `Tags: NOT GENERATED for any of these ${native.length} video(s). ` +
+ `Add "tags" to settings.digest.sections to generate them ` +
+ `(measured surcharge in the same pass: 5–15%).`,
+ );
+ } else {
+ const tagSum = (f: (v: VideoScore) => number): number =>
+ withTags.reduce((a, v) => a + f(v), 0);
+ const tags = tagSum((v) => v.tags);
+ const tagChunks = tagSum((v) => v.tagChunks);
+ const tagChunksOk = tagSum((v) => v.tagChunksOk);
+ const tagZeroYield = tagChunks - tagChunksOk;
+ const emptySection = withTags.filter((v) => v.tags === 0).length;
+ console.log(
+ `Tags: ${withTags.length}/${native.length} native video(s) have a tags section.`,
+ );
+ console.log(
+ `Zero-yield tag chunks: ${tagZeroYield}/${tagChunks} (${tagChunks > 0 ? pct(tagZeroYield / tagChunks) : "n/a"})`,
+ );
+ console.log(
+ `Tags per video: ${(tags / withTags.length).toFixed(2)} mean (${tags} total, ` +
+ `min ${Math.min(...withTags.map((v) => v.tags))}, max ${Math.max(...withTags.map((v) => v.tags))})`,
+ );
+ console.log(
+ `Empty tag sections: ${emptySection}/${withTags.length} (${pct(emptySection / withTags.length)})`,
+ );
+ console.log(
+ `Generic tags: ${tagSum((v) => v.tagGeneric)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagGeneric) / tags) : "n/a"})`,
+ );
+ console.log(
+ `Duplicate tags: ${tagSum((v) => v.tagDuplicate)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagDuplicate) / tags) : "n/a"}) — parseTags de-dups, so any of these is a parser bug`,
+ );
+ console.log(`Tag warnings: ${tagSum((v) => v.tagWarnings)}`);
+ // Vocabulary reuse is the whole point of tags: a tag used once is a label, a
+ // tag used across videos is an index. A corpus where every tag is unique has
+ // produced nothing searchable, and no per-video metric above can see that.
+ const freq = new Map<string, number>();
+ for (const v of withTags) {
+ // One count per VIDEO, not per occurrence: a tag repeated inside one video
+ // is already de-duplicated by parseTags, so counting occurrences would just
+ // re-measure that.
+ for (const t of new Set(tagsBySlug.get(v.slug) ?? [])) {
+ freq.set(t, (freq.get(t) ?? 0) + 1);
+ }
+ }
+ const reused = [...freq.values()].filter((n) => n > 1).length;
+ console.log(
+ `Distinct tags: ${freq.size} across ${withTags.length} video(s); ` +
+ `${reused} (${freq.size > 0 ? pct(reused / freq.size) : "n/a"}) appear in more than one video`,
+ );
+ const top = [...freq.entries()]
+ .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0]))
+ .slice(0, 12);
+ if (top.length > 0) {
+ console.log("Most reused tags:");
+ for (const [tag, n] of top) console.log(` ${String(n).padStart(4)}× ${tag}`);
+ }
+ }
+
console.log("");
console.log("Rejections by guard:");
if (byCode.size === 0) console.log(" (none)");
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -252,8 +252,18 @@ export type DigestSettings = {
// Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever
// consulted for a metered app.
spendCapUsd: number;
- // Which sections a sweep generates. Chapters alone is the default: tags double
- // the call count for a smaller payoff.
+ // Which sections a sweep generates.
+ //
+ // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that
+ // is not a contradiction: a tag call sends the same transcript as the chapter
+ // call before it, so it hits the engine's cached prefix and pays essentially no
+ // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it
+ // pays is decode, and a tag list is ~30 output tokens where a chapter list is
+ // ~200–290.
+ //
+ // The corollary matters more than the number: run them in the SAME pass. Tags
+ // generated later, on their own, pay full prefill again — measured at 44% of a
+ // whole chapters pass, i.e. 3–9× the marginal cost of just including them now.
sections: DigestSectionKind[];
// How each chunk's transcript markers are numbered — see DigestTimestampMode.
// Was a scored variable in the bake-off rather than a pre-applied fix; the
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -470,8 +470,13 @@ export function SettingsForm({ initial, apps, digestApps }: Props) {
))}
</span>
<span className="text-xs text-muted-foreground">
- Chapters alone is the default: tags roughly double the model calls
- for a smaller payoff. Unchecking everything falls back to chapters
+ Tags double the model calls but cost only 5–15% more time: a tag call
+ re-sends the same transcript as the chapter call before it, so it
+ reuses the engine's cached prompt and pays for decoding ~30
+ output tokens instead of ~250. Generating them <em>later</em>, in
+ their own pass, pays for the transcript again — measured at 44% of a
+ full chapters pass. So if you want tags at all, check them now
+ rather than after. Unchecking everything falls back to chapters
rather than generating nothing.
</span>
</label>
diff --git a/editor/e2e/digest.spec.ts b/editor/e2e/digest.spec.ts
@@ -666,3 +666,77 @@ test("corrections saved from the panel land in the overrides sidecar only", asyn
"Renamed from the panel",
);
});
+
+// ---------------------------------------------------------------------------
+// Tags — a capability that shipped and then never ran.
+//
+// `parseTags`, a tags JSON schema, a `sections` branch in digestVideo and a
+// per-section checkbox in the settings form all existed for months, and 0 of the
+// 109 sidecars on disk had ever contained a tags section. No metric counted them,
+// no bake-off round scored them, and nothing here covered them — so the whole
+// path could have been broken end to end and every test would still have passed.
+// This is the test that makes that impossible to repeat.
+// ---------------------------------------------------------------------------
+
+test("a digest run with tags enabled writes a tags section", async ({ page }) => {
+ await resetData(null);
+ // The one thing this spec varies. Chapters stay on so the test also proves the
+ // two sections coexist in ONE record rather than overwriting each other —
+ // writeDigestSection merges per section, and a regression there would silently
+ // cost whichever section was written first.
+ await writeSettings(digestSettings({ sections: ["chapters", "tags"] }));
+ await writeChannelConfig(CHANNEL);
+ await writeDigestVideo({ channelSlug: CHANNEL, videoId: VIDEO });
+
+ await runDigest(page, CHANNEL);
+
+ const record = await readJson<DigestRecord>(digestPath(CHANNEL, VIDEO));
+
+ const tags = record.sections.tags?.items ?? [];
+ expect(tags.length).toBeGreaterThan(0);
+ // Every tag is lowercase, non-empty and carries an id — parseTags' contract.
+ for (const t of tags) {
+ expect(t.tag).toBe(t.tag.toLowerCase());
+ expect(t.tag.trim().length).toBeGreaterThan(0);
+ expect(t.id).toBeTruthy();
+ expect(t.decidedBy).toBe("ai");
+ }
+ // Deduplicated: chunks overlap, so the same tag arrives repeatedly and the
+ // parser is responsible for collapsing it.
+ expect(new Set(tags.map((t) => t.id)).size).toBe(tags.length);
+
+ // Both sections present, each with its OWN provenance — they are generated by
+ // separate calls and expire independently.
+ expect(record.sections.chapters?.items.length).toBeGreaterThan(0);
+ expect(record.sections.tags?.provenance.appId).toBe("ollama-direct");
+ expect(record.sections.tags?.provenance.chunks).toBeGreaterThan(0);
+ expect(record.sections.tags?.provenance.chunksOk).toBeGreaterThan(0);
+
+ // The timing breakdown must NOT reach the sidecar: it is diagnostic only, and a
+ // field landing in provenance would change the freshness identity and silently
+ // invalidate every digest in the corpus.
+ const provenance = record.sections.tags?.provenance as
+ | Record<string, unknown>
+ | undefined;
+ for (const field of ["loadMs", "promptEvalMs", "evalMs", "totalMs", "durationMs"]) {
+ expect(provenance).not.toHaveProperty(field);
+ }
+});
+
+// Tags are OPT-IN and must stay that way: the default is chapters alone, and a
+// sweep that quietly started generating tags would be spending GPU-weeks nobody
+// asked for.
+test("tags are not generated unless the setting asks for them", async ({
+ page,
+}) => {
+ await resetData(null);
+ await writeSettings(digestSettings()); // sections: ["chapters"]
+ await writeChannelConfig(CHANNEL);
+ await writeDigestVideo({ channelSlug: CHANNEL, videoId: VIDEO });
+
+ await runDigest(page, CHANNEL);
+
+ const record = await readJson<DigestRecord>(digestPath(CHANNEL, VIDEO));
+ expect(record.sections.chapters?.items.length).toBeGreaterThan(0);
+ expect(record.sections.tags).toBeUndefined();
+});
diff --git a/editor/e2e/fixtures/ollama-stub.mjs b/editor/e2e/fixtures/ollama-stub.mjs
@@ -138,6 +138,14 @@ const server = createServer(async (req, res) => {
done: true,
prompt_eval_count: 100,
eval_count: 50,
+ // NANOSECONDS, as the real API reports them. Included so the timing
+ // breakdown digestApps.ts records is exercised by e2e rather than only by
+ // a live GPU — a run that silently stopped parsing these would otherwise
+ // look identical to one that worked.
+ total_duration: 1_500_000_000,
+ load_duration: 300_000_000,
+ prompt_eval_duration: 400_000_000,
+ eval_duration: 800_000_000,
}),
);
return;