Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a7aa7b1e9a629bad6382fedaa50dc5e72955c7f7
parent b975a7ba447882eef0c465da20cdc865e97fbfc0
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Wed, 29 Jul 2026 20:21:27 -0400

Make tags real: generate them, score them, cover them

Tags had never been generated. Not once. parseTags, a tags schema, a
sections branch in digestVideo and a settings checkbox had all shipped,
and 0 of 109 sidecars contained a tags section — the fourth
built-but-never-run defect this week.

They work, and they are good. 7/7 videos on teamrcn and 4/4 on
steven-crowder+nuxanor-kick came back with 0 warnings: parseTags accepted
every live output. The tags are specific where the chapters are generic —
an RC-car channel produced "losi comp crawler", "savage flux xs",
"mbx6e", against chapters titled "Introduction" three times.

THE SURCHARGE IS 5-15%, NOT ~100%. Both the settings comment and my own
prediction of ~55% were wrong, and the mechanism is worth knowing: a tag
call re-sends the same transcript as the chapter call before it, so it
hits the engine's cached prefix and pays +0.4s of prefill across 4 extra
calls against 22.4s for the first 4. All it pays is decode, and a tag
list is ~30 output tokens where a chapter list is ~250.

The corollary decides Stage 4: run tags in the SAME pass. Measured on the
same video with the model unloaded between runs, tags alone pay full
prefill (28.0s, identical to chapters) — 44% of a whole chapters pass, or
3-9x the marginal cost of just including them now.

Also: digest-validate.ts said the bake-off "generates into a scratch
directory and scores what it generated". It scores in memory and writes
no digests at all.

Quality finding not fixed here: 31.5 tags/video on multi-hour videos but
only 2.4% cross-video reuse, and "antisemitism"/"anti-semitism" land as
two tags. Specific, but the vocabulary does not converge — that is Phase
7's problem, and it is now measured rather than assumed.

editor e2e digest.spec: 14/14 (was 12).

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>

Diffstat:
Mcommon/bin/digest-validate.ts | 148++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++---
Mcommon/lib/settings.ts | 14++++++++++++--
Meditor/app/settings/components/SettingsForm.tsx | 9+++++++--
Meditor/e2e/digest.spec.ts | 74++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Meditor/e2e/fixtures/ollama-stub.mjs | 8++++++++
5 files changed, 244 insertions(+), 9 deletions(-)

diff --git a/common/bin/digest-validate.ts b/common/bin/digest-validate.ts @@ -2,10 +2,13 @@ // Score digests that ALREADY EXIST on disk, using the same metric definitions as // the bake-off harness (common/bin/digest-bakeoff.ts). // -// The bake-off generates into a scratch directory and scores what it generated; -// this reads what a real sweep wrote. That is the difference between "how does -// this configuration behave on a hand-picked sample" and "how did it behave on -// the corpus", and the second is what a validation run is for. +// The bake-off scores IN MEMORY and writes no digests at all; this reads what a +// real sweep wrote to disk. That is the difference between "how does this +// configuration behave on a hand-picked sample" and "how did it behave on the +// corpus", and the second is what a validation run is for. +// +// (An earlier version of this comment said the bake-off "generates into a scratch +// directory and scores what it generated". It does not, and never did.) // // Usage: pnpm exec tsx bin/digest-validate.ts <channel> [<channel> ...] @@ -33,6 +36,27 @@ function normalizeTitle(title: string): string { return title.trim().toLowerCase().replace(/\s+/g, " "); } +// The tag equivalent of isGenericTitle, and deliberately a SEPARATE regex rather +// than a reuse of it: a generic chapter title and a useless tag are different +// failures. "Introduction" is a bad chapter name; the tags that make a tag corpus +// worthless are the ones that describe the MEDIUM or the FORMAT rather than the +// subject ("video", "podcast", "commentary", "discussion"), because they match +// every video and so partition nothing. +// +// Anchored whole-string, unlike the chapter version's prefix match: a tag is a +// short noun phrase, so "video production" is a real topic and must not be +// condemned by the leading word — the first live run produced exactly that tag. +const GENERIC_TAG_RE = + /^(the\s+)?(video|videos|vid|clip|clips|stream|streams|livestream|vod|podcast|episode|show|channel|content|media|footage|recording|broadcast|commentary|discussion|conversation|talk|talking|chat|chatting|interview|review|reaction|reacting|opinion|opinions|thoughts|rant|update|updates|news|misc|miscellaneous|other|general|various|topics?|entertainment|social media|youtube|twitch)$/i; + +function isGenericTag(tag: string): boolean { + return GENERIC_TAG_RE.test(tag.trim()); +} + +function normalizeTag(tag: string): string { + return tag.trim().toLowerCase().replace(/\s+/g, " "); +} + type VideoScore = { slug: string; durationSeconds: number; @@ -44,6 +68,20 @@ type VideoScore = { duplicate: number; warnings: number; derived: boolean; + // Tags. Kept on the same row as the chapter metrics so one pass over the corpus + // scores both, and so `derived` excludes shared digests from both alike. + // + // `hasTags` distinguishes "this video has no tags section" from "it has one that + // came back empty" — before tags were ever generated the whole corpus was the + // first case, and a metric that could not tell them apart reported a 0% failure + // rate on a capability that had never run once. + hasTags: boolean; + tagChunks: number; + tagChunksOk: number; + tags: number; + tagGeneric: number; + tagDuplicate: number; + tagWarnings: number; }; async function readDuration(videoDir: string): Promise<number> { @@ -90,6 +128,22 @@ function scoreVideo( seen.add(n); } + // Tags, same accumulate-then-sum shape. parseTags already de-dups across the + // chunk overlap, so a duplicate surviving to disk is a PARSER failure rather + // than the expected overlap noise — which is why it is counted at all. + const tagSection = record.sections.tags; + const tagItems = tagSection?.items ?? []; + const tp = tagSection?.provenance; + let tagGeneric = 0; + let tagDuplicate = 0; + const tagsSeen = new Set<string>(); + for (const t of tagItems) { + if (isGenericTag(t.tag)) tagGeneric++; + const n = normalizeTag(t.tag); + if (tagsSeen.has(n)) tagDuplicate++; + tagsSeen.add(n); + } + return { slug, durationSeconds, @@ -101,6 +155,13 @@ function scoreVideo( duplicate, warnings: record.warnings.length, derived: record.derivedFrom != null, + hasTags: tagSection != null, + tagChunks: tp?.chunks ?? 0, + tagChunksOk: tp?.chunksOk ?? 0, + tags: tagItems.length, + tagGeneric, + tagDuplicate, + tagWarnings: record.warnings.filter((w) => w.section === "tags").length, }; } @@ -114,6 +175,9 @@ async function main(): Promise<void> { const videos: VideoScore[] = []; const byCode = new Map<DigestWarningCode, number>(); const provenanceSeen = new Map<string, number>(); + // Tag vocabulary, keyed the same way VideoScore.slug is, so the cross-video + // reuse metric can be computed after the shared-digest filter has been applied. + const tagsBySlug = new Map<string, string[]>(); for (const channelSlug of channels) { const dataDir = path.join(paths.channelsDir, channelSlug, "data"); @@ -123,7 +187,12 @@ async function main(): Promise<void> { const record = await loadDigest(videoDir); if (!record) continue; const duration = await readDuration(videoDir); - videos.push(scoreVideo(`${channelSlug}/${id}`, record, duration)); + const slug = `${channelSlug}/${id}`; + videos.push(scoreVideo(slug, record, duration)); + tagsBySlug.set( + slug, + (record.sections.tags?.items ?? []).map((t) => normalizeTag(t.tag)), + ); for (const w of record.warnings) { byCode.set(w.code, (byCode.get(w.code) ?? 0) + 1); } @@ -176,6 +245,75 @@ async function main(): Promise<void> { const medianGap = gaps.length > 0 ? gaps[Math.floor(gaps.length / 2)].maxGapSeconds : 0; console.log(`Median coverage gap: ${toHms(medianGap)}`); + // ---- Tags ------------------------------------------------------------- + // + // Reported unconditionally, INCLUDING the "not generated" case. Tags were a + // shipped capability for months with zero coverage — parseTags, a schema, a + // digestVideo branch and a settings checkbox all existed and 0 of 109 sidecars + // had ever contained a tags section. A metrics block that appears only when + // there is something to measure is exactly how that stays invisible. + console.log(""); + const withTags = native.filter((v) => v.hasTags); + if (withTags.length === 0) { + console.log( + `Tags: NOT GENERATED for any of these ${native.length} video(s). ` + + `Add "tags" to settings.digest.sections to generate them ` + + `(measured surcharge in the same pass: 5–15%).`, + ); + } else { + const tagSum = (f: (v: VideoScore) => number): number => + withTags.reduce((a, v) => a + f(v), 0); + const tags = tagSum((v) => v.tags); + const tagChunks = tagSum((v) => v.tagChunks); + const tagChunksOk = tagSum((v) => v.tagChunksOk); + const tagZeroYield = tagChunks - tagChunksOk; + const emptySection = withTags.filter((v) => v.tags === 0).length; + console.log( + `Tags: ${withTags.length}/${native.length} native video(s) have a tags section.`, + ); + console.log( + `Zero-yield tag chunks: ${tagZeroYield}/${tagChunks} (${tagChunks > 0 ? pct(tagZeroYield / tagChunks) : "n/a"})`, + ); + console.log( + `Tags per video: ${(tags / withTags.length).toFixed(2)} mean (${tags} total, ` + + `min ${Math.min(...withTags.map((v) => v.tags))}, max ${Math.max(...withTags.map((v) => v.tags))})`, + ); + console.log( + `Empty tag sections: ${emptySection}/${withTags.length} (${pct(emptySection / withTags.length)})`, + ); + console.log( + `Generic tags: ${tagSum((v) => v.tagGeneric)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagGeneric) / tags) : "n/a"})`, + ); + console.log( + `Duplicate tags: ${tagSum((v) => v.tagDuplicate)}/${tags} (${tags > 0 ? pct(tagSum((v) => v.tagDuplicate) / tags) : "n/a"}) — parseTags de-dups, so any of these is a parser bug`, + ); + console.log(`Tag warnings: ${tagSum((v) => v.tagWarnings)}`); + // Vocabulary reuse is the whole point of tags: a tag used once is a label, a + // tag used across videos is an index. A corpus where every tag is unique has + // produced nothing searchable, and no per-video metric above can see that. + const freq = new Map<string, number>(); + for (const v of withTags) { + // One count per VIDEO, not per occurrence: a tag repeated inside one video + // is already de-duplicated by parseTags, so counting occurrences would just + // re-measure that. + for (const t of new Set(tagsBySlug.get(v.slug) ?? [])) { + freq.set(t, (freq.get(t) ?? 0) + 1); + } + } + const reused = [...freq.values()].filter((n) => n > 1).length; + console.log( + `Distinct tags: ${freq.size} across ${withTags.length} video(s); ` + + `${reused} (${freq.size > 0 ? pct(reused / freq.size) : "n/a"}) appear in more than one video`, + ); + const top = [...freq.entries()] + .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])) + .slice(0, 12); + if (top.length > 0) { + console.log("Most reused tags:"); + for (const [tag, n] of top) console.log(` ${String(n).padStart(4)}× ${tag}`); + } + } + console.log(""); console.log("Rejections by guard:"); if (byCode.size === 0) console.log(" (none)"); diff --git a/common/lib/settings.ts b/common/lib/settings.ts @@ -252,8 +252,18 @@ export type DigestSettings = { // Hard ceiling on cumulative metered spend per job, USD. 0 = no cap. Only ever // consulted for a metered app. spendCapUsd: number; - // Which sections a sweep generates. Chapters alone is the default: tags double - // the call count for a smaller payoff. + // Which sections a sweep generates. + // + // Tags DOUBLE THE CALL COUNT but cost only 5–15% more TIME, measured, and that + // is not a contradiction: a tag call sends the same transcript as the chapter + // call before it, so it hits the engine's cached prefix and pays essentially no + // prefill (+0.4 s across 4 extra calls, against 22.4 s for the first 4). All it + // pays is decode, and a tag list is ~30 output tokens where a chapter list is + // ~200–290. + // + // The corollary matters more than the number: run them in the SAME pass. Tags + // generated later, on their own, pay full prefill again — measured at 44% of a + // whole chapters pass, i.e. 3–9× the marginal cost of just including them now. sections: DigestSectionKind[]; // How each chunk's transcript markers are numbered — see DigestTimestampMode. // Was a scored variable in the bake-off rather than a pre-applied fix; the diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx @@ -470,8 +470,13 @@ export function SettingsForm({ initial, apps, digestApps }: Props) { ))} </span> <span className="text-xs text-muted-foreground"> - Chapters alone is the default: tags roughly double the model calls - for a smaller payoff. Unchecking everything falls back to chapters + Tags double the model calls but cost only 5–15% more time: a tag call + re-sends the same transcript as the chapter call before it, so it + reuses the engine&apos;s cached prompt and pays for decoding ~30 + output tokens instead of ~250. Generating them <em>later</em>, in + their own pass, pays for the transcript again — measured at 44% of a + full chapters pass. So if you want tags at all, check them now + rather than after. Unchecking everything falls back to chapters rather than generating nothing. </span> </label> diff --git a/editor/e2e/digest.spec.ts b/editor/e2e/digest.spec.ts @@ -666,3 +666,77 @@ test("corrections saved from the panel land in the overrides sidecar only", asyn "Renamed from the panel", ); }); + +// --------------------------------------------------------------------------- +// Tags — a capability that shipped and then never ran. +// +// `parseTags`, a tags JSON schema, a `sections` branch in digestVideo and a +// per-section checkbox in the settings form all existed for months, and 0 of the +// 109 sidecars on disk had ever contained a tags section. No metric counted them, +// no bake-off round scored them, and nothing here covered them — so the whole +// path could have been broken end to end and every test would still have passed. +// This is the test that makes that impossible to repeat. +// --------------------------------------------------------------------------- + +test("a digest run with tags enabled writes a tags section", async ({ page }) => { + await resetData(null); + // The one thing this spec varies. Chapters stay on so the test also proves the + // two sections coexist in ONE record rather than overwriting each other — + // writeDigestSection merges per section, and a regression there would silently + // cost whichever section was written first. + await writeSettings(digestSettings({ sections: ["chapters", "tags"] })); + await writeChannelConfig(CHANNEL); + await writeDigestVideo({ channelSlug: CHANNEL, videoId: VIDEO }); + + await runDigest(page, CHANNEL); + + const record = await readJson<DigestRecord>(digestPath(CHANNEL, VIDEO)); + + const tags = record.sections.tags?.items ?? []; + expect(tags.length).toBeGreaterThan(0); + // Every tag is lowercase, non-empty and carries an id — parseTags' contract. + for (const t of tags) { + expect(t.tag).toBe(t.tag.toLowerCase()); + expect(t.tag.trim().length).toBeGreaterThan(0); + expect(t.id).toBeTruthy(); + expect(t.decidedBy).toBe("ai"); + } + // Deduplicated: chunks overlap, so the same tag arrives repeatedly and the + // parser is responsible for collapsing it. + expect(new Set(tags.map((t) => t.id)).size).toBe(tags.length); + + // Both sections present, each with its OWN provenance — they are generated by + // separate calls and expire independently. + expect(record.sections.chapters?.items.length).toBeGreaterThan(0); + expect(record.sections.tags?.provenance.appId).toBe("ollama-direct"); + expect(record.sections.tags?.provenance.chunks).toBeGreaterThan(0); + expect(record.sections.tags?.provenance.chunksOk).toBeGreaterThan(0); + + // The timing breakdown must NOT reach the sidecar: it is diagnostic only, and a + // field landing in provenance would change the freshness identity and silently + // invalidate every digest in the corpus. + const provenance = record.sections.tags?.provenance as + | Record<string, unknown> + | undefined; + for (const field of ["loadMs", "promptEvalMs", "evalMs", "totalMs", "durationMs"]) { + expect(provenance).not.toHaveProperty(field); + } +}); + +// Tags are OPT-IN and must stay that way: the default is chapters alone, and a +// sweep that quietly started generating tags would be spending GPU-weeks nobody +// asked for. +test("tags are not generated unless the setting asks for them", async ({ + page, +}) => { + await resetData(null); + await writeSettings(digestSettings()); // sections: ["chapters"] + await writeChannelConfig(CHANNEL); + await writeDigestVideo({ channelSlug: CHANNEL, videoId: VIDEO }); + + await runDigest(page, CHANNEL); + + const record = await readJson<DigestRecord>(digestPath(CHANNEL, VIDEO)); + expect(record.sections.chapters?.items.length).toBeGreaterThan(0); + expect(record.sections.tags).toBeUndefined(); +}); diff --git a/editor/e2e/fixtures/ollama-stub.mjs b/editor/e2e/fixtures/ollama-stub.mjs @@ -138,6 +138,14 @@ const server = createServer(async (req, res) => { done: true, prompt_eval_count: 100, eval_count: 50, + // NANOSECONDS, as the real API reports them. Included so the timing + // breakdown digestApps.ts records is exercised by e2e rather than only by + // a live GPU — a run that silently stopped parsing these would otherwise + // look identical to one that worked. + total_duration: 1_500_000_000, + load_duration: 300_000_000, + prompt_eval_duration: 400_000_000, + eval_duration: 800_000_000, }), ); return;