Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit a4f7145efbd053cfc8fc353eafc9b4674da516df
parent ce63f663c012fd7a3e572011fd5a9bedebbe508b
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 28 Sep 2026 18:00:50 -0400

mcp: pin get_video_metadata's stats block on a recomputed stat

No truncation note and the real cue count for a stat as buildStats now
recomputes it, and the note still fires for a transcript that stops early.

Co-Authored-By: Claude Fable 5.1 <noreply@anthropic.com>

Diffstat:
Mmcp/src/search.test.ts | 76++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
1 file changed, 76 insertions(+), 0 deletions(-)

diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts @@ -15,6 +15,7 @@ import type { } from "yt-dlp-transcript-common/lib/posts"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; import type { Cue } from "yt-dlp-transcript-common/lib/vtt"; +import type { VideoStat } from "yt-dlp-transcript-common/lib/stats"; import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups"; import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery"; import type { @@ -1296,6 +1297,81 @@ test("server: get_video_metadata reports a missing id as an error", async () => await client.close(); }); +// The "## Stats" block is read straight off the archive's stats pages, so its +// "covers only N% — truncated" warning is exactly as good as the stat. Before +// stats schema 6 the stat of a video transcribed after it was first indexed +// stayed at "no transcript, coverage 0" for good, and this warned that a +// complete transcript was truncated. buildStats.test.ts (a) pins the stat now +// being recomputed; this pins what the tool then says about it. +function statFor( + record: TranscriptDetail, + s: Pick<VideoStat, "hasTranscript" | "cueCount" | "coverage" | "transcribedDate">, +): VideoStat { + return { + slug: record.slug, + id: record.id, + channelSlug: record.channelSlug, + channel: "Channel A", + title: record.title, + platform: "youtube", + uploadDate: record.uploadDate, + downloadedDate: "20260711", + timestamp: null, + duration: 3600, + viewCount: null, + likeCount: null, + commentCount: null, + channelFollowerCount: null, + categories: [], + tags: [], + language: null, + isLivestream: false, + mediaType: "video", + status: "available", + ...s, + }; +} + +class StatsStubSource extends StubSource { + constructor(private stat: VideoStat) { + super(); + } + async statsIndex(): Promise<ReadonlyMap<string, VideoStat>> { + return new Map([[this.stat.slug, this.stat]]); + } +} + +test("server: get_video_metadata on a recomputed stat reports the cues and no truncation", async () => { + const a1 = CHAN_A[0]; + const client = await connectClient( + new StatsStubSource( + statFor(a1, { hasTranscript: true, cueCount: 7461, coverage: 0.998, transcribedDate: "20260918" }), + ), + ); + const out = firstText( + await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }), + ); + assert.match(out, /## Stats/); + assert.match(out, /- transcript cues: 7461/); + assert.match(out, /- transcribed: /); + assert.doesNotMatch(out, /COVERS ONLY/); + await client.close(); +}); + +test("server: get_video_metadata still warns for a transcript that really stops early", async () => { + const a1 = CHAN_A[0]; + const client = await connectClient( + new StatsStubSource( + statFor(a1, { hasTranscript: true, cueCount: 900, coverage: 0.41, transcribedDate: "20260918" }), + ), + ); + const out = firstText( + await client.callTool({ name: "get_video_metadata", arguments: { video_id: "a1" } }), + ); + assert.match(out, /TRANSCRIPT COVERS ONLY 41% OF THE RUNTIME/); + await client.close(); +}); + test("server: open_link decodes AND searches in one call, returning the handle", async () => { const client = await connectRegistry(); const out = firstText(