Archilyzer · Source

archilyzer

Archilyzer
git clone https://archilyzer.pages.dev/source/archilyzer.git
Log | Files | Refs | README | LICENSE

commit 5f15c6ad6d61a212ac52f6a204ca1d03a2017c65
parent 52d377e6956caf633b1b48f63ca943af6324cfe5
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date:   Mon, 21 Sep 2026 13:18:58 -0400

tags S2.10: the MCP learns curated tags

`tags: string[]` on search_transcripts and enumerate_matches, ORed, parsed
beside the other filters. A malformed id is REPORTED rather than dropped:
dropping the only id in the list would widen the scan back to the whole
corpus and read as a legitimate result — the same failure a typo'd state
would cause.

`list_tags` (right after list_channels) returns the vocabulary a source
publishes, grouped, with this site's count and the per-channel breakdown, so
an id is never guessed:

    4 curated tag(s) in Anilyzer:

    ### Eva
    - eva-collab · Collab · 7 video(s) — legal-mindset 5, nux-taku 2
    - eva-in-chat · In chat · 64 video(s) — legal-mindset 64

    ### (ungrouped)
    - lolcow · Lolcow · 3 video(s) — destiny 3

The empty case is answered in WORDS, in three places, because `[]` reads as
an answer and this one is not: list_tags says the site publishes none and
why, resolve_source says it on sight, and a search that filtered by a tag
anyway gets a footer warning that the result "is not evidence of absence".
A tag the source does not publish is named too.

Coverage is untouched by any of this. The page planner only ever PRUNES, and
a video its index has never heard of still gets its page read — the
invariant at search.ts's filter-first note — so a stale or pre-spec-4
summaries set costs time, never a missed hit. Both halves are pinned in
scanPlan.test.ts.

Co-Authored-By: Claude Opus <noreply@anthropic.com>

Diffstat:
Mmcp/src/instructions.ts | 7+++++--
Mmcp/src/protocol.test.ts | 91++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
Mmcp/src/scanPlan.test.ts | 70++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++
Mmcp/src/search.test.ts | 34++++++++++++++++++++++++++++++++++
Mmcp/src/server.ts | 171++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++++-
5 files changed, 369 insertions(+), 4 deletions(-)

diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts @@ -192,8 +192,11 @@ export function buildSweepInstructions( `\`["deleted","private","members_only","unlisted","maybe_missing"]\` for ` + `"what did the videos that are now GONE say"), \`date_from\`/` + `\`date_to\`, \`media_type\`, \`age\`, \`exclude\` (video-level NOT, ` + - `for "cup" but not "world cup"), and \`scopes\` to search descriptions, ` + - `tags or live chat instead of captions. Use them when the question ` + + `for "cup" but not "world cup"), \`tags\` — the operator's CURATED ` + + `per-video tags, cutting across channels ("every stream where X is on ` + + `mic"); call \`list_tags\` for the ids this corpus publishes rather ` + + `than guessing one — and \`scopes\` to search descriptions, ` + + `keywords or live chat instead of captions. Use them when the question ` + `implies them: a filtered scan reads only the shard pages that can hold ` + `a match, which is the difference between seconds and a minute per ` + `query — and it makes the answer narrower and more honest at the same ` + diff --git a/mcp/src/protocol.test.ts b/mcp/src/protocol.test.ts @@ -32,6 +32,7 @@ const TSX = path.join(HERE, "..", "node_modules", ".bin", "tsx"); // or an accidental reshuffle has to be a deliberate edit here. const EXPECTED_TOOLS = [ "list_channels", + "list_tags", "search_transcripts", "enumerate_matches", "get_transcript", @@ -47,7 +48,7 @@ const EXPECTED_TOOLS = [ ]; // A minimal composed public dir: one channel, two videos, real shard layout. -async function writeFixture(): Promise<string> { +async function writeFixture(opts: { tags?: boolean } = {}): Promise<string> { const dir = await mkdtemp(path.join(os.tmpdir(), "mcp-protocol-")); await writeFile( path.join(dir, "corpus.json"), @@ -68,6 +69,35 @@ async function writeFixture(): Promise<string> { slugToPage: { a1: 0, a2: 0 }, }), ); + // Curated tags, opt-in per fixture. The DEFAULT is a site with no + // /tags.json, because that is what every site built before corpus spec 4 + // looks like and it is the path most likely to be got wrong. + if (opts.tags) { + await writeFile( + path.join(dir, "tags.json"), + JSON.stringify({ + version: 1, + tags: [ + { + id: "eva-collab", + label: "Collab", + group: "eva", + groupLabel: "Eva", + order: 1, + count: 1, + channels: { "chan-a": 1 }, + }, + { + id: "loose-tag", + label: "Loose", + order: 2, + count: 1, + channels: { "chan-a": 1 }, + }, + ], + }), + ); + } const rec = (id: string, title: string, text: string) => ({ slug: `chan-a/${id}`, id, @@ -181,6 +211,65 @@ test("stdio: the legacy (2025) handshake still serves the same tools", async (t) assert.match(firstText(res), /Tea two/); }); +test("stdio: list_tags reports the vocabulary a site publishes", async (t) => { + const dir = await writeFixture({ tags: true }); + t.after(() => rm(dir, { recursive: true, force: true })); + const s = await connect(dir, { mode: "auto" }); + t.after(() => s.close()); + + const out = firstText(await s.client.callTool({ name: "list_tags", arguments: {} })); + assert.match(out, /2 curated tag\(s\)/); + // Grouped, with the label, this site's count and the per-channel breakdown. + assert.match(out, /### Eva/); + assert.match(out, /- eva-collab · Collab · 1 video\(s\) — chan-a 1/); + // An ungrouped tag still appears, under a bucket that says so. + assert.match(out, /### \(ungrouped\)/); + assert.match(out, /- loose-tag · Loose · 1 video\(s\) — chan-a 1/); + // And it names the filter it feeds, so the id never has to be guessed. + assert.match(out, /tags:\["eva-collab"\]/); +}); + +test("stdio: a site with no /tags.json says so instead of returning nothing", async (t) => { + const dir = await writeFixture(); + t.after(() => rm(dir, { recursive: true, force: true })); + const s = await connect(dir, { mode: "auto" }); + t.after(() => s.close()); + + const out = firstText(await s.client.callTool({ name: "list_tags", arguments: {} })); + assert.match(out, /publishes no curated tags/); + assert.match(out, /corpus spec 4/); + + // …and a search that filters by a tag anyway is warned, not quietly empty. + // This is the pre-spec-4 path: correct (no record carries a tag) but + // indistinguishable from "searched and found nothing" without the warning. + const search = firstText( + await s.client.callTool({ + name: "search_transcripts", + arguments: { query: "coffee", tags: ["eva-collab"] }, + }), + ); + assert.match(search, /No matches for "coffee"/); + assert.match(search, /publishes no \/tags\.json/); + assert.match(search, /not evidence of absence/); +}); + +test("stdio: a tag the site does not publish is named, not silently empty", async (t) => { + const dir = await writeFixture({ tags: true }); + t.after(() => rm(dir, { recursive: true, force: true })); + const s = await connect(dir, { mode: "auto" }); + t.after(() => s.close()); + + const out = firstText( + await s.client.callTool({ + name: "search_transcripts", + arguments: { query: "coffee", tags: ["eva-collab", "not-a-real-tag"] }, + }), + ); + assert.match(out, /tag\(s\) this source does not publish: not-a-real-tag/); + // The filter it DID understand is echoed too. + assert.match(out, /tags: eva-collab OR not-a-real-tag/); +}); + test("stdio: prompts are served on both eras", async (t) => { const dir = await writeFixture(); t.after(() => rm(dir, { recursive: true, force: true })); diff --git a/mcp/src/scanPlan.test.ts b/mcp/src/scanPlan.test.ts @@ -50,6 +50,7 @@ function video( isLivestream?: boolean; deleted?: boolean; text?: string; + curatedTags?: string[]; } = {}, ): TranscriptDetail { return { @@ -66,6 +67,7 @@ function video( webpageUrl: `https://www.youtube.com/watch?v=${id}`, description: "", tags: [], + ...(opts.curatedTags ? { curatedTags: opts.curatedTags } : {}), cues: cues([10, opts.text ?? "they filed a lawsuit today"]), }; } @@ -160,6 +162,9 @@ function indexed( uploadDate: rec.uploadDate, isLivestream: rec.isLivestream === true, ageRestricted: rec.ageRestricted === true, + ...(rec.curatedTags && rec.curatedTags.length > 0 + ? { curatedTags: rec.curatedTags } + : {}), }, ]; } @@ -380,6 +385,71 @@ test("a states filter reaches only the pages holding videos in those states", as assert.equal(result.hits[0].videoId, "gone1"); }); +test("a curated-tag filter reaches only the pages holding tagged videos", async () => { + const p0 = [video("chan", "plain")]; + const p1 = [video("chan", "collab", { curatedTags: ["eva-collab"] })]; + const p2 = [video("chan", "chatty", { curatedTags: ["eva-in-chat"] })]; + const source = new CountingSource( + { chan: [p0, p1, p2] }, + new Map([indexed(p0[0]), indexed(p1[0]), indexed(p2[0])]), + ); + const channels = await source.listChannels(); + const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] }; + + const plan = await buildScanPlan(source, channels, filters); + assert.equal(plan.pruned, true); + assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); + assert.equal(plan.unknownVideos, 0); + + const result = await searchTranscripts(source, { + query: "lawsuit", + contentTypes: ["video"], + filters, + }); + assert.deepEqual(source.reads, ["chan:1"]); + assert.equal(result.total, 1); + assert.equal(result.hits[0].videoId, "collab"); +}); + +test("a video the index has no tags for still gets its page read", async () => { + // The pre-spec-4 case in miniature: the summaries set is older than the + // transcripts and has never heard of `ghost`. "I don't know about this + // video" must mean READ THE PAGE and let the record predicate decide — the + // planner's one invariant — not "it carries no tags, skip it". `plain` IS + // known and genuinely untagged, so it is correctly pruned. + const p0 = [video("chan", "plain")]; + const p1 = [video("chan", "ghost", { curatedTags: ["eva-collab"] })]; + const source = new CountingSource( + { chan: [p0, p1] }, + new Map([indexed(p0[0])]), + ); + const channels = await source.listChannels(); + const filters: SearchFilters = { ...ALL_STATES, curatedTags: ["eva-collab"] }; + + const plan = await buildScanPlan(source, channels, filters); + assert.deepEqual(plan.perChannel.get("chan")?.pages, [1]); + assert.equal(plan.unknownVideos, 1); + + const result = await searchTranscripts(source, { + query: "lawsuit", + contentTypes: ["video"], + filters, + }); + assert.deepEqual(source.reads, ["chan:1"]); + assert.equal(result.total, 1, "the record the index missed still matched"); +}); + +test("an empty curated-tag list is not a filter and does not plan", async () => { + const p0 = [video("chan", "a1")]; + const source = new CountingSource({ chan: [p0] }, new Map()); + const plan = await buildScanPlan(source, await source.listChannels(), { + ...ALL_STATES, + curatedTags: [], + }); + assert.equal(plan.pruned, false); + assert.equal(plan.pagesPlanned, 1); +}); + // ─── 3. duplicate collapsing ─── test("a recording mirrored across channels is counted once and its mirror named", async () => { diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts @@ -107,10 +107,16 @@ const CHAN_A: TranscriptDetail[] = [ vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), { description: "a pour over brewing guide", tags: ["espresso", "beans"], + // CURATED tags (lib/curatedTags.ts) — the operator's vocabulary, which is + // a different field from the yt-dlp keywords one line above and is + // deliberately set on the same record so a test that confuses the two + // fails loudly. + curatedTags: ["eva-collab"], }), vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])), vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), { isLivestream: true, + curatedTags: ["eva-in-chat"], }), vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), { ageRestricted: true, @@ -623,6 +629,34 @@ test("spec filter fa: keep only age-restricted → a4", async () => { assert.deepEqual(ids, ["a4"]); }); +test("spec filter tg: a curated tag keeps only the videos carrying it", async () => { + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, curatedTags: ["eva-collab"] }, + }); + assert.deepEqual(ids, ["a1"]); +}); + +test("spec filter tg: several tags are ORed, never ANDed", async () => { + // a1 carries eva-collab, a3 carries eva-in-chat and neither carries both: + // an AND here would return nothing, which is the bug this pins. + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, curatedTags: ["eva-collab", "eva-in-chat"] }, + }); + assert.deepEqual(ids, ["a1", "a3"]); +}); + +test("spec filter tg: an untagged record carries none and passes none", async () => { + // Every record on a site built before corpus spec 4 looks like this, so the + // pre-spec-4 path is "an honest empty result", not "everything matches". + const src = new StubSource(); + const ids = await specIds(src, leafTree("coffee", "transcripts"), { + filters: { ...KEEP_ALL, curatedTags: ["never-assigned"] }, + }); + assert.deepEqual(ids, []); +}); + test("spec filter fav: available-only drops every missing state", async () => { const src = new StubSource(); const ids = await specIds(src, leafTree("coffee", "transcripts"), { diff --git a/mcp/src/server.ts b/mcp/src/server.ts @@ -11,6 +11,8 @@ import type { VideoStat } from "yt-dlp-transcript-common/lib/stats"; import { momentUrl, momentBaseUrl } from "yt-dlp-transcript-common/lib/momentUrl"; import type { Platform } from "yt-dlp-transcript-common/lib/platform"; import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases"; +import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags"; +import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags"; import { VIDEO_STATES, isVideoState, @@ -215,6 +217,20 @@ const SEARCH_FILTER_ARGS = { "state. ('maybe_missing' = fell out of the channel listing but was never " + "individually confirmed.)", }, + tags: { + type: "array", + items: { type: "string" }, + description: + "Keep only videos carrying one of these CURATED TAGS — the operator's " + + "cross-channel vocabulary (e.g. tags:['eva-collab'] is 'every stream " + + "where she is on mic', on any channel, including channels that are not " + + "hers). ORed: naming two tags keeps a video with EITHER. Call " + + "`list_tags` for the ids this source publishes and how many videos each " + + "one has — do not guess an id. NOT the yt-dlp keywords searched by " + + "scopes:['tags']; those are metadata from the uploader, these are " + + "curation. A source that publishes no tags says so in the footer rather " + + "than returning a quiet zero.", + }, date_from: { type: "string", description: "Keep only videos uploaded on or after this date (YYYYMMDD).", @@ -306,6 +322,27 @@ export const TOOLS: Tool[] = [ }, }, { + name: "list_tags", + description: + "List the CURATED TAGS this archive publishes — the operator's " + + "cross-channel vocabulary for marking individual videos (e.g. " + + "'eva-collab' = she is on mic), which is what `tags` on " + + "search_transcripts / enumerate_matches filters by. Returns each tag's " + + "id, label, group, how many videos carry it ON THIS SOURCE, and the " + + "per-channel breakdown — so a tag can be scoped, counted and cited " + + "without guessing an id. These are NOT the yt-dlp keywords searched by " + + "scopes:['tags']: those come from the uploader's metadata, these are " + + "curation applied after the fact and cut across channels. A source that " + + "publishes none (nothing curated yet, or a site built before corpus " + + "spec 4) says so plainly rather than returning an empty list that reads " + + "like an answer.", + inputSchema: { + type: "object", + properties: { ...SOURCE_ARG }, + additionalProperties: false, + }, + }, + { name: "search_transcripts", description: "Search the archive for a term or phrase. The corpus holds video " + @@ -934,6 +971,8 @@ export function createServer( switch (name) { case "list_channels": return handleListChannels(source, args); + case "list_tags": + return handleListTags(source); case "search_transcripts": return handleSearch(source, args); case "enumerate_matches": @@ -1087,6 +1126,94 @@ async function handleListChannels( ); } +// The curated-tag vocabulary a source publishes, with its own counts. +// +// The empty case is answered in WORDS, not with an empty list. An agent that +// gets `[]` back reads it as "no tags here" and moves on; what it actually +// needs to know is that a tag filter against this source will match nothing and +// why — the site publishes none, or it was built before the corpus spec that +// carries them. That is the same failure the footer warning below exists to +// stop, said once up front. +async function handleListTags(source: ShardSource): Promise<ToolResult> { + const tags = await loadSourceTags(source); + if (tags.length === 0) { + return text( + `${source.label} publishes no curated tags.\n\n` + + `Either nothing has been tagged for this site yet, or it was built ` + + `before curated tags existed (corpus spec 4) — /tags.json is absent. ` + + `A \`tags\` filter against this source would match nothing, so do not ` + + `use one here; scope with channel/group/date instead.`, + ); + } + + const groups = groupPublishedTags(tags); + const sections = groups.map((g) => { + const head = `### ${g.label || "(ungrouped)"}`; + const lines = g.tags.map((t) => { + const per = Object.entries(t.channels) + .sort((a, b) => b[1] - a[1] || a[0].localeCompare(b[0])) + .map(([slug, n]) => `${slug} ${n}`) + .join(", "); + return ( + `- ${t.id} · ${t.label} · ${t.count} video(s)` + + (per ? ` — ${per}` : "") + ); + }); + return `${head}\n${lines.join("\n")}`; + }); + + return text( + `${tags.length} curated tag(s) in ${source.label}:\n\n` + + `${sections.join("\n\n")}\n\n` + + `(filter with tags:["${tags[0].id}"] on search_transcripts or ` + + `enumerate_matches; several tags are ORed. A video may carry more than ` + + `one tag, so these counts do not sum to a video count.)`, + ); +} + +// Every tag read goes through here so "this source cannot report tags at all" +// (an in-memory stub, which has no such concept) and "this source publishes +// none" land in the same place, as the same empty list. +async function loadSourceTags(source: ShardSource): Promise<PublishedTag[]> { + if (typeof source.loadTags !== "function") return []; + try { + return await source.loadTags(); + } catch { + return []; + } +} + +// A tag filter against a source that publishes no /tags.json matches nothing. +// That is CORRECT — a record with no curated tags carries none — but a bare +// zero is indistinguishable from "searched, found nothing", so it is named. +// +// The scan itself is unaffected and stays honest: the page planner only ever +// PRUNES, and a video its index has never heard of gets its page read anyway +// (search.ts's load-bearing invariant), so this is a reporting fix, never a +// coverage one. +async function describeTagCoverage( + source: ShardSource, + parsed: ParsedSearchArgs, +): Promise<string> { + const asked = parsed.filters?.curatedTags ?? []; + if (asked.length === 0) return ""; + const published = await loadSourceTags(source); + if (published.length === 0) { + return ( + `⚠ this source publishes no /tags.json (nothing curated, or built ` + + `before corpus spec 4) — a tag filter matches NOTHING here, so this ` + + `result is not evidence of absence` + ); + } + const known = new Set(published.map((t) => t.id)); + const unknown = asked.filter((t) => !known.has(t)); + if (unknown.length === 0) return ""; + return ( + `⚠ tag(s) this source does not publish: ${unknown.join(", ")} — they ` + + `match nothing here; call list_tags for the ids it has` + ); +} + async function handleSearch( source: ShardSource, args: Record<string, unknown>, @@ -1119,6 +1246,7 @@ async function handleSearch( const scopeNote = describeScope(result.selection); const postsNote = describePostsPass(result); const filterNote = describeSearchFilters(parsed); + const tagNote = await describeTagCoverage(source, parsed); const prunedNote = describeCoverage(result); const dupNote = describeDuplicates(result); const rangeStart = result.total === 0 ? 0 : result.offset + 1; @@ -1136,6 +1264,7 @@ async function handleSearch( (filterNote ? `; ${filterNote}` : "") + (postsNote ? `; ${postsNote}` : "") + (aliasNote ? `; ${aliasNote}` : "") + + (tagNote ? `; ${tagNote}` : "") + parsed.warnings.map((w) => `; ⚠ ${w}`).join("") + ")"; @@ -1321,6 +1450,7 @@ async function handleEnumerateMatches( const aliasNote = describeFiredAliases(result.firedAliases); const postsNote = describePostsPass(result); const filterNote = describeSearchFilters(parsed); + const tagNote = await describeTagCoverage(source, parsed); const prunedNote = describeCoverage(result); const dupNote = describeDuplicates(result); @@ -1335,6 +1465,7 @@ async function handleEnumerateMatches( (filterNote ? `; ${filterNote}` : "") + (postsNote ? `; ${postsNote}` : "") + (aliasNote ? `; ${aliasNote}` : "") + + (tagNote ? `; ${tagNote}` : "") + parsed.warnings.map((w) => `; ⚠ ${w}`).join("") + ")"; @@ -1426,12 +1557,35 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs { const from = /^\d{8}$/.test(dateFrom) ? dateFrom : undefined; const to = /^\d{8}$/.test(dateTo) ? dateTo : undefined; + // Curated tags. A malformed id is REPORTED, never silently dropped: dropping + // the only tag in the list would widen the search back to the whole corpus + // and read as a legitimate result, which is the same failure a typo'd state + // would cause. + const rawTags = strArray(args.tags); + let tags: string[] | undefined; + if (rawTags) { + const normalized = rawTags.map((t) => t.trim().toLowerCase()); + tags = normalized.filter((t) => isTagId(t)); + const bad = normalized.filter((t) => !isTagId(t)); + if (bad.length > 0) { + warnings.push( + `not a tag id, ignored: ${bad.join(", ")} (ids are lowercase, ` + + `[a-z0-9._-]; call list_tags)`, + ); + } + if (tags.length === 0) { + tags = undefined; + warnings.push("no valid tags given — the tag filter was NOT applied"); + } + } + const anyFilter = states !== undefined || mediaType !== undefined || age !== undefined || from !== undefined || - to !== undefined; + to !== undefined || + tags !== undefined; const filters: SearchFilters | null = anyFilter ? { @@ -1442,6 +1596,7 @@ function parseSearchArgs(args: Record<string, unknown>): ParsedSearchArgs { states: new Set<VideoState>(states ?? VIDEO_STATES), ...(from ? { dateFrom: from } : {}), ...(to ? { dateTo: to } : {}), + ...(tags ? { curatedTags: tags } : {}), } : null; @@ -1490,6 +1645,9 @@ function describeSearchFilters(p: ParsedSearchArgs): string { if (f.dateFrom || f.dateTo) { parts.push(`uploaded ${f.dateFrom ?? "…"}–${f.dateTo ?? "…"}`); } + if (f.curatedTags && f.curatedTags.length > 0) { + parts.push(`tags: ${f.curatedTags.join(" OR ")}`); + } } if (p.exclude && p.exclude.length > 0) { parts.push(`excluding ${p.exclude.map((e) => `"${e}"`).join(", ")}`); @@ -2190,6 +2348,17 @@ async function handleResolveSource( ` reachable: yes — ${channels.length} channel(s)` + (groups.length > 0 ? `, ${groups.length} group(s)` : ""), ); + // Named here so a caller learns whether a `tags` filter is even available + // BEFORE it writes one and reads the empty result as an answer. The two + // states are different and both are said: some tags, or none at all. + const tags = await loadSourceTags(resolved.source); + lines.push( + tags.length > 0 + ? ` curated tags: ${tags.length} — ` + + tags.map((t) => `${t.id} (${t.count})`).join(", ") + + ` · list_tags for the breakdown` + : ` curated tags: none published (a \`tags\` filter matches nothing here)`, + ); } catch (e) { return errorText( `${lines.join("\n")}\n reachable: NO — ${(e as Error).message}`,