commit b56621327931d93e958b90df8a457195d3f97777
parent f2b4f358f29e31b1f59788588548222d43051ffd
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Tue, 6 Oct 2026 09:55:01 -0400
mcp: search reads every English track; get_transcript takes an optional track
search_transcripts and the query-tree tools match a record's alternate tracks
(lib/captionTracks.ts) and tag a snippet from one with its track ("in
uploaded captions"). get_transcript names a record's tracks in its header and
reads another with `track`; get_transcripts windows a match only an alternate
holds, under its name; get_video_metadata lists the other tracks without their
cues. The sweep plan says what such a hit is before anyone quotes it.
Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
Diffstat:
4 files changed, 200 insertions(+), 14 deletions(-)
diff --git a/mcp/src/instructions.ts b/mcp/src/instructions.ts
@@ -232,6 +232,16 @@ export function buildSweepInstructions(
);
steps.push(
+ `**A hit can come from another caption track.** Where a video has more ` +
+ `than one English track whose words differ, search reads them all; a ` +
+ `snippet tagged \`in uploaded captions\` (or another track) matched ` +
+ `words the primary transcript — the original audio's captions — does ` +
+ `not have there. Uploaded captions are not always what was said: read ` +
+ `the primary around that moment (\`get_transcript\`; \`track\` reads ` +
+ `the other one) before quoting, and say which track the words are from.`,
+ );
+
+ steps.push(
`**State the plan.** Report N (the enumerated total) and ` +
`\`ceil(N / ${req.batchSize})\` batches before you start. The report may ` +
`only ever claim the coverage this number justifies: N videos ` +
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -237,7 +237,7 @@ class StubSource implements ShardSource {
];
}
- private pages(ch: ChannelRef): TranscriptDetail[][] {
+ protected pages(ch: ChannelRef): TranscriptDetail[][] {
// One record per page so paging exercises multiple shard pages.
const recs = ch.slug === "chan-a" ? CHAN_A : CHAN_B;
return recs.map((r) => [r]);
@@ -1685,3 +1685,80 @@ test("server: the unadvertised channel/group singulars are still parsed", async
assert.match(out, /scope: 1 channel/);
await client.close();
});
+
+// ─── Alternate tracks (lib/captionTracks.ts) ───
+
+// chan-b plus b2: an en-orig primary and an uploaded `en` that says a word the
+// primary never does.
+const ALT_REC = vid(
+ "b2",
+ "Two tracks",
+ "chan-b",
+ cues([5, "the harbor bridge opened"]),
+ {
+ track: "en-orig",
+ altTracks: [{ track: "en", cues: cues([6, "the harbor bridge opened"], [90, "a zeppelin flew over"]) }],
+ },
+);
+class AltTrackSource extends StubSource {
+ protected override pages(ch: ChannelRef): TranscriptDetail[][] {
+ const base = super.pages(ch);
+ return ch.slug === "chan-b" ? [...base, [ALT_REC]] : base;
+ }
+}
+
+test("searchTranscripts: a word only an alternate track holds is found there, the track named", async () => {
+ const src = new AltTrackSource();
+ const res = await searchTranscripts(src, { query: "zeppelin" });
+ assert.equal(res.hits.length, 1);
+ assert.equal(res.hits[0].videoId, "b2");
+ assert.deepEqual(
+ res.hits[0].snippets.map((s) => [s.seconds, s.track]),
+ [[90, "en"]],
+ );
+ // Said by both tracks at the same moment: once, from the primary.
+ const both = await searchTranscripts(src, { query: "harbor bridge" });
+ assert.deepEqual(both.hits[0].snippets.map((s) => s.track), [undefined]);
+});
+
+test("server: a hit from an alternate track says which one", async () => {
+ const client = await connectClient(new AltTrackSource());
+ const out = firstText(
+ await client.callTool({ name: "search_transcripts", arguments: { query: "zeppelin" } }),
+ );
+ assert.match(out, /\[in uploaded captions \[1:30\]\(https:\/\/example.test\/b2\?t=90s\)\] a zeppelin flew over/);
+ await client.close();
+});
+
+test("server: get_transcript reads the primary by default and another track by `track`", async () => {
+ const client = await connectClient(new AltTrackSource());
+ const primary = firstText(
+ await client.callTool({ name: "get_transcript", arguments: { video_id: "b2" } }),
+ );
+ assert.doesNotMatch(primary, /zeppelin/);
+ assert.match(primary, /track: en-orig \(original audio captions\) — the primary/);
+ assert.match(primary, /other tracks: en \(uploaded captions\) — pass track to read one/);
+
+ const en = firstText(
+ await client.callTool({ name: "get_transcript", arguments: { video_id: "b2", track: "en" } }),
+ );
+ assert.match(en, /a zeppelin flew over/);
+ assert.match(en, /track: en \(uploaded captions\)/);
+
+ const bad = await client.callTool({
+ name: "get_transcript",
+ arguments: { video_id: "b2", track: "en-GB" },
+ });
+ assert.match(firstText(bad), /has no track "en-GB"; its tracks are: en-orig \(original audio captions\), en \(uploaded captions\)/);
+
+ // get_transcripts with a query windows a match only the alternate holds.
+ const batch = firstText(
+ await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b2"], query: "zeppelin" },
+ }),
+ );
+ assert.match(batch, /1 matching line\(s\) only in uploaded captions \(track en\), windowed/);
+ assert.match(batch, /a zeppelin flew over/);
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -10,6 +10,7 @@
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
import { postConversation, type Post } from "yt-dlp-transcript-common/lib/posts";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import { hitsAcrossTracks } from "yt-dlp-transcript-common/lib/captionTracks";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import {
type GroupNode,
@@ -65,6 +66,10 @@ export type Snippet = {
// reader has to be able to tell "the word appears in the description" from
// "the word was said at 0:00" — they license completely different citations.
scope?: LayerScope;
+ // A transcript hit from one of the record's ALTERNATE English tracks
+ // (lib/captionTracks.ts) — words its primary does not have there. Absent for
+ // a hit in the primary.
+ track?: string;
};
// A scope selector for a search/sweep: any mix of channel handles (slug / key /
@@ -614,13 +619,17 @@ export async function searchTranscripts(
let otherHit = false;
if (wantCues) {
- for (const cue of rec.cues ?? []) {
- if (!match(cue.text)) continue;
+ // Every English track of the record (lib/captionTracks.ts): a
+ // match only an alternate holds names that alternate.
+ for (const cue of hitsAcrossTracks(rec, (cues) =>
+ cues.filter((c) => match(c.text)),
+ )) {
matches++;
push({
clock: clock(cue.start),
seconds: cue.start,
text: truncate(cue.text, policy.snippetChars),
+ ...(cue.track ? { track: cue.track } : {}),
});
}
}
@@ -1124,6 +1133,7 @@ export async function runSearchSpec(
description: rec.description ?? "",
tags: (rec.tags ?? []).join(", "),
cues: rec.cues ?? [],
+ ...(rec.altTracks ? { altTracks: rec.altTracks } : {}),
chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [],
snippetsPerVideo,
includeSnippets,
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -15,6 +15,14 @@ import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import { isTagId, type PublishedTag } from "yt-dlp-transcript-common/lib/curatedTags";
import { groupPublishedTags } from "yt-dlp-transcript-common/lib/publishedTags";
import {
+ cuesOfTrack,
+ hitsAcrossTracks,
+ inTrackLabel,
+ recordTracks,
+ trackLabel,
+} from "yt-dlp-transcript-common/lib/captionTracks";
+import { windowedTranscript } from "yt-dlp-transcript-common/lib/search/window";
+import {
VIDEO_STATES,
isVideoState,
type VideoState,
@@ -44,6 +52,7 @@ import {
findVideo,
buildMatcher,
getWindowedTranscript,
+ MCP_POLICY,
runSearchSpec,
type SearchFilters,
type SearchResult,
@@ -582,6 +591,14 @@ export const TOOLS: Tool[] = [
type: "boolean",
description: "Prefix each caption line with a timestamp (default true).",
},
+ track: {
+ type: "string",
+ description:
+ "Optional: read one of the video's other English tracks instead of its " +
+ "primary transcript (e.g. \"en\" for the uploaded captions beside the " +
+ "original-audio \"en-orig\"). The header lists the tracks a video has; " +
+ "only tracks whose words differ from the primary are kept. Omit for the primary.",
+ },
},
required: ["video_id"],
additionalProperties: false,
@@ -1635,7 +1652,11 @@ async function handleSearch(
const stamp = base
? baseStamp(s.clock, s.seconds)
: stampMarkup(source, h, s.clock, s.seconds);
- const tag = s.scope && s.scope !== "transcripts" ? `${s.scope} ` : "";
+ const tag = s.scope && s.scope !== "transcripts"
+ ? `${s.scope} `
+ : s.track
+ ? `${inTrackLabel(s.track)} `
+ : "";
return ` - [${tag}${stamp}] ${s.text}`;
})
.join("\n");
@@ -2207,12 +2228,37 @@ async function handleGetTranscripts(
counts.length > 1
? counts.map((c) => `"${c.query}": ${c.n}`).join(", ")
: `${matchCount} matching line(s)`;
+ // The record's alternate tracks (lib/captionTracks.ts): a match only
+ // an alternate holds is windowed from that track, under its name.
+ const altBlocks: string[] = [];
+ let altMatches = 0;
+ const across = hitsAcrossTracks(record, (list) =>
+ list.filter((c) => matcher.match(c.text)),
+ );
+ for (const alt of record.altTracks ?? []) {
+ const own = new Set(across.filter((h) => h.track === alt.track).map((h) => h.text));
+ if (own.size === 0) continue;
+ const w = windowedTranscript(alt.cues, (t) => own.has(t), {
+ before,
+ after,
+ timestamps,
+ stamp,
+ maxLines: maxLines ?? MCP_POLICY.windowLineCap,
+ });
+ altMatches += w.matchCount;
+ altBlocks.push(
+ `_(${w.matchCount} matching line(s) only ${inTrackLabel(alt.track)} (track ${alt.track}), windowed)_\n${w.lines.join("\n")}`,
+ );
+ }
const body =
- matchCount === 0
+ matchCount === 0 && altMatches === 0
? `_(no lines matched ${
counts.length > 1 ? "any query" : "the query"
} in this transcript${counts.length > 1 ? ` — ${countNote}` : ""})_`
- : `_(${countNote}, windowed)_\n${lines.join("\n")}`;
+ : [
+ ...(matchCount > 0 ? [`_(${countNote}, windowed)_\n${lines.join("\n")}`] : []),
+ ...altBlocks,
+ ].join("\n\n");
blocks.push(`${head}\n\n${body}`);
} else {
const md = transcriptToMarkdown(
@@ -2377,11 +2423,38 @@ async function handleGetTranscript(
...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
...(record.platform ? { platform: record.platform } : {}),
};
- const md = transcriptToMarkdown(record, {
- timestamps: args.timestamps !== false,
- includeTags: true,
- linkForCue: (seconds) => momentLinkFor(source, link, seconds),
- });
+ // The record's tracks (lib/captionTracks.ts): the primary unless `track`
+ // names an alternate it holds.
+ const tracks = recordTracks(record);
+ const asked = typeof args.track === "string" ? args.track.trim() : "";
+ const cues = cuesOfTrack(record, asked);
+ if (asked && cues === undefined) {
+ return errorText(
+ tracks.length > 1
+ ? `video ${videoId} has no track "${asked}"; its tracks are: ${tracks.map((t) => `${t} (${trackLabel(t)})`).join(", ")}`
+ : `video ${videoId} has no track "${asked}": it has only its primary transcript`,
+ );
+ }
+ const shown = asked || record.track;
+ const extraMeta =
+ tracks.length > 1 && shown
+ ? [
+ `track: ${shown} (${trackLabel(shown)})${shown === record.track ? " — the primary" : ""}`,
+ `other tracks: ${tracks
+ .filter((t) => t !== shown)
+ .map((t) => `${t} (${trackLabel(t)}${t === record.track ? ", the primary" : ""})`)
+ .join(", ")} — pass track to read one`,
+ ]
+ : undefined;
+ const md = transcriptToMarkdown(
+ { ...record, cues },
+ {
+ timestamps: args.timestamps !== false,
+ includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
+ ...(extraMeta ? { extraMeta } : {}),
+ },
+ );
return text(md);
}
@@ -2405,12 +2478,23 @@ async function handleGetMetadata(
typeof args.channel === "string" ? args.channel : undefined,
);
if (!found) return errorText(`video not found: ${videoId}`);
- const { cues, ...meta } = found.record;
+ const { cues, altTracks, ...meta } = found.record;
const slug = found.record.slug;
+ // An alternate's cues are not metadata; its id and label are.
+ const otherTracks = (altTracks ?? []).map((t) => ({
+ track: t.track,
+ label: trackLabel(t.track),
+ cueCount: t.cues.length,
+ }));
const lines: string[] = [
JSON.stringify(
- { ...meta, channelName: found.ch.name, cueCount: cues?.length ?? 0 },
+ {
+ ...meta,
+ channelName: found.ch.name,
+ cueCount: cues?.length ?? 0,
+ ...(otherTracks.length > 0 ? { otherTracks } : {}),
+ },
null,
2,
),
@@ -2912,7 +2996,12 @@ function renderScopedSnippet(
s: ScopedSnippet,
): string {
if (s.seconds > 0) {
- const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `;
+ const tag =
+ s.scope === "transcripts"
+ ? s.track
+ ? `${inTrackLabel(s.track)} `
+ : ""
+ : `${s.track ?? s.scope} `;
return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`;
}
return ` - [${s.scope}] ${s.text}`;