commit 09ad64e928b303f70bf0ead35cb5a70a232a2439
parent 0a97fc07e198af8cedebed4a98586785e716052c
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 22 Jul 2026 00:02:52 -0400
Merge worktree-mcp-token-efficiency: MCP compact base-links + dumb-extractor sweep batches
Diffstat:
8 files changed, 578 insertions(+), 58 deletions(-)
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -108,3 +108,73 @@ export function momentUrl(input: MomentUrlInput): string | null {
if (viewer) return viewer;
return platformMomentUrl(input.webpageUrl, input.platform, input.seconds);
}
+
+// ─── Base (appendable) forms ───
+//
+// A "moment base" is a URL that ends in `t=`, so appending integer seconds
+// yields a valid moment link (`<base>156` ≡ the momentUrl for 156s). Used by
+// the MCP's compact `link_style:"base"` output: one base per video instead of
+// a full URL per line, expanded back to full links in the final report.
+
+// The viewer deep-link base: same normalization as viewerMomentUrl, with `t`
+// set last and empty so the caller can append seconds. Null when the origin or
+// slug is missing/unparseable.
+export function viewerMomentBaseUrl(
+ siteOrigin: string | null | undefined,
+ slug: string | null | undefined,
+): string | null {
+ if (!siteOrigin || !slug) return null;
+ let u: URL;
+ try {
+ u = new URL(siteOrigin);
+ } catch {
+ return null;
+ }
+ u.pathname = "/";
+ u.search = "";
+ u.hash = "";
+ u.searchParams.set("v", slug);
+ u.searchParams.set("t", "");
+ return u.toString();
+}
+
+// The platform base. Unlike platformMomentUrl (which falls back to the bare
+// webpage URL), a base MUST be appendable — so only platforms whose time param
+// takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs`
+// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown
+// have no dependable start param at all. Null in every non-appendable case.
+export function platformMomentBaseUrl(
+ webpageUrl: string | null | undefined,
+ platform: Platform | null | undefined,
+): string | null {
+ if (!webpageUrl) return null;
+ const plat = platform ?? detectPlatform(webpageUrl);
+ let u: URL;
+ try {
+ u = new URL(webpageUrl);
+ } catch {
+ return null;
+ }
+ switch (plat) {
+ case "youtube": // accepts raw seconds in `t` (the `s` suffix is optional)
+ case "odysee": {
+ // delete-then-set so a pre-existing `t` is re-appended LAST — the base
+ // must end in `t=` for the append rule to hold.
+ u.searchParams.delete("t");
+ u.searchParams.set("t", "");
+ return u.toString();
+ }
+ default:
+ return null;
+ }
+}
+
+// The preferred moment base: viewer first, platform fallback — mirroring
+// momentUrl's preference order. Null when neither can be built.
+export function momentBaseUrl(
+ input: Omit<MomentUrlInput, "seconds">,
+): string | null {
+ const viewer = viewerMomentBaseUrl(input.siteOrigin, input.slug);
+ if (viewer) return viewer;
+ return platformMomentBaseUrl(input.webpageUrl, input.platform);
+}
diff --git a/common/lib/transcriptToMarkdown.test.ts b/common/lib/transcriptToMarkdown.test.ts
@@ -48,3 +48,33 @@ test("maxCues truncates and notes it", () => {
assert.doesNotMatch(md, /General Kenobi/);
assert.match(md, /truncated: showing 1 of 2 cues/);
});
+
+test("stampForCue renders the compact form and floors fractional seconds", () => {
+ const md = transcriptToMarkdown(base, {
+ stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ });
+ assert.match(md, /^\[0:00\|0\] Hello there$/m);
+ // 3661.25s floors to 3661 — the same integer momentUrl would use, so the
+ // compact stamp and the inline link cite the identical second.
+ assert.match(md, /^\[1:01:01\|3661\] General Kenobi$/m);
+});
+
+test("stampForCue takes precedence over linkForCue (no inline links)", () => {
+ const md = transcriptToMarkdown(base, {
+ stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ linkForCue: (seconds) => `https://example.test/abc123?t=${seconds}s`,
+ });
+ assert.ok(!md.includes("](https://example.test/"), "no per-line links");
+ assert.match(md, /^\[0:00\|0\] Hello there$/m);
+});
+
+test("extraMeta lines render as `- <line>` in the metadata header", () => {
+ const md = transcriptToMarkdown(base, {
+ extraMeta: ["moment_base: https://example.test/abc123?t="],
+ });
+ assert.match(md, /^- moment_base: https:\/\/example\.test\/abc123\?t=$/m);
+ assert.ok(
+ md.indexOf("moment_base:") < md.indexOf("## Description"),
+ "extra meta sits in the header, before the body",
+ );
+});
diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts
@@ -36,6 +36,14 @@ export type TranscriptMarkdownOptions = {
// (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when
// `timestamps` is false or when it returns null. See `momentUrl`.
linkForCue?: (seconds: number) => string | null;
+ // Optional formatter for the bracketed label's CONTENTS (e.g. "2:36|156" →
+ // rendered as `[2:36|156] text`). Receives the pre-formatted clock and the
+ // cue's raw start seconds. Takes precedence over `linkForCue`. Ignored when
+ // `timestamps` is false.
+ stampForCue?: (clock: string, seconds: number) => string;
+ // Extra metadata lines rendered as `- <line>` after the tags line (e.g.
+ // "moment_base: <url>").
+ extraMeta?: string[];
};
// [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so
@@ -59,6 +67,8 @@ export function transcriptToMarkdown(
includeTags = false,
maxCues,
linkForCue,
+ stampForCue,
+ extraMeta,
} = options;
const lines: string[] = [];
@@ -76,6 +86,7 @@ export function transcriptToMarkdown(
if (includeTags && input.tags && input.tags.length > 0) {
meta.push(`- Tags: ${input.tags.join(", ")}`);
}
+ for (const line of extraMeta ?? []) meta.push(`- ${line}`);
lines.push(...meta);
if (includeDescription && input.description && input.description.trim()) {
@@ -107,6 +118,10 @@ export function transcriptToMarkdown(
lines.push(text);
continue;
}
+ if (stampForCue) {
+ lines.push(`[${stampForCue(stamp(cue.start), cue.start)}] ${text}`);
+ continue;
+ }
const url = linkForCue ? linkForCue(cue.start) : null;
const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start);
lines.push(`[${label}] ${text}`);
diff --git a/mcp/README.md b/mcp/README.md
@@ -13,8 +13,8 @@ reads the site's already-published static JSON shards (`corpus.json` +
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
-| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment**. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
-| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`), or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
+| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
+| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`, or compact `link_style:"base"` stamps; `max_lines` caps the excerpt), or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). |
| `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. |
| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity). **Previews** a plan by default; **applies** it (switch source + search, linked results) on `apply:true`. Adjust in natural language via `overrides`. |
@@ -29,6 +29,19 @@ otherwise the video's platform watch page with a per-platform time param
(YouTube `&t=<sec>s`, Odysee/Twitch equivalents). A `--local` source with no
origin falls back to platform links.
+**Compact base links (`link_style:"base"`).** Inline links are ~70–90 chars *per
+line* — bulk an agent pipeline shouldn't pay for. `search_transcripts` and
+`get_transcripts` accept `link_style:"base"`: each video gets **one**
+`- moment_base:` header line (a URL ending in `t=`) and every stamp becomes the
+compact `[mm:ss|<seconds>]`. The expansion rule: **full moment link =
+`<moment_base><seconds>`** — append the integer after the `|`, e.g.
+`[title @ 2:36](<moment_base>156)`. The seconds are floored exactly like the
+inline links', so both styles cite the identical second. A video whose base
+can't be built (no viewer origin and a platform whose time param doesn't take
+raw seconds — Twitch — or doesn't exist — Rumble/Kick) omits the line: cite its
+`- source:` URL plain instead. `open_link` results and `get_transcript` stay
+inline-linked (future work).
+
### `search_transcripts`
Beyond `query`, `regex`, and `limit`:
@@ -50,8 +63,11 @@ Beyond `query`, `regex`, and `limit`:
full `total` and `has_more`, so you can enumerate a query's *entire* match set:
page with `offset += limit` until `has_more` is `no`.
- **`include_snippets`** (default true) — set `false` for a cheap worklist
- (id / title / channel / date / match count, no cue text). Ideal for the
- planning pass of a sweep.
+ (id / title / channel / date / match count, no cue text — the per-video
+ `- source:` line is dropped too). Ideal for the planning pass of a sweep.
+- **`link_style`** (default `"inline"`) — `"base"` switches to the compact
+ agent-pipeline form: a `- moment_base:` line per hit and `[mm:ss|<seconds>]`
+ snippet stamps (see *Compact base links* above).
- **`use_aliases`** (default true) — expand the query through the site's curated
search aliases. A plain query that matches a curated trigger also searches the
alias's regex, so mis-transcribed spellings are caught (e.g. `k cups` also
@@ -74,6 +90,12 @@ video's full transcript comes back as markdown. Missing ids are reported inline.
lookup when a batch spans several channels (the ids are already scoped by the
search that produced them).
+- **`link_style`** (default `"inline"`) — `"base"` emits one `- moment_base:`
+ line per video and compact `[mm:ss|<seconds>]` stamps instead of a full
+ Markdown link per line (see *Compact base links* above).
+- **`max_lines`** (default 200) — cap on merged excerpt lines per video when a
+ query is given (the earliest lines are kept). Ignored without a query.
+
### `open_link` — paste a viewer share link, at full fidelity
The archilyzer viewer's **Share** button produces a URL that encodes the whole
@@ -119,7 +141,9 @@ Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query`
(required *unless* a `link` is given), `link?` (an archilyzer viewer share URL to
seed the sweep from), `channel?`, `channels?` (comma-separated slugs/names),
`group?` (a channel group id or name), `directive?` (default *"key claims &
-contradictions"*), `batch_size?` (default 8), `report_path?` (default
+contradictions"*), `batch_size?` (default 8), `parse_model?` (default
+`haiku` — the model requested for the per-batch extractor subagents; they only
+quote verbatim, so the cheapest model wins), `report_path?` (default
`./sweep-report.md`).
```
@@ -151,19 +175,27 @@ worklist by paging with `include_snippets:false` until `has_more` is false →
a short summary. The MCP stays read-only; only the report file is written, in
Claude's working directory.
-**Subagent-compacted batches.** Each batch is handed to a **subagent** (Claude
-Code's Task tool) that calls `get_transcripts`, extracts findings for the
-directive, and returns *only a compact fragment of cited, linked findings* — the
-heavy transcript text lives and dies inside the subagent, so the orchestrator's
-context keeps just the distilled report. The orchestrator merges each fragment
-into the report and discards it; batches are independent, so several can run in
-parallel. If no subagent tool is available, the batch is processed inline and the
-raw text dropped after folding (the original behaviour).
+**Dumb extractors on the cheap model.** Each batch is handed to a **subagent**
+(Claude Code's Task tool) spawned as a *verbatim extractor* on the cheapest
+model — the prompt requests `parse_model` (default `haiku`) via the Task tool's
+model override, and gracefully spawns on the default when no override exists.
+The extractor calls `get_transcripts` with `link_style:"base"` and returns
+*only* the directive-relevant lines, **verbatim** — grouped per video under its
+`moment_base:` line, in their compact `[mm:ss|sec]` form, under a hard budget of
+**≤40 lines (~600 words) per batch**. No analysis, no summarizing: the heavy
+transcript bulk lives and dies inside the cheap subagent, and nothing irrelevant
+is ever reprocessed. The **orchestrator does all the synthesis** — claims,
+contradictions, cross-referencing — expanding each kept citation to a full link
+by appending the seconds to the video's `moment_base`, then discards the
+fragment. Batches are independent, so several can run in parallel. If no
+subagent tool is available, the batch is processed inline (still
+`link_style:"base"`) and the raw excerpt text dropped after folding.
**Linked citations.** Every source is cited as a clickable
-`[title @ mm:ss](<moment url>)` link — the moment URL comes straight from the
-`[mm:ss](url)` links in the `get_transcripts` output, so a click seeks to the
-exact second (see *Clickable moment links* above).
+`[title @ mm:ss](<moment url>)` link — built by appending the cited integer
+seconds to the video's `moment_base` from the tool output, so a click seeks to
+the exact second (see *Compact base links* above; a video with no base is cited
+by its `source:` URL). Note `open_link` results stay inline-linked.
## Data source (pick one)
diff --git a/mcp/src/momentUrl.test.ts b/mcp/src/momentUrl.test.ts
@@ -4,6 +4,9 @@ import {
momentUrl,
viewerMomentUrl,
platformMomentUrl,
+ momentBaseUrl,
+ viewerMomentBaseUrl,
+ platformMomentBaseUrl,
} from "yt-dlp-transcript-common/lib/momentUrl";
// ─── Archilyzer viewer link (preferred) ───
@@ -125,3 +128,92 @@ test("momentUrl: null when neither a viewer nor a platform link can be built", (
const url = momentUrl({ siteOrigin: null, slug: "ch/vid", seconds: 12 });
assert.equal(url, null);
});
+
+// ─── Base (appendable) forms — the link_style:"base" building blocks ───
+
+test("viewer base: ends in t= with the slug encoded; appending seconds parses", () => {
+ const base = viewerMomentBaseUrl(
+ "https://rekietalyzer.pages.dev",
+ "rekietalaw/uamorPe6hSc",
+ );
+ assert.ok(base);
+ assert.ok(base!.endsWith("&t="), "the base ends in t= so seconds append cleanly");
+ assert.ok(
+ base!.includes("v=rekietalaw%2FuamorPe6hSc"),
+ "the slug's / is %2F-encoded by URLSearchParams",
+ );
+ // Appending integer seconds yields exactly the viewer deep-link params.
+ const u = new URL(base! + "754");
+ assert.equal(u.searchParams.get("v"), "rekietalaw/uamorPe6hSc");
+ assert.equal(u.searchParams.get("t"), "754");
+});
+
+test("viewer base: null without an origin/slug or with an unparseable origin", () => {
+ assert.equal(viewerMomentBaseUrl(null, "ch/vid"), null);
+ assert.equal(viewerMomentBaseUrl("https://site.example", null), null);
+ assert.equal(viewerMomentBaseUrl("not a url", "ch/vid"), null);
+});
+
+test("platform base: YouTube ends in t= (existing query → &t=, none → ?t=)", () => {
+ const withQuery = platformMomentBaseUrl(
+ "https://www.youtube.com/watch?v=abc123",
+ "youtube",
+ );
+ assert.ok(withQuery!.endsWith("&t="));
+ const noQuery = platformMomentBaseUrl("https://youtu.be/abc123", "youtube");
+ assert.ok(noQuery!.endsWith("?t="));
+});
+
+test("platform base: a pre-existing t param is stripped and re-appended LAST", () => {
+ const base = platformMomentBaseUrl(
+ "https://www.youtube.com/watch?t=99s&v=abc",
+ "youtube",
+ );
+ assert.ok(base!.endsWith("&t="), "t= is the last param");
+ assert.ok(!base!.includes("99s"), "the old seek value is gone");
+ const u = new URL(base! + "42");
+ assert.equal(u.searchParams.get("v"), "abc");
+ assert.equal(u.searchParams.get("t"), "42");
+});
+
+test("platform base: Odysee ends in t= (raw-seconds param)", () => {
+ const base = platformMomentBaseUrl("https://odysee.com/@chan/video", "odysee");
+ assert.ok(base!.endsWith("?t="));
+});
+
+test("platform base: non-appendable cases → null (never an invalid seek)", () => {
+ // Twitch's t takes an XhYmZs token — appending raw seconds would mis-seek.
+ assert.equal(
+ platformMomentBaseUrl("https://www.twitch.tv/videos/12345", "twitch"),
+ null,
+ );
+ // Rumble has no dependable start param at all.
+ assert.equal(
+ platformMomentBaseUrl("https://rumble.com/v123-title.html", "rumble"),
+ null,
+ );
+ assert.equal(platformMomentBaseUrl(null, "youtube"), null);
+ assert.equal(platformMomentBaseUrl("not a url", "youtube"), null);
+});
+
+test("momentBaseUrl: viewer base wins, platform fallback, else null", () => {
+ const viewer = momentBaseUrl({
+ siteOrigin: "https://site.example",
+ slug: "ch/vid",
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.ok(viewer!.startsWith("https://site.example/?v="));
+ assert.ok(viewer!.endsWith("&t="));
+
+ const platform = momentBaseUrl({
+ siteOrigin: null,
+ slug: "ch/vid",
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.ok(platform!.startsWith("https://www.youtube.com/watch?v=vid"));
+ assert.ok(platform!.endsWith("&t="));
+
+ assert.equal(momentBaseUrl({ siteOrigin: null, slug: "ch/vid" }), null);
+});
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -404,6 +404,35 @@ test("getWindowedTranscript: no matches yields no lines", async () => {
assert.equal(lines.length, 0);
});
+test("getWindowedTranscript: a stamp formatter renders the bracket contents", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "coffee" });
+ const { lines } = getWindowedTranscript(rec, match, {
+ before: 30,
+ after: 30,
+ stamp: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ });
+ assert.ok(
+ lines.includes("[1:40|100] here is the coffee moment"),
+ lines.join("\n"),
+ );
+});
+
+test("getWindowedTranscript: maxLines keeps the earliest merged lines", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "coffee" });
+ const { lines, matchCount } = getWindowedTranscript(rec, match, {
+ before: 30,
+ after: 30,
+ maxLines: 2,
+ });
+ assert.equal(matchCount, 1, "matchCount is unaffected by the line cap");
+ assert.equal(lines.length, 2);
+ assert.ok(lines[0].includes("right before the moment"));
+ assert.ok(lines[1].includes("here is the coffee moment"));
+ assert.ok(!lines.join("\n").includes("right after the moment"));
+});
+
// ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ───
async function allChannels(src: StubSource): Promise<ChannelRef[]> {
@@ -680,6 +709,7 @@ test("server: the sweep prompt lists with its arguments and renders the query",
"directive",
"group",
"link",
+ "parse_model",
"query",
"report_path",
]);
@@ -751,3 +781,127 @@ test("server: the sweep prompt accepts a link= seed and drives open_link", async
assert.match(text, /share link/);
await client.close();
});
+
+// ─── link_style:"base" — compact stamps, moment_base, max_lines, worklist trim ───
+
+test("server: get_transcripts link_style base emits moment_base + compact stamps", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee", link_style: "base" },
+ });
+ const out = firstText(res);
+ // StubSource has no viewer origin → the platform (YouTube-param) base.
+ assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m);
+ assert.match(out, /\[1:40\|100\] here is the coffee moment/);
+ assert.ok(!out.includes("](http"), "no full inline links in base excerpts");
+ assert.match(out, /<moment_base><seconds>/, "the expansion note is present");
+ await client.close();
+});
+
+test("server: get_transcripts default output stays inline-linked (regression pin)", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee" },
+ });
+ const out = firstText(res);
+ assert.ok(out.includes("](https://example.test/b1?t=100s)"), out);
+ assert.ok(!out.includes("moment_base"), "no base artifacts in inline mode");
+ await client.close();
+});
+
+test("server: get_transcripts base full transcript uses compact stamps + header base", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], link_style: "base" },
+ });
+ const out = firstText(res);
+ assert.match(out, /\[0:00\|0\] intro chatter/);
+ assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m);
+ assert.ok(!out.includes("](http"), "no inline links in base full transcripts");
+ await client.close();
+});
+
+test("server: get_transcripts max_lines caps the merged excerpt lines", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee", max_lines: 1 },
+ });
+ const out = firstText(res);
+ assert.match(out, /1 matching line\(s\), windowed/);
+ assert.ok(out.includes("right before the moment"), "the earliest line is kept");
+ assert.ok(!out.includes("right after the moment"), "later lines are dropped");
+ await client.close();
+});
+
+test("server: search_transcripts link_style base emits moment_base + compact snippets", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", link_style: "base", limit: 20 },
+ });
+ const out = firstText(res);
+ assert.match(out, /- moment_base: https:\/\/example\.test\/a1\?t=$/m);
+ assert.ok(out.includes(" - [0:10|10] i love coffee"), out);
+ assert.ok(!out.includes("](http"), "no full inline links in base snippets");
+ assert.match(out, /<moment_base><seconds>/, "the expansion note is present");
+ await client.close();
+});
+
+test("server: search_transcripts worklist trims the source line (and no moment_base)", async () => {
+ const client = await connectClient(new StubSource());
+ const withSnips = firstText(
+ await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", limit: 5 },
+ }),
+ );
+ assert.match(withSnips, /- source: /, "source line present by default");
+
+ const worklist = firstText(
+ await client.callTool({
+ name: "search_transcripts",
+ arguments: {
+ query: "coffee",
+ limit: 5,
+ include_snippets: false,
+ link_style: "base",
+ },
+ }),
+ );
+ assert.ok(!worklist.includes("- source:"), "worklist mode drops the source line");
+ assert.ok(!worklist.includes("moment_base"), "…and emits no moment_base either");
+ await client.close();
+});
+
+test("server: the sweep prompt drives dumb extractors on the cheap model", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /haiku/, "the default parse model is requested");
+ assert.match(text, /DUMB EXTRACTOR/);
+ assert.match(text, /link_style/);
+ assert.match(text, /moment_base/);
+ assert.match(text, /\[mm:ss\|seconds\]/);
+ assert.match(text, /40 lines/, "the extractor's hard line budget");
+ assert.match(text, /<moment_base><seconds>/, "the expansion rule is spelled out");
+ await client.close();
+});
+
+test("server: the sweep prompt honors parse_model", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a", parse_model: "sonnet" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /sonnet/);
+ assert.ok(!/haiku/.test(text), "the default model name is fully replaced");
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -398,9 +398,14 @@ export function getWindowedTranscript(
after?: number;
maxCues?: number;
timestamps?: boolean;
- // Optional builder turning a line's start seconds into a moment deep link;
- // when it returns a URL the timestamp is rendered as a Markdown link.
- link?: (seconds: number) => string | null;
+ // Optional formatter for the bracketed stamp's CONTENTS (e.g. an inline
+ // Markdown link "[m:ss](url)" or the compact base form "m:ss|156"), given
+ // the line's clock + start seconds. Bare clock when omitted.
+ stamp?: (clock: string, seconds: number) => string;
+ // Cap on merged excerpt lines emitted for this video (default
+ // WINDOW_LINE_CAP). The earliest lines are kept; `maxCues` above stays the
+ // per-window bound.
+ maxLines?: number;
} = {},
): { lines: string[]; matchCount: number } {
const cues = record.cues ?? [];
@@ -417,12 +422,11 @@ export function getWindowedTranscript(
maxCues: opts.maxCues,
}),
);
- merged = mergeSnippets(merged, win, WINDOW_LINE_CAP);
+ merged = mergeSnippets(merged, win, opts.maxLines ?? WINDOW_LINE_CAP);
}
const lines = merged.map((s) => {
if (!timestamps) return s.text;
- const url = opts.link ? opts.link(s.seconds) : null;
- const stamp = url ? `[${s.clock}](${url})` : s.clock;
+ const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock;
return `[${stamp}] ${s.text}`;
});
return { lines, matchCount };
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -7,7 +7,7 @@ import {
} from "@modelcontextprotocol/sdk/types.js";
import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
-import { momentUrl } from "yt-dlp-transcript-common/lib/momentUrl";
+import { momentUrl, momentBaseUrl } from "yt-dlp-transcript-common/lib/momentUrl";
import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import {
@@ -81,6 +81,39 @@ function stampMarkup(
return url ? `[${clock}](${url})` : clock;
}
+// The requested timestamp-link form: "base" is the compact agent-pipeline form
+// (one `moment_base` per video + `[m:ss|seconds]` lines); anything else — the
+// default — is the full inline Markdown links.
+function linkStyleOf(args: Record<string, unknown>): "inline" | "base" {
+ return args.link_style === "base" ? "base" : "inline";
+}
+
+// The compact stamp contents for link_style:"base": `m:ss|156`. The integer is
+// floored exactly like momentUrl floors its seconds, so appending it to the
+// video's moment_base cites the identical second the inline link would.
+function baseStamp(clock: string, seconds: number): string {
+ return `${clock}|${Math.floor(seconds)}`;
+}
+
+// The appendable moment base for a hit (mirrors momentLinkFor), or null when
+// none can be built — no viewer origin AND the platform's time param doesn't
+// take raw seconds (Twitch) or doesn't exist (Rumble/Kick/no URL).
+function momentBaseFor(source: ShardSource, h: LinkableHit): string | null {
+ return momentBaseUrl({
+ siteOrigin: h.siteUrl ?? source.publicOrigin(),
+ slug: h.slug,
+ webpageUrl: h.webpageUrl,
+ platform: h.platform,
+ });
+}
+
+// Footer note stating the expansion rule for link_style:"base" output.
+const BASE_EXPANSION_NOTE =
+ "link_style base: full moment link = `<moment_base><seconds>` — append the " +
+ "integer after the `|` to that video's moment_base, e.g. `[title @ 2:36]" +
+ "(<moment_base>156)`; a video with no moment_base line → cite its source " +
+ "url plain";
+
type ToolResult = {
content: { type: "text"; text: string }[];
isError?: boolean;
@@ -183,6 +216,18 @@ const TOOLS = [
"Max shard pages to scan before stopping (default 400). Reaching " +
"it marks coverage partial.",
},
+ link_style: {
+ type: "string",
+ enum: ["inline", "base"],
+ description:
+ "Timestamp link form (default 'inline': every [m:ss] is a full " +
+ "Markdown moment link). 'base' is a compact agent-pipeline form: " +
+ "each video gets one '- moment_base:' header line (a URL ending " +
+ "in 't=') and snippet stamps become [m:ss|<seconds>]; expand to a " +
+ "full link by appending the integer seconds to the moment_base " +
+ "([title @ m:ss](<moment_base><seconds>)). A video with no " +
+ "buildable base omits the line — cite its source url plain.",
+ },
},
required: ["query"],
additionalProperties: false,
@@ -265,6 +310,25 @@ const TOOLS = [
type: "number",
description: "Seconds of context after each match (default 30).",
},
+ link_style: {
+ type: "string",
+ enum: ["inline", "base"],
+ description:
+ "Timestamp link form (default 'inline': every [m:ss] is a full " +
+ "Markdown moment link). 'base' is a compact agent-pipeline form: " +
+ "each video gets one '- moment_base:' header line (a URL ending " +
+ "in 't=') and line stamps become [m:ss|<seconds>]; expand to a " +
+ "full link by appending the integer seconds to the moment_base " +
+ "([title @ m:ss](<moment_base><seconds>)). A video with no " +
+ "buildable base omits the line — cite its source url plain.",
+ },
+ max_lines: {
+ type: "number",
+ description:
+ "Cap on merged excerpt lines per video when a query is given " +
+ "(default 200; the earliest lines are kept). Ignored without a " +
+ "query.",
+ },
},
required: ["video_ids"],
additionalProperties: false,
@@ -620,6 +684,8 @@ async function handleSearch(
): Promise<ToolResult> {
const query = String(args.query ?? "").trim();
if (!query) return errorText("query is required");
+ const includeSnippets = args.include_snippets !== false;
+ const base = linkStyleOf(args) === "base";
const result = await searchTranscripts(source, {
query,
channel: typeof args.channel === "string" ? args.channel : undefined,
@@ -629,7 +695,7 @@ async function handleSearch(
regex: args.regex === true,
limit: typeof args.limit === "number" ? args.limit : undefined,
offset: typeof args.offset === "number" ? args.offset : undefined,
- includeSnippets: args.include_snippets !== false,
+ includeSnippets,
useAliases: args.use_aliases !== false,
maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined,
});
@@ -657,19 +723,29 @@ async function handleSearch(
return text(`${head}${footer}`);
}
const blocks = result.hits.map((h) => {
+ // Worklist mode (include_snippets:false) trims the source line too — the
+ // caller only wants ids/titles/counts, so no per-video URLs at all.
+ const baseUrl = base && includeSnippets ? momentBaseFor(source, h) : null;
const head =
`### ${h.title}\n` +
`- video_id: ${h.videoId} | channel: ${h.channelName}` +
(h.siteTitle ? ` | site: ${h.siteTitle}` : "") +
` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` +
- (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "");
+ (includeSnippets && h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "") +
+ (baseUrl ? `\n- moment_base: ${baseUrl}` : "");
const snips = h.snippets
- .map((s) => ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`)
+ .map((s) =>
+ base
+ ? ` - [${baseStamp(s.clock, s.seconds)}] ${s.text}`
+ : ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`,
+ )
.join("\n");
return snips ? `${head}\n${snips}` : head;
});
+ // Worklist mode has no stamps (or bases) to expand — skip the note too.
+ const baseNote = base && includeSnippets ? `\n\n(${BASE_EXPANSION_NOTE})` : "";
return text(
- `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}`,
+ `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}${baseNote}`,
);
}
@@ -751,6 +827,11 @@ async function handleGetTranscripts(
const before = typeof args.before === "number" ? args.before : 30;
const after = typeof args.after === "number" ? args.after : 30;
+ const base = linkStyleOf(args) === "base";
+ const maxLinesArg =
+ typeof args.max_lines === "number" ? Math.floor(args.max_lines) : undefined;
+ const maxLines =
+ maxLinesArg !== undefined && maxLinesArg >= 1 ? maxLinesArg : undefined;
const blocks: string[] = [];
const missing: string[] = [];
@@ -767,21 +848,30 @@ async function handleGetTranscripts(
...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
...(record.platform ? { platform: record.platform } : {}),
};
- const linkForSeconds = (seconds: number): string | null =>
- momentLinkFor(source, link, seconds);
+ const baseUrl = base ? momentBaseFor(source, link) : null;
+ // The bracketed stamp's contents: the full inline Markdown link (byte-
+ // identical to the pre-link_style output), or the compact base form.
+ const stamp = base
+ ? baseStamp
+ : (clock: string, seconds: number): string => {
+ const url = momentLinkFor(source, link, seconds);
+ return url ? `[${clock}](${url})` : clock;
+ };
const head =
`## ${record.title || id}\n` +
`- video_id: ${id} | channel: ${ch.name}` +
(ch.siteTitle ? ` | site: ${ch.siteTitle}` : "") +
` | uploaded: ${formatDate(record.uploadDate)}` +
- (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "");
+ (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "") +
+ (baseUrl ? `\n- moment_base: ${baseUrl}` : "");
if (matcher) {
const { lines, matchCount } = getWindowedTranscript(record, matcher.match, {
before,
after,
timestamps,
- link: linkForSeconds,
+ stamp,
+ maxLines,
});
const body =
matchCount === 0
@@ -789,11 +879,21 @@ async function handleGetTranscripts(
: `_(${matchCount} matching line(s), windowed)_\n${lines.join("\n")}`;
blocks.push(`${head}\n\n${body}`);
} else {
- const md = transcriptToMarkdown(record, {
- timestamps,
- includeTags: true,
- linkForCue: linkForSeconds,
- });
+ const md = transcriptToMarkdown(
+ record,
+ base
+ ? {
+ timestamps,
+ includeTags: true,
+ stampForCue: baseStamp,
+ extraMeta: baseUrl ? [`moment_base: ${baseUrl}`] : undefined,
+ }
+ : {
+ timestamps,
+ includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
+ },
+ );
blocks.push(md.trim());
}
}
@@ -805,6 +905,7 @@ async function handleGetTranscripts(
}
if (missing.length > 0) notes.push(`not found: ${missing.join(", ")}`);
if (dropped > 0) notes.push(`${dropped} extra id(s) beyond the 20-cap dropped`);
+ if (base) notes.push(BASE_EXPANSION_NOTE);
const footer = notes.length > 0 ? `\n\n(${notes.join("; ")})` : "";
if (blocks.length === 0) {
@@ -1228,6 +1329,8 @@ function renderScopedSnippet(
return ` - [${s.scope}] ${s.text}`;
}
+// open_link results (and get_transcript) stay inline-linked for now — a
+// link_style:"base" form for them is future work.
function renderSpecResults(source: ShardSource, hits: SpecHit[]): string {
return hits
.map((h) => {
@@ -1397,6 +1500,14 @@ const PROMPTS = [
required: false,
},
{
+ name: "parse_model",
+ description:
+ "Model to request for the per-batch extractor subagents (default " +
+ "'haiku'). The extractors only quote verbatim, so the cheapest " +
+ "model wins; all synthesis stays with the orchestrator.",
+ required: false,
+ },
+ {
name: "report_path",
description: "Report file to write (default ./sweep-report.md).",
required: false,
@@ -1424,6 +1535,7 @@ function buildSweepPrompt(args: Record<string, unknown>) {
: [];
const directive = argStr(args, "directive") ?? "key claims & contradictions";
const batchSize = argStr(args, "batch_size") ?? "8";
+ const parseModel = argStr(args, "parse_model") ?? "haiku";
const reportPath = argStr(args, "report_path") ?? "./sweep-report.md";
const subject = query ? `"${query}"` : "the share link's search";
@@ -1508,30 +1620,40 @@ function buildSweepPrompt(args: Record<string, unknown>) {
`count) before you start.`,
);
- // ── Feature 2: subagent-per-batch map-reduce so the heavy transcript text
- // lives only in ephemeral subagents; the orchestrator keeps just the distilled
- // cited findings. Feature 1: linked citations built from the moment URLs.
+ // ── Feature 2: dumb-extractor-per-batch map-reduce on the cheapest model —
+ // the extractor only quotes verbatim base-form excerpts, so the heavy
+ // transcript text never reaches the orchestrator (and never costs the big
+ // model). Feature 1: linked citations, expanded from moment_base + seconds.
steps.push(
`**Per batch (map-reduce), for each group of up to ${batchSize} video ids:**\n` +
- ` - **Spawn a subagent** (the Task tool) for the batch. Give it the ` +
- `batch's ids, the query, and the directive, and tell it to: call ` +
- `\`get_transcripts\` with those ids **and the query** (bounded, ` +
- `timestamped, alias-correct excerpt windows), extract only what serves the ` +
- `directive, and **return a compact markdown fragment of cited, linked ` +
- `findings and nothing else** — the raw transcript text stays inside the ` +
- `subagent and never enters your context. Every citation must be a link: ` +
- `**\`[title @ mm:ss](<moment url>)\`**, where the moment URL is taken ` +
- `straight from the \`[mm:ss](url)\` links in that batch's ` +
- `\`get_transcripts\` output (they seek to the exact second).\n` +
- ` - **Merge** the returned fragment into \`${reportPath}\`: cross-` +
- `reference it against the report so far and upsert findings — claims, and ` +
- `contradictions with earlier claims — into well-titled \`## sections\` ` +
- `(Write/Edit). Then discard the fragment. Batches are independent, so you ` +
+ ` - **Spawn a subagent as a DUMB EXTRACTOR on the cheapest model** — ` +
+ `use the Task tool and request model "${parseModel}" (its model ` +
+ `parameter, or a "${parseModel}"-backed agent type); if no model ` +
+ `override is available, spawn it anyway on the default. Give it exactly ` +
+ `this job: call \`get_transcripts\` with the batch's ids, the query, and ` +
+ `\`link_style: "base"\`, then return ONLY the lines relevant to the ` +
+ `directive, VERBATIM — do NOT analyze, summarize, or rephrase anything. ` +
+ `Group the kept lines per video as \`### <title>\` + that video's ` +
+ `\`moment_base:\` line copied exactly (or its \`source:\` line when ` +
+ `there is no moment_base) + the kept lines in their \`[mm:ss|seconds]\` ` +
+ `form. Hard budget: at most 40 lines (~600 words) per batch — if more ` +
+ `match, keep the strongest and end with \`(+N more matching lines)\`. ` +
+ `Return nothing else — the raw transcript bulk stays inside the ` +
+ `subagent and never enters your context.\n` +
+ ` - **Merge — you (the orchestrator) do ALL the synthesis.** Cross-` +
+ `reference the returned fragment against the report so far and upsert ` +
+ `findings — claims, and contradictions with earlier claims — into ` +
+ `well-titled \`## sections\` of \`${reportPath}\` (Write/Edit). Cite ` +
+ `every finding as **\`[title @ mm:ss](<moment url>)\`**, expanding each ` +
+ `kept \`[mm:ss|seconds]\` stamp by appending the integer after the ` +
+ `\`|\` to that video's moment_base (full link = ` +
+ `\`<moment_base><seconds>\`; no moment_base → link the \`source:\` URL ` +
+ `instead). Then discard the fragment. Batches are independent, so you ` +
`may dispatch several subagents in parallel.\n` +
` - **Fallback:** if no subagent/Task tool is available, do the batch ` +
- `inline — call \`get_transcripts\` yourself, fold the cited linked ` +
- `findings into the report, then **drop the raw transcript text** before ` +
- `moving on (don't carry it forward).`,
+ `inline — call \`get_transcripts\` with \`link_style: "base"\` yourself, ` +
+ `fold the expanded cited findings into the report, then **drop the raw ` +
+ `excerpt text** before moving on (don't carry it forward).`,
);
steps.push(
@@ -1552,8 +1674,9 @@ function buildSweepPrompt(args: Record<string, unknown>) {
`set methodically, using the transcript MCP tools for evidence and your own ` +
`Write/Edit tools for the report. The MCP is read-only; never try to change ` +
`the archive. **Cite every finding as a clickable ` +
- `\`[title @ mm:ss](<moment url>)\` link** (the moment URLs come straight ` +
- `from the tool output).\n\n` +
+ `\`[title @ mm:ss](<moment url>)\` link** (build each moment URL by ` +
+ `appending the cited integer seconds to that video's \`moment_base\` from ` +
+ `the tool output).\n\n` +
`Follow these steps:\n\n${numbered}`;
return {