// Decode an archilyzer viewer **share link** into a search spec + source target. // // A share URL is `origin` + a `qt=` composite query tree (or a legacy `q`/`re`) // + the v1 filter params (`fc/ft/fa/fav/fk/fdf/fdt`, gated by `fv=1`). This // module turns that URL into the pieces the MCP needs to reproduce the search at // full fidelity: // • the origin (→ probed by the server for hub vs single-site), // • the query tree (parsed via the browser's own `parseRoot`), // • the decoded filters (via the browser's own `parseShareV1`), // • the requested channel NAMES (`fc`, resolved against the live corpus later), // • and any ignored/warned pieces (`fk` tracks are vestigial in the composite // share model — decoded and reported as ignored, not enforced). // // Pure: no network. The server does the origin probe and channel resolution. import { parseRoot, rootFromLegacy, emptyRoot, isLeaf, isGroup, isNodeActive, newGroup, newLeaf, type GroupNode, type QueryNode, } from "yt-dlp-transcript-common/lib/searchQuery"; import { hasShareV1, parseShareV1, } from "yt-dlp-transcript-common/components/shareUrl"; import { VIDEO_STATES, VIDEO_STATE_LABELS, } from "yt-dlp-transcript-common/lib/availability"; import type { SearchFilters } from "./search"; export type DecodedLink = { // The pasted URL's origin (scheme + host[:port]). The source target. origin: string; // The composite query tree — from `qt=`, a synthesized legacy leaf, or an // empty root (filter-only browse). tree: GroupNode; querySource: "qt" | "legacy" | "empty"; // Positive share filters (ft/fa/fav + dates), or null when the link carries no // v1 filter block (`fv=1`) — i.e. an unfiltered search. filters: SearchFilters | null; // Channel NAMES the link selects (`fc`). Resolved against the live corpus by // the caller. Empty with `filters === null` means "whole corpus". channelNames: string[]; // Whether the link constrained channels at all (had a v1 filter block). When // false, channel scope is the whole corpus regardless of `channelNames`. hasChannelFilter: boolean; // Vestigial `fk` subtitle-track tokens — decoded but a no-op. ignoredTracks: string[]; // Human-facing notes (malformed qt, ignored fk, …). warnings: string[]; }; // Decode a share link. Never throws for a malformed query tree (falls back to a // legacy/empty tree with a warning); throws only if `link` isn't a URL at all. export function decodeShareLink(link: string): DecodedLink { const url = new URL(link); const search = url.search; const params = new URLSearchParams(search); const warnings: string[] = []; // ── Query: qt= tree preferred, else legacy q/re, else empty (filter-only). let tree: GroupNode = emptyRoot(); let querySource: DecodedLink["querySource"] = "empty"; const qt = params.get("qt"); if (qt) { const parsed = parseRoot(qt); if (parsed) { tree = parsed; querySource = "qt"; } else { warnings.push("malformed `qt` query tree — ignored"); } } if (querySource === "empty") { const q = params.get("q") ?? ""; const mode = params.get("m") === "subs" ? "subs" : "transcripts"; const re = params.get("re") === "1"; if (q.trim() !== "") { tree = rootFromLegacy(q, mode, re); querySource = "legacy"; } } // ── Filters: only meaningful when the v1 block is present (`fv=1`). const v1 = hasShareV1(search); // Pass the requested fc names AS the channel universe so none are dropped — // real validation against the live corpus happens at apply time. const requestedChannels = params.getAll("fc"); const sel = parseShareV1(search, requestedChannels); const ignoredTracks = [...sel.tracks]; if (ignoredTracks.length > 0) { warnings.push( `\`fk\` subtitle-track filter is vestigial in the composite share model — ` + `${ignoredTracks.length} track token(s) decoded but ignored: ${ignoredTracks.join(", ")}`, ); } const filters: SearchFilters | null = v1 ? { videos: sel.videos, livestreams: sel.livestreams, allAges: sel.allAges, restricted: sel.restricted, states: sel.states, ...(sel.dateFrom ? { dateFrom: sel.dateFrom } : {}), ...(sel.dateTo ? { dateTo: sel.dateTo } : {}), } : null; return { origin: url.origin, tree, querySource, filters, channelNames: v1 ? [...sel.selectedChannels] : [], hasChannelFilter: v1, ignoredTracks, warnings, }; } // Structured adjustments an agent can pass to honor a natural-language edit // ("remove the availability filter", "only channel X", "search 'foo' instead"). // Every field is optional; only the provided facets are changed. export type LinkOverrides = { // Drop every decoded filter (ft/fa/fav/dates) → unfiltered. clearFilters?: boolean; // Reset a single filter facet to its keep-everything default. clearAvailability?: boolean; // fav clearType?: boolean; // ft clearAge?: boolean; // fa clearDates?: boolean; // fdf/fdt // Widen the channel scope to the whole corpus (ignore fc). clearChannels?: boolean; // Replace the channel scope with these names (validated against the corpus). channels?: string[]; // Replace/clear the upload-date bounds ("YYYYMMDD"; null clears that bound). dateFrom?: string | null; dateTo?: string | null; // Replace the whole query with a single leaf (query + optional regex/scope). query?: string; regex?: boolean; queryScope?: | "transcripts" | "chat" | "posts" | "metadata" | "description" | "tags"; }; const KEEP_ALL_FILTERS: SearchFilters = { videos: true, livestreams: true, allAges: true, restricted: true, states: new Set(VIDEO_STATES), }; // Apply overrides to a decoded link, returning a new DecodedLink. Records what // changed as warnings so the plan can echo it. export function applyLinkOverrides( decoded: DecodedLink, overrides: LinkOverrides | undefined, ): DecodedLink { if (!overrides) return decoded; const next: DecodedLink = { ...decoded, warnings: [...decoded.warnings], channelNames: [...decoded.channelNames], filters: decoded.filters ? { ...decoded.filters } : null, }; const note = (s: string): void => { next.warnings.push(`override: ${s}`); }; // ── Query replacement. if (typeof overrides.query === "string" && overrides.query.trim() !== "") { next.tree = newGroup({ children: [ newLeaf({ query: overrides.query, scope: overrides.queryScope ?? "transcripts", useRegex: overrides.regex === true, }), ], }); next.querySource = "qt"; note(`query replaced with "${overrides.query}"`); } // ── Channel scope. if (overrides.clearChannels) { next.channelNames = []; next.hasChannelFilter = false; note("channel scope widened to the whole corpus"); } if (Array.isArray(overrides.channels)) { next.channelNames = overrides.channels.filter((c) => c.trim() !== ""); next.hasChannelFilter = true; note(`channel scope set to [${next.channelNames.join(", ")}]`); } // ── Filters. if (overrides.clearFilters) { next.filters = null; note("all filters removed"); } else if (next.filters) { const f = next.filters; if (overrides.clearAvailability) { f.states = new Set(VIDEO_STATES); note("availability filter removed"); } if (overrides.clearType) { f.videos = true; f.livestreams = true; note("type filter removed"); } if (overrides.clearAge) { f.allAges = true; f.restricted = true; note("age filter removed"); } if (overrides.clearDates) { delete f.dateFrom; delete f.dateTo; note("date filter removed"); } if (overrides.dateFrom !== undefined) { if (overrides.dateFrom === null) delete f.dateFrom; else f.dateFrom = overrides.dateFrom; note(`dateFrom set to ${overrides.dateFrom ?? "(none)"}`); } if (overrides.dateTo !== undefined) { if (overrides.dateTo === null) delete f.dateTo; else f.dateTo = overrides.dateTo; note(`dateTo set to ${overrides.dateTo ?? "(none)"}`); } } else if ( overrides.dateFrom !== undefined || overrides.dateTo !== undefined ) { // Setting a date on an otherwise-unfiltered link creates a filter block. next.filters = { ...KEEP_ALL_FILTERS }; if (typeof overrides.dateFrom === "string") next.filters.dateFrom = overrides.dateFrom; if (typeof overrides.dateTo === "string") next.filters.dateTo = overrides.dateTo; note("date filter added"); } return next; } // ── Human-readable renderings for the preview plan ── const SCOPE_LABEL: Record = { transcripts: "transcript", chat: "live-chat", metadata: "title/channel", description: "description", tags: "tags", }; // Render a query tree readably, e.g. // transcript:"coffee" AND tags:"espresso" AND NOT chat:"spam" export function renderQueryTree(node: QueryNode): string { if (isLeaf(node)) { const label = SCOPE_LABEL[node.scope] ?? node.scope; const kind = node.useRegex ? "/" : '"'; const end = node.useRegex ? "/" : '"'; const body = `${label}:${kind}${node.query}${end}`; return node.negate ? `NOT ${body}` : body; } if (!isGroup(node)) return ""; const parts = node.children .filter(isNodeActive) .map((c) => { const s = renderQueryTree(c); return isGroup(c) ? `(${s})` : s; }) .filter((s) => s !== ""); if (parts.length === 0) return "(everything in scope)"; const joined = parts.join(` ${node.op} `); return node.negate ? `NOT (${joined})` : joined; } // A compact list of the active filters for the plan, or [] when unfiltered. export function describeFilters(f: SearchFilters | null): string[] { if (!f) return []; const out: string[] = []; // Type (ft): only worth noting when it excludes one side. if (f.videos !== f.livestreams) { out.push(`type: ${f.videos ? "videos only" : "livestreams only"}`); } else if (!f.videos && !f.livestreams) { out.push("type: none kept (videos and livestreams both excluded)"); } // Audience (fa). if (f.allAges !== f.restricted) { out.push(`audience: ${f.allAges ? "all-ages only" : "age-restricted only"}`); } else if (!f.allAges && !f.restricted) { out.push("audience: none kept"); } // Availability (fav). const av = VIDEO_STATES.filter((s) => f.states.has(s)); if (av.length < VIDEO_STATES.length) { out.push( `availability: ${av.length ? av.map((s) => VIDEO_STATE_LABELS[s].toLowerCase()).join(" + ") : "none"} kept`, ); } // Dates (fdf/fdt). if (f.dateFrom || f.dateTo) { out.push(`uploaded: ${f.dateFrom ?? "…"} → ${f.dateTo ?? "…"}`); } return out; }