commit 025ed41d78557f162d403cbefd7209190c3291b4
parent 4782bea368f426a1f31f0133910023c725a6d0da
Author: I Mean I'm Just Saying <imeanimjustsaying@kiwifarms.st>
Date: Wed, 22 Jul 2026 11:00:57 -0400
Merge branch 'main' into worktree-feat+byo-ai-corpus
# Conflicts:
# export/CHANGELOG.md
Diffstat:
67 files changed, 5891 insertions(+), 252 deletions(-)
diff --git a/common/components/SearchSessionContext.tsx b/common/components/SearchSessionContext.tsx
@@ -909,11 +909,25 @@ function useSearchSessionState() {
if (stored) {
setProfiles(stored.profiles);
setActiveProfileName(stored.activeProfileName);
- setCollapsedGroups(new Set(stored.collapsedGroups ?? []));
if (typeof stored.filtersCollapsed === "boolean") {
setFiltersCollapsed(stored.filtersCollapsed);
}
}
+ if (stored?.collapsedGroups !== undefined) {
+ // User preference honored (a stored [] — all groups open — stays).
+ setCollapsedGroups(new Set(stored.collapsedGroups));
+ } else {
+ // Inline groups render as loose chips with no collapse affordance, so
+ // they're excluded from both the default-collapse set and the >1
+ // heuristic (a lone collapsible group among inline chips stays open).
+ const collapsible = channelGroupings.filter((g) => !g.group.inline);
+ if (collapsible.length > 1) {
+ // Collapse state never touched on a multi-group site: start compact.
+ // The chips still show selection state, so nothing important is
+ // hidden.
+ setCollapsedGroups(new Set(collapsible.map((g) => g.group.id)));
+ }
+ }
const search = typeof window !== "undefined" ? window.location.search : "";
const params = new URLSearchParams(search);
@@ -1039,7 +1053,7 @@ function useSearchSessionState() {
setHydrated(true);
// eslint-disable-next-line react-hooks/exhaustive-deps
- }, [channelOptions.length === 0, hydrated]);
+ }, [channelOptions.length === 0, hydrated, channelGroupings]);
// Reflect the chart view + shape in the URL live (post-hydration only), so the
// address bar always carries the current chart — mirrors how qt= is written.
diff --git a/common/components/WorkspaceSearchBar.tsx b/common/components/WorkspaceSearchBar.tsx
@@ -6,9 +6,12 @@
// (and keeps its committed search) across `/` ⇄ `/ask`. All of its state comes
// from the shared SearchSession.
+import { Fragment } from "react";
+import { ChevronRightIcon } from "lucide-react";
import { Button } from "./ui/button";
import { Checkbox } from "./ui/checkbox";
import QueryBuilder from "./QueryBuilder";
+import { cn } from "../lib/utils";
import { clearAdvanced } from "./exportAdvancedStorage";
import { ymdToInput, inputToYmd } from "../lib/ymd";
import {
@@ -233,8 +236,38 @@ export default function WorkspaceSearchBar() {
Reset everything
</Button>
</div>
- <div className="flex flex-col gap-1.5">
+ <div className="flex flex-wrap items-start gap-1.5">
{channelGroupings.map(({ group, channels }) => {
+ if (group.inline) {
+ // Inline group: no collapsible box — each member channel
+ // renders as its own loose chip flowing among the group
+ // chips, styled like a collapsed group chip.
+ return (
+ <Fragment key={group.id}>
+ {channels.map((key) => {
+ const label = channelLabelByKey.get(key) ?? key;
+ return (
+ <label
+ key={key}
+ title={label}
+ data-testid="inline-channel-chip"
+ className="flex items-center gap-2 rounded border border-border bg-card/60 px-2.5 py-1.5 text-sm cursor-pointer select-none min-w-0 max-w-full w-full sm:w-auto sm:min-w-48"
+ >
+ <span className="flex items-center shrink-0">
+ <Checkbox
+ checked={!draftExcludedChannels.has(key)}
+ onCheckedChange={(value) =>
+ setSelected([key], value === true)
+ }
+ />
+ </span>
+ <span className="truncate">{label}</span>
+ </label>
+ );
+ })}
+ </Fragment>
+ );
+ }
const isOpen = !collapsedGroups.has(group.id);
const selectedCount = channels.reduce(
(n, name) =>
@@ -245,17 +278,57 @@ export default function WorkspaceSearchBar() {
<details
key={group.id}
open={isOpen}
- onToggle={(e) =>
- toggleGroupCollapsed(
- group.id,
- (e.target as HTMLDetailsElement).open,
- )
- }
- className="rounded border border-border bg-card/60"
+ className={cn(
+ "rounded border border-border bg-card/60 min-w-0 max-w-full w-full",
+ !isOpen && "sm:w-auto sm:min-w-48",
+ )}
>
- <summary className="cursor-pointer select-none flex flex-wrap items-center gap-x-3 gap-y-1 px-3 py-1.5 text-sm">
+ <summary
+ onClick={(e) => {
+ // Drive the open state from React rather than the
+ // browser's default toggle action — same rationale
+ // as the outer Filters summary above.
+ e.preventDefault();
+ toggleGroupCollapsed(group.id, !isOpen);
+ }}
+ className="cursor-pointer select-none flex items-center gap-2 px-2.5 py-1.5 text-sm min-w-0"
+ title={!isOpen ? group.description : undefined}
+ >
+ <ChevronRightIcon
+ aria-hidden="true"
+ className={cn(
+ "size-3.5 shrink-0 text-muted-foreground transition-transform",
+ isOpen && "rotate-90",
+ )}
+ />
+ <span
+ onClick={(e) => {
+ // Isolate checkbox clicks from the summary: no
+ // expand/collapse, just the group bulk-toggle.
+ e.preventDefault();
+ e.stopPropagation();
+ }}
+ className="flex items-center shrink-0"
+ >
+ <Checkbox
+ checked={
+ selectedCount === 0
+ ? false
+ : selectedCount === channels.length
+ ? true
+ : "indeterminate"
+ }
+ onCheckedChange={(value) =>
+ setSelected(channels, value === true)
+ }
+ aria-label={`Select all in ${group.name || "group"}`}
+ />
+ </span>
{group.name && (
- <span className="font-medium text-foreground inline-flex items-center gap-1.5">
+ <span
+ className="font-medium text-foreground flex items-center gap-1.5 min-w-0"
+ title={group.name}
+ >
{group.accent && (
<span
aria-hidden="true"
@@ -263,40 +336,48 @@ export default function WorkspaceSearchBar() {
style={{ background: group.accent }}
/>
)}
- {group.name}
+ <span className="truncate">{group.name}</span>
</span>
)}
- <span className="text-xs text-muted-foreground">
- ({selectedCount}/{channels.length} selected)
- </span>
- <span className="ml-auto flex items-center gap-2">
- <Button
- type="button"
- variant="link"
- size="sm"
- onClick={(e) => {
- e.preventDefault();
- e.stopPropagation();
- setSelected(channels, true);
- }}
- className="text-xs text-muted-foreground"
- >
- All
- </Button>
- <Button
- type="button"
- variant="link"
- size="sm"
- onClick={(e) => {
- e.preventDefault();
- e.stopPropagation();
- setSelected(channels, false);
- }}
- className="text-xs text-muted-foreground"
- >
- None
- </Button>
+ <span
+ className={cn(
+ "text-xs text-muted-foreground shrink-0",
+ !isOpen && "ml-auto",
+ )}
+ >
+ {selectedCount}/{channels.length}
+ <span className="sr-only"> selected</span>
</span>
+ {isOpen && (
+ <span className="ml-auto flex items-center gap-2">
+ <Button
+ type="button"
+ variant="link"
+ size="sm"
+ onClick={(e) => {
+ e.preventDefault();
+ e.stopPropagation();
+ setSelected(channels, true);
+ }}
+ className="text-xs text-muted-foreground"
+ >
+ All
+ </Button>
+ <Button
+ type="button"
+ variant="link"
+ size="sm"
+ onClick={(e) => {
+ e.preventDefault();
+ e.stopPropagation();
+ setSelected(channels, false);
+ }}
+ className="text-xs text-muted-foreground"
+ >
+ None
+ </Button>
+ </span>
+ )}
</summary>
{group.description && (
<div className="px-3 pt-0 pb-1 text-xs text-muted-foreground">
diff --git a/common/components/ui/checkbox.tsx b/common/components/ui/checkbox.tsx
@@ -1,7 +1,7 @@
"use client"
import * as React from "react"
-import { CheckIcon } from "lucide-react"
+import { CheckIcon, MinusIcon } from "lucide-react"
import { Checkbox as CheckboxPrimitive } from "radix-ui"
import { cn } from "../../lib/utils"
@@ -14,7 +14,7 @@ function Checkbox({
<CheckboxPrimitive.Root
data-slot="checkbox"
className={cn(
- "peer size-4 shrink-0 rounded-[4px] border border-input shadow-xs transition-shadow outline-none focus-visible:border-ring focus-visible:ring-[3px] focus-visible:ring-ring/50 disabled:cursor-not-allowed disabled:opacity-50 aria-invalid:border-destructive aria-invalid:ring-destructive/20 data-[state=checked]:border-primary data-[state=checked]:bg-primary data-[state=checked]:text-primary-foreground dark:bg-input/30 dark:aria-invalid:ring-destructive/40 dark:data-[state=checked]:bg-primary",
+ "peer group size-4 shrink-0 rounded-[4px] border border-input shadow-xs transition-shadow outline-none focus-visible:border-ring focus-visible:ring-[3px] focus-visible:ring-ring/50 disabled:cursor-not-allowed disabled:opacity-50 aria-invalid:border-destructive aria-invalid:ring-destructive/20 data-[state=checked]:border-primary data-[state=checked]:bg-primary data-[state=checked]:text-primary-foreground dark:bg-input/30 dark:aria-invalid:ring-destructive/40 dark:data-[state=checked]:bg-primary data-[state=indeterminate]:border-primary data-[state=indeterminate]:bg-primary data-[state=indeterminate]:text-primary-foreground dark:data-[state=indeterminate]:bg-primary",
className
)}
{...props}
@@ -23,7 +23,8 @@ function Checkbox({
data-slot="checkbox-indicator"
className="grid place-content-center text-current transition-none"
>
- <CheckIcon className="size-3.5" />
+ <CheckIcon className="size-3.5 group-data-[state=indeterminate]:hidden" />
+ <MinusIcon className="hidden size-3.5 group-data-[state=indeterminate]:block" />
</CheckboxPrimitive.Indicator>
</CheckboxPrimitive.Root>
)
diff --git a/common/controller/autoRunner.ts b/common/controller/autoRunner.ts
@@ -32,6 +32,7 @@ import {
pruneExpired,
} from "../jobs/platformBackoff";
import { type DownloadFailureClass } from "../lib/availability";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
import { type DownloadOutcomeStatus } from "../lib/downloadOutcome";
import { downloadQueueKey } from "../lib/queueKeys";
import { readChannelConfig } from "./channels";
@@ -622,7 +623,7 @@ async function launchUnit(args: LaunchArgs): Promise<UnitResult> {
// The child job's own abort signal: registry.cancel(childJobId) aborts
// it (→ kills yt-dlp) when the runner is hard-cancelled.
signal,
- globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ cookiePolicy: resolveCookiePolicy(settings, config),
inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
globalSkipLiveDownloads: settings.skipLiveDownloads,
downloadFormatPreset: resolveDownloadFormatPreset({
diff --git a/common/controller/channelSnapshot.ts b/common/controller/channelSnapshot.ts
@@ -11,10 +11,13 @@ import {
type VideoFiles,
} from "../lib/videoStatus";
import {
+ AUTH_RETRY_CLASSES,
AVAILABILITY_VALUES,
EXCLUDED_FROM_DOWNLOAD,
type Availability,
} from "../lib/availability";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
+import { getSettings } from "../lib/settings";
import {
loadAvailability,
resolveEffectiveAvailability,
@@ -106,6 +109,14 @@ export type ChannelSnapshot = {
// user can re-download with a different format (e.g. Original). Optional:
// older snapshots lack it; readers must default to [].
shortAudio: string[];
+ // Undownloaded playlist videos whose effective availability says browser
+ // cookies could recover them (AUTH_RETRY_CLASSES: needs_auth, members_only,
+ // private). Populated in EVERY cookie mode — members_only/private are
+ // batch-excluded regardless, and a subscribed/owning account's cookies can
+ // still fetch them — and drives the "Download with cookies" bucket button
+ // (retry-bucket with forceCookies). Optional: older snapshots lack it;
+ // readers must default to [].
+ needsCookies: string[];
};
undownloadedIds: string[];
excludedFromDownload?: ExcludedFromDownload;
@@ -339,7 +350,11 @@ export async function generateChannelSnapshot(
// "Missing metadata" or "Failed transcriptions" creates noise the user
// cannot resolve.
const excludedById = new Map<string, Availability>();
+ const effectiveById = new Map<string, Availability>();
for (const v of perVideo) {
+ if (v.effectiveAvailability) {
+ effectiveById.set(v.id, v.effectiveAvailability);
+ }
if (
v.effectiveAvailability &&
(EXCLUDED_FROM_DOWNLOAD as ReadonlyArray<Availability>).includes(
@@ -535,7 +550,13 @@ export async function generateChannelSnapshot(
excludedFromDownload.deleted.sort();
excludedFromDownload.private.sort();
+ // The channel's resolved cookie mode. In "defer" mode, needs_auth videos are
+ // ALSO dropped from undownloadedIds (which the auto-download runner
+ // consumes), so they aren't re-attempted every run — they wait in the
+ // needsCookies bucket for the manual cookie run instead.
+ const cookiePolicy = resolveCookiePolicy(getSettings(), config ?? undefined);
const undownloadedIds: string[] = [];
+ const needsCookies: string[] = [];
for (const url of urls) {
// Post-reconcile a video's dir is its canonical id, so the URL's canonical
// id is the dir name directly.
@@ -543,7 +564,12 @@ export async function generateChannelSnapshot(
if (!dirId) continue;
const f = filesById.get(dirId);
if (f && videoHasAnyArtifact(f)) continue;
+ const effective = effectiveById.get(dirId);
+ if (effective && AUTH_RETRY_CLASSES.has(effective)) {
+ needsCookies.push(dirId);
+ }
if (excludedById.has(dirId)) continue;
+ if (cookiePolicy.mode === "defer" && effective === "needs_auth") continue;
undownloadedIds.push(dirId);
}
@@ -589,6 +615,7 @@ export async function generateChannelSnapshot(
skippedByFilter: skippedByFilter.sort(),
incompleteTranscript: incompleteTranscript.sort(),
shortAudio: shortAudio.sort(),
+ needsCookies: needsCookies.sort(),
},
undownloadedIds,
excludedFromDownload,
diff --git a/common/controller/checkAvailability.ts b/common/controller/checkAvailability.ts
@@ -14,6 +14,12 @@ import {
recordAvailability,
resolveEffectiveAvailability,
} from "../lib/availability-server";
+import {
+ alwaysCookies,
+ cookieArgs,
+ resolveCookiePolicy,
+} from "../lib/cookiePolicy";
+import { getSettings } from "../lib/settings";
import { readChannelConfig } from "./channels";
import { resolveShardItems } from "./shard";
import type { Paths } from "../lib/paths";
@@ -115,6 +121,14 @@ export async function runAvailabilityCheck({
const dataDir = path.join(channelDir, "data");
const config = await readChannelConfig(paths, channelSlug);
const extraArgs = config?.ytdlpExtraArgs ?? [];
+ // Probe cookies in "always" mode ONLY. when-required/defer probes stay
+ // cookie-free deliberately, so auth gating keeps being OBSERVED as
+ // needs_auth — defer mode's exclusion + Needs-cookies bucket depend on that
+ // signal. (Always-mode probes may report a cookie-recoverable video as
+ // public; that's the trade-off of prophylactic cookies.)
+ const probeCookies = alwaysCookies(
+ resolveCookiePolicy(getSettings(), config ?? undefined),
+ );
const limit = pLimit(concurrency && concurrency > 0 ? Math.floor(concurrency) : 1);
@@ -201,6 +215,7 @@ export async function runAvailabilityCheck({
"--dump-json",
"--skip-download",
"--no-warnings",
+ ...cookieArgs(probeCookies),
...extraArgs,
"--",
url,
diff --git a/common/controller/persistKept.ts b/common/controller/persistKept.ts
@@ -2,6 +2,7 @@ import path from "node:path";
import type { Paths } from "../lib/paths";
import type { ChannelConfig } from "../lib/channelConfig";
import { getSettings } from "../lib/settings";
+import { resolveCookiePolicy } from "../lib/cookiePolicy";
import { isSavedVideo } from "../lib/savedVideo-server";
import { computeKeptVideoIds } from "./keptVideos";
import { findVideoSourceUrl } from "./undownloadedVideos";
@@ -99,7 +100,7 @@ export async function persistKept({
videoUrl: url,
onLog: downloadLog,
signal: downloadSignal,
- globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ cookiePolicy: resolveCookiePolicy(settings, channelConfig),
inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
diff --git a/common/jobs/jobSpec.test.ts b/common/jobs/jobSpec.test.ts
@@ -0,0 +1,50 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import { parseJobSpec } from "./jobSpec";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/jobs/jobSpec.test.ts
+
+test("parses a minimal spec and rejects malformed input", () => {
+ assert.deepEqual(parseJobSpec({ kind: "sync", slug: "chan" }), {
+ kind: "sync",
+ slug: "chan",
+ });
+ assert.equal(parseJobSpec(null), null);
+ assert.equal(parseJobSpec("sync"), null);
+ assert.equal(parseJobSpec({ kind: "", slug: "chan" }), null);
+ assert.equal(parseJobSpec({ kind: "sync" }), null);
+});
+
+test("accepts every replay bucket, including needsCookies", () => {
+ for (const bucket of [
+ "partialDownloads",
+ "noTranscript",
+ "downloadedNoTranscript",
+ "incompleteTranscript",
+ "shortAudio",
+ "needsCookies",
+ ]) {
+ const spec = parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket });
+ assert.equal(spec?.bucket, bucket);
+ }
+ // Unknown buckets invalidate the whole spec.
+ assert.equal(
+ parseJobSpec({ kind: "retry-bucket", slug: "chan", bucket: "nope" }),
+ null,
+ );
+});
+
+test("carries flag params verbatim (e.g. forceCookies)", () => {
+ const spec = parseJobSpec({
+ kind: "retry-bucket",
+ slug: "chan",
+ bucket: "needsCookies",
+ params: { queueKey: "youtube", forceCookies: true },
+ });
+ assert.deepEqual(spec?.params, { queueKey: "youtube", forceCookies: true });
+ // Array params are rejected (params must be a plain object).
+ assert.equal(
+ parseJobSpec({ kind: "sync", slug: "chan", params: [] })?.params,
+ undefined,
+ );
+});
diff --git a/common/jobs/jobSpec.ts b/common/jobs/jobSpec.ts
@@ -20,7 +20,8 @@ export type ReplayBucket =
| "noTranscript"
| "downloadedNoTranscript"
| "incompleteTranscript"
- | "shortAudio";
+ | "shortAudio"
+ | "needsCookies";
export type JobSpec = {
kind: string;
@@ -40,6 +41,7 @@ const REPLAY_BUCKETS: ReadonlySet<string> = new Set<ReplayBucket>([
"downloadedNoTranscript",
"incompleteTranscript",
"shortAudio",
+ "needsCookies",
]);
// Defensive parse for a spec read back from JSON (a sidecar or the bookmarks
diff --git a/common/jobs/syncSchedulerState.ts b/common/jobs/syncSchedulerState.ts
@@ -46,6 +46,10 @@ export type SchedulerState = {
// per-channel, job). Drives the savedVideoBackup.intervalMinutes cadence.
// null = never run. See editor/app/scheduler/runTick.ts.
lastSavedVideoBackupAt: number | null;
+ // Epoch ms of the last manual full "Sync all" sweep (syncAllChannelsAction).
+ // A global freshness marker distinct from any single channel's lastSyncedAt;
+ // surfaced by the monitor widget's last-sync readout. null = never run.
+ lastSyncAllAt: number | null;
};
export const SCHEDULER_RUN_LOG_LIMIT = 50;
@@ -63,7 +67,12 @@ export function emptyChannelSyncState(): ChannelSyncState {
}
export function emptySchedulerState(): SchedulerState {
- return { channels: {}, runs: [], lastSavedVideoBackupAt: null };
+ return {
+ channels: {},
+ runs: [],
+ lastSavedVideoBackupAt: null,
+ lastSyncAllAt: null,
+ };
}
// Read the state file, tolerating a missing/corrupt file by returning an empty
@@ -89,7 +98,12 @@ export async function readSchedulerState(paths: Paths): Promise<SchedulerState>
const runs: SchedulerRun[] = Array.isArray(r.runs)
? r.runs.map(coerceRun).slice(0, SCHEDULER_RUN_LOG_LIMIT)
: [];
- return { channels, runs, lastSavedVideoBackupAt: num(r.lastSavedVideoBackupAt) };
+ return {
+ channels,
+ runs,
+ lastSavedVideoBackupAt: num(r.lastSavedVideoBackupAt),
+ lastSyncAllAt: num(r.lastSyncAllAt),
+ };
}
// Write the state atomically (tmp file + rename), creating the .scheduler dir on
@@ -102,6 +116,7 @@ export async function writeSchedulerState(
channels: state.channels,
runs: state.runs.slice(0, SCHEDULER_RUN_LOG_LIMIT),
lastSavedVideoBackupAt: state.lastSavedVideoBackupAt ?? null,
+ lastSyncAllAt: state.lastSyncAllAt ?? null,
};
await mkdir(path.dirname(paths.schedulerStateFile), { recursive: true });
const tmp = `${paths.schedulerStateFile}.tmp-${process.pid}`;
diff --git a/common/lib/availability.ts b/common/lib/availability.ts
@@ -31,6 +31,18 @@ export const EXCLUDED_FROM_DOWNLOAD: ReadonlyArray<Availability> = [
"private",
];
+// Availability classes a cookie (auth) retry can potentially recover — the
+// gate for the managed downloader's auth-retry attempts and the membership
+// rule for the per-channel "Needs cookies" snapshot bucket. Note the overlap
+// with EXCLUDED_FROM_DOWNLOAD: members_only/private are batch-excluded, but
+// cookies from a subscribed/owning account can still fetch them via the
+// bucket's manual cookie run.
+export const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([
+ "needs_auth",
+ "members_only",
+ "private",
+]);
+
export type AvailabilityHistorySource = "check" | "backfill" | "download";
export type AvailabilityHistoryEntry = {
diff --git a/common/lib/channelConfig.ts b/common/lib/channelConfig.ts
@@ -3,6 +3,7 @@ import {
isDownloadFormatPreset,
type DownloadFormatPreset,
} from "../ytdlp/downloadFormat";
+import { isCookieMode, type CookieMode } from "./cookiePolicy";
export type ChannelHandling = "youtube" | "transcribe";
@@ -96,6 +97,14 @@ export type ChannelConfig = {
// omitted, the global SiteSettings value is used. Set false to allow this
// channel to download currently-live/upcoming videos.
skipLiveDownloads?: boolean;
+ // Per-channel override of the global cookies-from-browser browser spec
+ // (SiteSettings.cookiesFromBrowser). Omitted/empty = inherit the global
+ // value. See common/lib/cookiePolicy.ts.
+ cookiesFromBrowser?: string;
+ // Per-channel override of the global cookie mode
+ // (SiteSettings.cookieMode). Omitted = inherit. See
+ // common/lib/cookiePolicy.ts for the mode semantics.
+ cookieMode?: CookieMode;
// Opt-in audio-integrity checking for sources that intermittently serve
// corrupt audio mid-download (e.g. Odysee "original" format). When
// enabled, the managed downloader periodically validates the in-progress
@@ -233,6 +242,12 @@ export function parseChannelConfig(raw: unknown): ChannelConfig | null {
if (typeof r.skipLiveDownloads === "boolean") {
config.skipLiveDownloads = r.skipLiveDownloads;
}
+ if (typeof r.cookiesFromBrowser === "string" && r.cookiesFromBrowser.trim()) {
+ config.cookiesFromBrowser = r.cookiesFromBrowser.trim();
+ }
+ if (isCookieMode(r.cookieMode)) {
+ config.cookieMode = r.cookieMode;
+ }
if (
typeof r.sleepBetweenDownloadsSeconds === "number" &&
Number.isFinite(r.sleepBetweenDownloadsSeconds) &&
diff --git a/common/lib/channelGroups.ts b/common/lib/channelGroups.ts
@@ -15,16 +15,22 @@ export type ChannelGroup = {
// group header so origin is legible in the channel filter. Absent in
// single-site mode.
accent?: string;
+ // Render this group's channels as loose individual chips instead of a
+ // collapsible group chip; absent = false.
+ inline?: boolean;
};
export const DEFAULT_GROUP_FALLBACK_ID = "default";
// Synthesized fallback used when settings has no groups configured at all,
-// so the UI always has a non-empty grouping to render.
+// so the UI always has a non-empty grouping to render. Inline by default:
+// with no configured grouping the channels render as loose individual chips
+// rather than one big collapsible "All channels" box.
export const FALLBACK_GROUP: ChannelGroup = {
id: DEFAULT_GROUP_FALLBACK_ID,
name: "All channels",
selectedByDefault: true,
+ inline: true,
};
const ID_RE = /^[a-z0-9][a-z0-9-]*$/;
@@ -51,6 +57,9 @@ export function parseChannelGroup(raw: unknown): ChannelGroup | null {
if (typeof r.order === "number" && Number.isFinite(r.order)) {
group.order = Math.floor(r.order);
}
+ // Absent/false stays absent so serialized JSON stays clean and existing
+ // configured groups default off.
+ if (r.inline === true) group.inline = true;
return group;
}
diff --git a/common/lib/cookiePolicy.test.ts b/common/lib/cookiePolicy.test.ts
@@ -0,0 +1,109 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ DEFAULT_COOKIE_MODE,
+ alwaysCookies,
+ authRetryCookies,
+ cookieArgs,
+ isCookieMode,
+ resolveCookiePolicy,
+ type ResolvedCookiePolicy,
+} from "./cookiePolicy";
+
+// Run with: pnpm --filter yt-dlp-transcript-common exec tsx --test common/lib/cookiePolicy.test.ts
+// (or `node_modules/.bin/tsx --test common/lib/cookiePolicy.test.ts` from the repo root)
+
+test("isCookieMode accepts exactly the three modes", () => {
+ assert.equal(isCookieMode("always"), true);
+ assert.equal(isCookieMode("when-required"), true);
+ assert.equal(isCookieMode("defer"), true);
+ assert.equal(isCookieMode(""), false);
+ assert.equal(isCookieMode("never"), false);
+ assert.equal(isCookieMode(undefined), false);
+ assert.equal(isCookieMode(null), false);
+ assert.equal(isCookieMode(42), false);
+});
+
+test("value resolution: channel > global > undefined; empty/whitespace = inherit", () => {
+ assert.equal(
+ resolveCookiePolicy({ cookiesFromBrowser: "firefox" }).cookies,
+ "firefox",
+ );
+ assert.equal(
+ resolveCookiePolicy(
+ { cookiesFromBrowser: "firefox" },
+ { cookiesFromBrowser: "chrome:Default" },
+ ).cookies,
+ "chrome:Default",
+ );
+ // Empty / whitespace channel value inherits the global.
+ assert.equal(
+ resolveCookiePolicy(
+ { cookiesFromBrowser: "firefox" },
+ { cookiesFromBrowser: "" },
+ ).cookies,
+ "firefox",
+ );
+ assert.equal(
+ resolveCookiePolicy(
+ { cookiesFromBrowser: "firefox" },
+ { cookiesFromBrowser: " " },
+ ).cookies,
+ "firefox",
+ );
+ // Nothing configured anywhere.
+ assert.equal(resolveCookiePolicy({}).cookies, undefined);
+ assert.equal(resolveCookiePolicy({ cookiesFromBrowser: " " }).cookies, undefined);
+ // Values are trimmed.
+ assert.equal(
+ resolveCookiePolicy({ cookiesFromBrowser: " firefox " }).cookies,
+ "firefox",
+ );
+});
+
+test("mode resolution: channel > global > default", () => {
+ assert.equal(resolveCookiePolicy({}).mode, DEFAULT_COOKIE_MODE);
+ assert.equal(resolveCookiePolicy({ cookieMode: "always" }).mode, "always");
+ assert.equal(
+ resolveCookiePolicy({ cookieMode: "always" }, { cookieMode: "defer" }).mode,
+ "defer",
+ );
+ // Absent channel mode inherits the global.
+ assert.equal(
+ resolveCookiePolicy({ cookieMode: "defer" }, {}).mode,
+ "defer",
+ );
+ assert.equal(resolveCookiePolicy({ cookieMode: "defer" }, null).mode, "defer");
+ // Junk mode (e.g. hand-edited file that bypassed parsing) falls to default.
+ assert.equal(
+ resolveCookiePolicy({ cookieMode: "sometimes" as never }).mode,
+ DEFAULT_COOKIE_MODE,
+ );
+});
+
+test("accessors encode the mode semantics", () => {
+ const p = (mode: ResolvedCookiePolicy["mode"], cookies?: string) =>
+ ({ cookies, mode }) as ResolvedCookiePolicy;
+
+ // always: cookies on every invocation, and on retries.
+ assert.equal(alwaysCookies(p("always", "firefox")), "firefox");
+ assert.equal(authRetryCookies(p("always", "firefox")), "firefox");
+ // when-required: never prophylactically, yes on auth retries.
+ assert.equal(alwaysCookies(p("when-required", "firefox")), undefined);
+ assert.equal(authRetryCookies(p("when-required", "firefox")), "firefox");
+ // defer: never.
+ assert.equal(alwaysCookies(p("defer", "firefox")), undefined);
+ assert.equal(authRetryCookies(p("defer", "firefox")), undefined);
+
+ // No value configured: every accessor degrades to undefined in every mode.
+ for (const mode of ["always", "when-required", "defer"] as const) {
+ assert.equal(alwaysCookies(p(mode)), undefined);
+ assert.equal(authRetryCookies(p(mode)), undefined);
+ }
+});
+
+test("cookieArgs emits the flag pair or nothing", () => {
+ assert.deepEqual(cookieArgs("firefox"), ["--cookies-from-browser", "firefox"]);
+ assert.deepEqual(cookieArgs(undefined), []);
+ assert.deepEqual(cookieArgs(""), []);
+});
diff --git a/common/lib/cookiePolicy.ts b/common/lib/cookiePolicy.ts
@@ -0,0 +1,86 @@
+// Client-safe cookie-policy resolution. No node-only imports — the editor's
+// settings/channel forms are "use client" files that pull the mode constants
+// in. Mirrors availability.ts in that respect.
+//
+// One browser-cookie spec (yt-dlp `--cookies-from-browser`) and one MODE decide
+// how every yt-dlp spawn uses cookies:
+//
+// "always" — pass cookies on every invocation (enumeration, metadata
+// prefetch, availability probes, downloads).
+// "when-required" — (default; the historical behavior) cookies only to retry
+// an attempt that failed with an auth/age error. Covers the
+// download auth-retry AND the metadata-prefetch retry.
+// "defer" — never use cookies in normal runs. Auth-gated videos are
+// recorded (needs_auth), excluded from subsequent batch
+// runs, and collected into the per-channel "Needs cookies"
+// bucket whose button re-runs them with cookies forced on.
+//
+// The VALUE resolves channel-over-global (non-empty channel spec wins; empty =
+// inherit). The MODE resolves the same way (absent channel mode = inherit).
+// With no value configured, "always"/"when-required" degrade to the no-cookie
+// behavior; "defer" still defers (the bucket run then warns that no cookie
+// value is configured).
+
+export type CookieMode = "always" | "when-required" | "defer";
+
+export const COOKIE_MODE_VALUES: ReadonlyArray<CookieMode> = [
+ "always",
+ "when-required",
+ "defer",
+];
+
+export const DEFAULT_COOKIE_MODE: CookieMode = "when-required";
+
+export function isCookieMode(v: unknown): v is CookieMode {
+ return v === "always" || v === "when-required" || v === "defer";
+}
+
+export type ResolvedCookiePolicy = {
+ // The resolved browser spec, or undefined when neither the channel nor the
+ // global settings carry a non-empty one.
+ cookies: string | undefined;
+ mode: CookieMode;
+};
+
+// The subset of SiteSettings / ChannelConfig this module reads. Structural, so
+// neither settings.ts nor channelConfig.ts needs to be imported here (both
+// import CookieMode from this file).
+export type CookiePolicyInputs = {
+ cookiesFromBrowser?: string;
+ cookieMode?: CookieMode;
+};
+
+export function resolveCookiePolicy(
+ settings: CookiePolicyInputs,
+ channelConfig?: CookiePolicyInputs | null,
+): ResolvedCookiePolicy {
+ const channelValue = channelConfig?.cookiesFromBrowser?.trim() ?? "";
+ const globalValue = settings.cookiesFromBrowser?.trim() ?? "";
+ const cookies = channelValue || globalValue || undefined;
+ const mode = isCookieMode(channelConfig?.cookieMode)
+ ? channelConfig!.cookieMode!
+ : isCookieMode(settings.cookieMode)
+ ? settings.cookieMode!
+ : DEFAULT_COOKIE_MODE;
+ return { cookies, mode };
+}
+
+// Cookies for an ordinary (non-retry) invocation: only "always" mode passes
+// them prophylactically. Undefined in every other mode — including "defer",
+// whose whole point is that normal runs stay cookie-free.
+export function alwaysCookies(p: ResolvedCookiePolicy): string | undefined {
+ return p.mode === "always" ? p.cookies : undefined;
+}
+
+// Cookies for retrying an attempt that failed with an auth/age error:
+// available in "always" and "when-required", never in "defer" (the failure is
+// recorded instead, feeding the Needs-cookies bucket).
+export function authRetryCookies(p: ResolvedCookiePolicy): string | undefined {
+ return p.mode === "defer" ? undefined : p.cookies;
+}
+
+// The argv fragment for a resolved cookie spec. Empty when there is none, so
+// spawn sites can unconditionally spread it.
+export function cookieArgs(cookies: string | undefined): string[] {
+ return cookies ? ["--cookies-from-browser", cookies] : [];
+}
diff --git a/common/lib/downloadOutcome.ts b/common/lib/downloadOutcome.ts
@@ -48,7 +48,11 @@ export type DownloadAttemptKind =
| "audio-checked-primary"
// The metadata-only pass that runs before the real download so app-level
// filters can decide, and so the download can reuse it via --load-info-json.
- | "metadata-prefetch";
+ | "metadata-prefetch"
+ // A cookie re-run of a metadata prefetch that failed with an auth/age error
+ // (cookie mode "always"/"when-required" with a cookie value configured).
+ // Recorded with n: 0 alongside the failed prefetch it retries.
+ | "metadata-prefetch-auth-retry";
export type AudioCheckProbeVerdict = "clean" | "partial" | "malformed";
diff --git a/common/lib/momentUrl.ts b/common/lib/momentUrl.ts
@@ -0,0 +1,180 @@
+// Build a timestamped deep link for a cited moment in a video.
+//
+// Two shapes, in preference order:
+//
+// 1. **Archilyzer viewer link** — `${siteOrigin}/?v=<slug>&t=<seconds>`. This
+// is the exact scheme `common/components/urlState.ts` reads: `v` is the
+// video slug (`<channelSlug>/<id>`, the value the modal compares against
+// `detail.slug`) and `t` is whole seconds. Opening it lands in the archive's
+// transcript modal at the cited moment. Used whenever we know the public
+// origin of the viewer that owns the video (a deployed site, or a hub
+// member's url).
+//
+// 2. **Platform link** — the video's own `webpageUrl` plus a per-platform time
+// parameter, mirroring the seek patterns the in-app players use
+// (`common/components/{Odysee,Rumble,Twitch}Player.tsx`). Used as a fallback
+// when there is no viewer origin (e.g. an MCP pointed at a local build).
+//
+// Pure and dependency-light so it can be reused by the MCP server, the browser
+// report-citation UI, and build tools alike.
+
+import { detectPlatform, type Platform } from "./platform";
+
+export type MomentUrlInput = {
+ // Public origin of the archilyzer viewer that owns this video (a RemoteSource
+ // base, or a hub member's url). When present and non-empty, we build a viewer
+ // deep link. Null/undefined ⇒ fall back to a platform link.
+ siteOrigin?: string | null;
+ // The video's slug as the viewer's `?v=` param expects it (`channelSlug/id`).
+ slug?: string | null;
+ // The cited moment, in seconds.
+ seconds: number;
+ // Fallback building blocks when there is no viewer origin.
+ webpageUrl?: string | null;
+ platform?: Platform | null;
+};
+
+// Twitch watch/VOD URLs take an `XhYmZs` time token, not raw seconds.
+function twitchTime(totalSeconds: number): string {
+ const s = Math.max(0, Math.floor(totalSeconds));
+ const h = Math.floor(s / 3600);
+ const m = Math.floor((s % 3600) / 60);
+ return `${h}h${m}m${s % 60}s`;
+}
+
+// Best-effort deep link into a platform's own watch page at `seconds`. Mirrors
+// the per-platform time params the in-app players build. Platforms whose watch
+// page has no reliable start param (Rumble, Kick) get the bare `webpageUrl`.
+// Returns null only when there is no `webpageUrl` to work from.
+export function platformMomentUrl(
+ webpageUrl: string | null | undefined,
+ platform: Platform | null | undefined,
+ seconds: number,
+): string | null {
+ if (!webpageUrl) return null;
+ const secs = Math.max(0, Math.floor(seconds || 0));
+ if (secs <= 0) return webpageUrl;
+ const plat = platform ?? detectPlatform(webpageUrl);
+ let u: URL;
+ try {
+ u = new URL(webpageUrl);
+ } catch {
+ return webpageUrl;
+ }
+ switch (plat) {
+ case "youtube":
+ u.searchParams.set("t", `${secs}s`);
+ return u.toString();
+ case "odysee":
+ u.searchParams.set("t", String(secs));
+ return u.toString();
+ case "twitch":
+ u.searchParams.set("t", twitchTime(secs));
+ return u.toString();
+ // Rumble / Kick / unknown: the watch page has no dependable start param —
+ // return the plain webpage URL rather than an invalid seek.
+ default:
+ return webpageUrl;
+ }
+}
+
+// The archilyzer viewer deep link, or null when `siteOrigin`/`slug` are missing
+// or the origin can't be parsed.
+export function viewerMomentUrl(
+ siteOrigin: string | null | undefined,
+ slug: string | null | undefined,
+ seconds: number,
+): string | null {
+ if (!siteOrigin || !slug) return null;
+ const secs = Math.max(0, Math.floor(seconds || 0));
+ let u: URL;
+ try {
+ u = new URL(siteOrigin);
+ } catch {
+ return null;
+ }
+ u.pathname = "/";
+ u.search = "";
+ u.hash = "";
+ u.searchParams.set("v", slug);
+ if (secs > 0) u.searchParams.set("t", String(secs));
+ return u.toString();
+}
+
+// The preferred moment link: viewer deep link when we have an origin+slug,
+// else the platform fallback. Null when neither can be built.
+export function momentUrl(input: MomentUrlInput): string | null {
+ const viewer = viewerMomentUrl(input.siteOrigin, input.slug, input.seconds);
+ if (viewer) return viewer;
+ return platformMomentUrl(input.webpageUrl, input.platform, input.seconds);
+}
+
+// ─── Base (appendable) forms ───
+//
+// A "moment base" is a URL that ends in `t=`, so appending integer seconds
+// yields a valid moment link (`<base>156` ≡ the momentUrl for 156s). Used by
+// the MCP's compact `link_style:"base"` output: one base per video instead of
+// a full URL per line, expanded back to full links in the final report.
+
+// The viewer deep-link base: same normalization as viewerMomentUrl, with `t`
+// set last and empty so the caller can append seconds. Null when the origin or
+// slug is missing/unparseable.
+export function viewerMomentBaseUrl(
+ siteOrigin: string | null | undefined,
+ slug: string | null | undefined,
+): string | null {
+ if (!siteOrigin || !slug) return null;
+ let u: URL;
+ try {
+ u = new URL(siteOrigin);
+ } catch {
+ return null;
+ }
+ u.pathname = "/";
+ u.search = "";
+ u.hash = "";
+ u.searchParams.set("v", slug);
+ u.searchParams.set("t", "");
+ return u.toString();
+}
+
+// The platform base. Unlike platformMomentUrl (which falls back to the bare
+// webpage URL), a base MUST be appendable — so only platforms whose time param
+// takes raw seconds qualify. Twitch is excluded (its `t` takes an `XhYmZs`
+// token, so appending an integer would be an invalid seek); Rumble/Kick/unknown
+// have no dependable start param at all. Null in every non-appendable case.
+export function platformMomentBaseUrl(
+ webpageUrl: string | null | undefined,
+ platform: Platform | null | undefined,
+): string | null {
+ if (!webpageUrl) return null;
+ const plat = platform ?? detectPlatform(webpageUrl);
+ let u: URL;
+ try {
+ u = new URL(webpageUrl);
+ } catch {
+ return null;
+ }
+ switch (plat) {
+ case "youtube": // accepts raw seconds in `t` (the `s` suffix is optional)
+ case "odysee": {
+ // delete-then-set so a pre-existing `t` is re-appended LAST — the base
+ // must end in `t=` for the append rule to hold.
+ u.searchParams.delete("t");
+ u.searchParams.set("t", "");
+ return u.toString();
+ }
+ default:
+ return null;
+ }
+}
+
+// The preferred moment base: viewer first, platform fallback — mirroring
+// momentUrl's preference order. Null when neither can be built.
+export function momentBaseUrl(
+ input: Omit<MomentUrlInput, "seconds">,
+): string | null {
+ const viewer = viewerMomentBaseUrl(input.siteOrigin, input.slug);
+ if (viewer) return viewer;
+ return platformMomentBaseUrl(input.webpageUrl, input.platform);
+}
diff --git a/common/lib/settings.ts b/common/lib/settings.ts
@@ -24,6 +24,11 @@ import {
defaultAutoQueue,
sanitizeAutoQueue,
} from "../jobs/autoQueuePolicy";
+import {
+ DEFAULT_COOKIE_MODE,
+ isCookieMode,
+ type CookieMode,
+} from "./cookiePolicy";
export type { Worker } from "./workers";
export type { AutoQueueSettings } from "../jobs/autoQueuePolicy";
@@ -63,9 +68,18 @@ export type SiteSettings = {
// app (see defaultWorkersFromApps). See common/lib/workers.ts.
workers: Worker[];
// Browser spec (e.g. "firefox", "chrome:Default") passed to
- // `yt-dlp --cookies-from-browser` ONLY on the auth-retry attempt of the
- // managed per-video download flow. Empty string = disabled.
+ // `yt-dlp --cookies-from-browser`. WHEN it is passed is governed by
+ // `cookieMode` below. Empty string = no cookies configured. Per-channel
+ // override available (ChannelConfig.cookiesFromBrowser).
cookiesFromBrowser: string;
+ // How yt-dlp invocations use the configured cookies (see
+ // common/lib/cookiePolicy.ts): "always" passes them on every invocation,
+ // "when-required" (default; the historical behavior) only to retry an
+ // auth/age failure, "defer" never in normal runs — auth-gated videos are
+ // excluded from batches and collected into the per-channel "Needs cookies"
+ // bucket for a manual cookie run. Per-channel override available
+ // (ChannelConfig.cookieMode).
+ cookieMode: CookieMode;
// Pause (seconds) inserted between per-video yt-dlp invocations in
// managed batch downloads. yt-dlp's own `-t sleep` only paces requests
// within one invocation, so without this the managed loop hammers the
@@ -455,6 +469,7 @@ function defaults(): SiteSettings {
transcriptionApps: {},
workers: [],
cookiesFromBrowser: "",
+ cookieMode: DEFAULT_COOKIE_MODE,
sleepBetweenDownloadsSeconds: SLEEP_BETWEEN_DOWNLOADS_DEFAULT_SECONDS,
downloadFormat: "auto",
minFreeDiskGB: MIN_FREE_DISK_GB_DEFAULT,
@@ -635,6 +650,11 @@ export function getSettings(): SiteSettings {
if (typeof merged.cookiesFromBrowser !== "string") {
merged.cookiesFromBrowser = "";
}
+ // A settings.json predating cookieMode (or carrying junk) gets the default,
+ // which preserves the historical retry-only behavior.
+ if (!isCookieMode(merged.cookieMode)) {
+ merged.cookieMode = DEFAULT_COOKIE_MODE;
+ }
merged.sleepBetweenDownloadsSeconds = clampSleepBetweenDownloadsSeconds(
merged.sleepBetweenDownloadsSeconds,
);
@@ -826,6 +846,9 @@ export async function writeSettings(next: SiteSettings): Promise<void> {
typeof next.cookiesFromBrowser === "string"
? next.cookiesFromBrowser.trim()
: "",
+ cookieMode: isCookieMode(next.cookieMode)
+ ? next.cookieMode
+ : DEFAULT_COOKIE_MODE,
sleepBetweenDownloadsSeconds: clampSleepBetweenDownloadsSeconds(
next.sleepBetweenDownloadsSeconds,
),
diff --git a/common/lib/transcriptToMarkdown.test.ts b/common/lib/transcriptToMarkdown.test.ts
@@ -48,3 +48,33 @@ test("maxCues truncates and notes it", () => {
assert.doesNotMatch(md, /General Kenobi/);
assert.match(md, /truncated: showing 1 of 2 cues/);
});
+
+test("stampForCue renders the compact form and floors fractional seconds", () => {
+ const md = transcriptToMarkdown(base, {
+ stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ });
+ assert.match(md, /^\[0:00\|0\] Hello there$/m);
+ // 3661.25s floors to 3661 — the same integer momentUrl would use, so the
+ // compact stamp and the inline link cite the identical second.
+ assert.match(md, /^\[1:01:01\|3661\] General Kenobi$/m);
+});
+
+test("stampForCue takes precedence over linkForCue (no inline links)", () => {
+ const md = transcriptToMarkdown(base, {
+ stampForCue: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ linkForCue: (seconds) => `https://example.test/abc123?t=${seconds}s`,
+ });
+ assert.ok(!md.includes("](https://example.test/"), "no per-line links");
+ assert.match(md, /^\[0:00\|0\] Hello there$/m);
+});
+
+test("extraMeta lines render as `- <line>` in the metadata header", () => {
+ const md = transcriptToMarkdown(base, {
+ extraMeta: ["moment_base: https://example.test/abc123?t="],
+ });
+ assert.match(md, /^- moment_base: https:\/\/example\.test\/abc123\?t=$/m);
+ assert.ok(
+ md.indexOf("moment_base:") < md.indexOf("## Description"),
+ "extra meta sits in the header, before the body",
+ );
+});
diff --git a/common/lib/transcriptToMarkdown.ts b/common/lib/transcriptToMarkdown.ts
@@ -31,6 +31,19 @@ export type TranscriptMarkdownOptions = {
// Cap the number of cue lines emitted (for fitting a context window). When
// truncated, a marker line is appended. Default: no cap.
maxCues?: number;
+ // Optional builder turning a cue's start seconds into a deep link. When it
+ // returns a URL, the timestamp is rendered as a Markdown link
+ // (`[m:ss](url)`) so a reader can jump to the exact moment. Ignored when
+ // `timestamps` is false or when it returns null. See `momentUrl`.
+ linkForCue?: (seconds: number) => string | null;
+ // Optional formatter for the bracketed label's CONTENTS (e.g. "2:36|156" →
+ // rendered as `[2:36|156] text`). Receives the pre-formatted clock and the
+ // cue's raw start seconds. Takes precedence over `linkForCue`. Ignored when
+ // `timestamps` is false.
+ stampForCue?: (clock: string, seconds: number) => string;
+ // Extra metadata lines rendered as `- <line>` after the tags line (e.g.
+ // "moment_base: <url>").
+ extraMeta?: string[];
};
// [h:mm:ss] / [m:ss] label for a cue start. formatDuration returns "" for 0, so
@@ -53,6 +66,9 @@ export function transcriptToMarkdown(
includeDescription = true,
includeTags = false,
maxCues,
+ linkForCue,
+ stampForCue,
+ extraMeta,
} = options;
const lines: string[] = [];
@@ -70,6 +86,7 @@ export function transcriptToMarkdown(
if (includeTags && input.tags && input.tags.length > 0) {
meta.push(`- Tags: ${input.tags.join(", ")}`);
}
+ for (const line of extraMeta ?? []) meta.push(`- ${line}`);
lines.push(...meta);
if (includeDescription && input.description && input.description.trim()) {
@@ -97,7 +114,17 @@ export function transcriptToMarkdown(
const cue = cues[i];
const text = cue.text.trim();
if (!text) continue;
- lines.push(timestamps ? `[${stamp(cue.start)}] ${text}` : text);
+ if (!timestamps) {
+ lines.push(text);
+ continue;
+ }
+ if (stampForCue) {
+ lines.push(`[${stampForCue(stamp(cue.start), cue.start)}] ${text}`);
+ continue;
+ }
+ const url = linkForCue ? linkForCue(cue.start) : null;
+ const label = url ? `[${stamp(cue.start)}](${url})` : stamp(cue.start);
+ lines.push(`[${label}] ${text}`);
}
if (limit < cues.length) {
lines.push("");
diff --git a/common/ytdlp/downloadOneManaged.ts b/common/ytdlp/downloadOneManaged.ts
@@ -3,11 +3,17 @@ import { appendFile, mkdir, readdir, readFile, rm } from "node:fs/promises";
import { createWriteStream, type WriteStream } from "node:fs";
import { execa } from "execa";
import {
+ AUTH_RETRY_CLASSES,
classifyDownloadFailure,
parseUnavailableFromStderr,
- type Availability,
} from "../lib/availability";
import {
+ DEFAULT_COOKIE_MODE,
+ alwaysCookies,
+ authRetryCookies,
+ type ResolvedCookiePolicy,
+} from "../lib/cookiePolicy";
+import {
type AudioFormat,
type ChannelConfig,
} from "../lib/channelConfig";
@@ -82,9 +88,12 @@ export type ManagedDownloadOpts = {
videoUrl: string;
onLog: (s: string) => void;
signal: AbortSignal;
- // Resolved global setting; only used when the primary attempt fails with an
- // auth/age error AND the channel doesn't already have its own cookies set.
- globalCookiesFromBrowser?: string;
+ // Resolved cookie policy (value + mode), already collapsed channel-over-
+ // global by the caller (resolveCookiePolicy). Governs which attempts pass
+ // --cookies-from-browser: "always" on every invocation, "when-required"
+ // (the default when omitted) only on auth-retry passes, "defer" never —
+ // failures are recorded for the Needs-cookies bucket instead.
+ cookiePolicy?: ResolvedCookiePolicy;
// When false, suppress appending to the channel archive file. Mirrors
// `ignoreArchive` from the playlist-level callers.
appendArchive?: boolean;
@@ -349,12 +358,6 @@ function attemptSucceeded(exitCode: number | null): boolean {
return exitCode === 0 || exitCode === 101;
}
-const AUTH_RETRY_CLASSES: ReadonlySet<Availability> = new Set<Availability>([
- "needs_auth",
- "members_only",
- "private",
-]);
-
async function hasAnyTranscriptOnDisk(videoDir: string): Promise<boolean> {
const entries = await readdir(videoDir).catch(() => [] as string[]);
return entries.some((e) => {
@@ -467,6 +470,17 @@ async function runManagedDownload(
const attempts: DownloadAttempt[] = [];
let status: DownloadOutcomeStatus = "failed";
let fellBackToTranscribe = false;
+ // Cookie policy for every yt-dlp spawn in this download. An omitted policy
+ // behaves like the historical default: no prophylactic cookies, retry-only
+ // (and with no value configured the retries degrade to no-ops).
+ const cookiePolicy: ResolvedCookiePolicy = opts.cookiePolicy ?? {
+ cookies: undefined,
+ mode: DEFAULT_COOKIE_MODE,
+ };
+ // Set when a failed metadata prefetch was recovered by its cookie retry: the
+ // real download then gets cookies immediately (skipping a doomed cookie-less
+ // attempt) and a success is reported as "ok-with-cookies".
+ let prefetchNeededCookies = false;
// Set when the download-time duration guard trips: the measured shortfall,
// recorded on the outcome so the UI can explain it without re-probing.
let shortAudioInfo: NonNullable<DownloadOutcomeRecord["shortAudio"]> | undefined;
@@ -498,7 +512,10 @@ async function runManagedDownload(
opts.channelConfig.platform ?? detectPlatform(opts.videoUrl);
if (canonicalId && !opts.signal.aborted) {
const videoDir = path.join(channelDir, "data", canonicalId);
- const prefetchArgs = [
+ // In "always" mode the prefetch (like every other invocation) carries the
+ // configured cookies up front.
+ const prefetchCookies = alwaysCookies(cookiePolicy);
+ const buildPrefetchArgs = (cookies: string | undefined) => [
"--ignore-config",
"--restrict-filenames",
...outputArgsForUrl(opts.videoUrl),
@@ -506,11 +523,15 @@ async function runManagedDownload(
"--skip-download",
"--no-write-subs",
"--no-write-auto-subs",
- ...channelConfigArgs(opts.channelConfig),
+ ...channelConfigArgs(opts.channelConfig, cookies),
"--",
opts.videoUrl,
];
- const prefetchRes = await runOneYtdlp(opts, channelDir, prefetchArgs);
+ const prefetchRes = await runOneYtdlp(
+ opts,
+ channelDir,
+ buildPrefetchArgs(prefetchCookies),
+ );
lastFullTail = prefetchRes.stderrTail;
const prefetchAvail = attemptSucceeded(prefetchRes.exitCode)
? undefined
@@ -519,7 +540,7 @@ async function runManagedDownload(
n: 0,
kind: "metadata-prefetch",
handling: opts.channelConfig.handling,
- usedCookies: false,
+ usedCookies: Boolean(prefetchCookies),
ytdlpExitCode: prefetchRes.exitCode,
availabilityClass: prefetchAvail,
error: attemptSucceeded(prefetchRes.exitCode)
@@ -527,6 +548,49 @@ async function runManagedDownload(
: trimError(prefetchRes.stderrTail),
});
+ // Prefetch auth retry: a prefetch that failed with an auth/age error is
+ // re-run once with cookies (when the mode allows it and the failed pass
+ // didn't already use them). A recovery here lets the real download start
+ // with cookies immediately instead of burning a doomed cookie-less
+ // attempt first. Defer mode lands here with no retry cookies, so the
+ // failure stands and feeds the Needs-cookies bucket.
+ const prefetchRetryCookies = authRetryCookies(cookiePolicy);
+ if (
+ !attemptSucceeded(prefetchRes.exitCode) &&
+ prefetchAvail !== undefined &&
+ AUTH_RETRY_CLASSES.has(prefetchAvail) &&
+ prefetchRetryCookies !== undefined &&
+ !prefetchCookies &&
+ !opts.signal.aborted
+ ) {
+ opts.onLog(
+ `Metadata prefetch auth-required (${prefetchAvail}); retrying with --cookies-from-browser ${prefetchRetryCookies}\n`,
+ );
+ const retryRes = await runOneYtdlp(
+ opts,
+ channelDir,
+ buildPrefetchArgs(prefetchRetryCookies),
+ );
+ lastFullTail = retryRes.stderrTail;
+ const retryAvail = attemptSucceeded(retryRes.exitCode)
+ ? undefined
+ : parseUnavailableFromStderr(retryRes.stderrTail);
+ attempts.push({
+ n: 0,
+ kind: "metadata-prefetch-auth-retry",
+ handling: opts.channelConfig.handling,
+ usedCookies: true,
+ ytdlpExitCode: retryRes.exitCode,
+ availabilityClass: retryAvail,
+ error: attemptSucceeded(retryRes.exitCode)
+ ? undefined
+ : trimError(retryRes.stderrTail),
+ });
+ if (attemptSucceeded(retryRes.exitCode)) {
+ prefetchNeededCookies = true;
+ }
+ }
+
const metaPath = path.join(videoDir, "metadata.info.json");
const metadata = await loadRawMetadata(metaPath);
// Only wire --load-info-json into the real attempts when we actually have
@@ -629,6 +693,13 @@ async function runManagedDownload(
}
}
+ // Cookies for the real download attempts: always-mode passes them on every
+ // invocation; otherwise a cookie-recovered prefetch means this video needs
+ // them, so don't burn a doomed cookie-less attempt first.
+ const primaryCookies =
+ alwaysCookies(cookiePolicy) ??
+ (prefetchNeededCookies ? cookiePolicy.cookies : undefined);
+
let primaryRes: AttemptOutcome;
let audioCheckStats: AudioCheckAttemptStats | undefined;
let audioCheckCorruptSource = false;
@@ -652,7 +723,7 @@ async function runManagedDownload(
),
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
- ...channelConfigArgs(opts.channelConfig),
+ ...channelConfigArgs(opts.channelConfig, primaryCookies),
"--",
opts.videoUrl,
];
@@ -696,7 +767,7 @@ async function runManagedDownload(
...mediaArgs,
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
- ...channelConfigArgs(opts.channelConfig),
+ ...channelConfigArgs(opts.channelConfig, primaryCookies),
...sourceArgs(opts.videoUrl, infoJsonPath),
];
primaryRes = await runOneYtdlp(opts, channelDir, primaryArgs);
@@ -710,7 +781,7 @@ async function runManagedDownload(
n: 1,
kind: audioCheckEnabled ? "audio-checked-primary" : "primary",
handling: opts.channelConfig.handling,
- usedCookies: false,
+ usedCookies: Boolean(primaryCookies),
ytdlpExitCode: primaryRes.exitCode,
availabilityClass: primaryAvail,
error: attemptSucceeded(primaryRes.exitCode)
@@ -725,7 +796,14 @@ async function runManagedDownload(
!audioCheckCorruptFullSource &&
attemptSucceeded(primaryRes.exitCode);
if (lastSucceeded) {
- status = audioCheckEnabled ? "ok-audio-checked" : "ok";
+ // "ok-with-cookies" keeps meaning "cookies were NEEDED" (the prefetch only
+ // succeeded with them) — always-mode prophylactic cookies on a clean
+ // success stay a plain "ok".
+ status = audioCheckEnabled
+ ? "ok-audio-checked"
+ : prefetchNeededCookies
+ ? "ok-with-cookies"
+ : "ok";
} else if (audioCheckCorruptSource) {
status = "failed-corrupt-source";
} else if (audioCheckCorruptFullSource) {
@@ -735,15 +813,20 @@ async function runManagedDownload(
}
// ---------- Attempt 2: auth retry ----------
+ // Off in defer mode (authRetryCookies -> undefined), so the failure is
+ // recorded as-is and feeds the Needs-cookies bucket. Also skipped when the
+ // failed attempt already used cookies — retrying identically is futile.
+ const downloadRetryCookies = authRetryCookies(cookiePolicy);
const shouldAuthRetry =
!lastSucceeded &&
primaryAvail !== undefined &&
AUTH_RETRY_CLASSES.has(primaryAvail) &&
- Boolean(opts.globalCookiesFromBrowser);
+ downloadRetryCookies !== undefined &&
+ !primaryCookies;
if (shouldAuthRetry && !opts.signal.aborted) {
opts.onLog(
- `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${opts.globalCookiesFromBrowser}\n`,
+ `Auth-required (${primaryAvail}); retrying with --cookies-from-browser ${downloadRetryCookies}\n`,
);
const retryArgs = [
"--ignore-config",
@@ -752,7 +835,7 @@ async function runManagedDownload(
...mediaArgs,
"--print",
`after_video:${ARCHIVE_MARKER} %(extractor)s %(id)s`,
- ...channelConfigArgs(opts.channelConfig, opts.globalCookiesFromBrowser),
+ ...channelConfigArgs(opts.channelConfig, downloadRetryCookies),
...sourceArgs(opts.videoUrl, infoJsonPath),
];
const retryRes = await runOneYtdlp(opts, channelDir, retryArgs);
@@ -862,11 +945,12 @@ async function runManagedDownload(
...opts.channelConfig,
handling: "transcribe",
};
- // Use cookies if the auth retry succeeded with them; otherwise channel-level.
+ // Cookies for the fallback: always-mode passes them like every other
+ // invocation; otherwise only when this video demonstrably needed them
+ // (an attempt succeeded with cookies -> "ok-with-cookies").
const fallbackCookieOverride =
- status === "ok-with-cookies"
- ? opts.globalCookiesFromBrowser
- : undefined;
+ alwaysCookies(cookiePolicy) ??
+ (status === "ok-with-cookies" ? cookiePolicy.cookies : undefined);
// Feed yt-dlp the metadata it already wrote during the primary
// attempt instead of re-querying the extractor — saves a network
// round-trip per video, which adds up across batch runs and helps
diff --git a/common/ytdlp/runYtdlp.ts b/common/ytdlp/runYtdlp.ts
@@ -20,6 +20,12 @@ import {
classifyDownloadFailure,
type Availability,
} from "../lib/availability";
+import {
+ alwaysCookies,
+ cookieArgs,
+ resolveCookiePolicy,
+ type ResolvedCookiePolicy,
+} from "../lib/cookiePolicy";
import { resolveEffectiveAvailability } from "../lib/availability-server";
import { backfillAvailabilityFromMetadata } from "../controller/backfillAvailability";
import { resolveShardItems } from "../controller/shard";
@@ -96,6 +102,11 @@ export type RunYtdlpOpts = {
// saved playlist by extracted ID). IDs not present in the playlist are
// reported and skipped.
bucketIds?: ReadonlyArray<string>;
+ // retry-bucket only (the "Needs cookies" bucket button): force cookie mode
+ // "always" for this run, so every invocation carries the configured cookie
+ // value and the defer-mode batch exclusion is bypassed. Warns (and behaves
+ // like a plain run) when no cookie value is configured.
+ forceCookies?: boolean;
// retry-bucket only: override channelConfig.handling for this run without
// mutating the channel config on disk. Useful for retrying old "youtube"
// videos as "transcribe".
@@ -170,12 +181,46 @@ function abortableSleep(ms: number, signal: AbortSignal): Promise<void> {
});
}
-function configArgs(config: ChannelConfig): string[] {
+function configArgs(config: ChannelConfig, cookies?: string): string[] {
const args: string[] = [];
+ args.push(...cookieArgs(cookies));
if (config.ytdlpExtraArgs?.length) args.push(...config.ytdlpExtraArgs);
return args;
}
+// The run-level cookie policy: settings + channel overrides, with the
+// retry-bucket forceCookies flag overriding the mode to "always" (that's the
+// "Download with cookies" button). Logs a warning when cookies are forced but
+// no value is configured anywhere — the run then proceeds cookie-less.
+function resolveRunCookiePolicy(
+ opts: RunYtdlpOpts,
+ channelConfig: ChannelConfig,
+): ResolvedCookiePolicy {
+ const policy = resolveCookiePolicy(getSettings(), channelConfig);
+ if (!opts.forceCookies) return policy;
+ if (!policy.cookies) {
+ opts.onLog(
+ "Force-cookies run requested, but no cookies-from-browser value is configured (global or channel). Proceeding without cookies.\n",
+ );
+ }
+ return { ...policy, mode: "always" };
+}
+
+// Defer-mode batch exclusion: a video whose effective availability is
+// needs_auth is skipped by batch runs when the resolved cookie mode is
+// "defer" (it lands in the snapshot's Needs-cookies bucket instead of being
+// re-attempted on every sync/download-missing). members_only/private are
+// already covered by EXCLUDED_FROM_DOWNLOAD. forceCookies runs resolve to
+// mode "always" and therefore bypass this.
+export async function isDeferredAuthExcluded(
+ videoDir: string,
+ policy: ResolvedCookiePolicy,
+): Promise<boolean> {
+ if (policy.mode !== "defer") return false;
+ const cls = await resolveEffectiveAvailability(videoDir);
+ return cls === "needs_auth";
+}
+
const OUTPUT_ARGS: string[] = [
"-o",
"data/%(id)s/audio.%(ext)s",
@@ -247,7 +292,15 @@ async function enumeratePlaylistUrls(
if (range) {
args.push("--lazy-playlist", "-I", `${range.start}:${range.end}`);
}
- args.push(...configArgs(opts.channelConfig), opts.channelConfig.url!);
+ // Cookie mode "always" covers enumeration too (a members-only or otherwise
+ // gated channel may not even list without cookies).
+ args.push(
+ ...configArgs(
+ opts.channelConfig,
+ alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)),
+ ),
+ opts.channelConfig.url!,
+ );
opts.onLog(`$ ${opts.paths.ytdlpBin} ${args.join(" ")}\n`);
const child = execa(opts.paths.ytdlpBin, args, {
@@ -436,8 +489,14 @@ async function downloadPlaylistManaged(
// Exclude videos whose effective availability marks them as permanently
// unavailable (members_only, deleted, private). Applies to both prefilter
// modes: failed download attempts don't write to the archive, so the
- // archive-prefilter path would also keep retrying them.
+ // archive-prefilter path would also keep retrying them. Under cookie mode
+ // "defer", needs_auth videos are excluded too — they wait in the snapshot's
+ // Needs-cookies bucket for a manual cookie run instead of being re-attempted
+ // (and re-failing) on every batch. A forceCookies run resolves to mode
+ // "always" and so bypasses the defer exclusion.
+ const runCookiePolicy = resolveRunCookiePolicy(opts, effectiveChannelConfig);
const excludedCounts = { members_only: 0, deleted: 0, private: 0 };
+ let deferredAuthCount = 0;
const filteredTofetch: string[] = [];
for (const url of tofetch) {
const dirId = extractVideoId(url);
@@ -452,6 +511,10 @@ async function downloadPlaylistManaged(
excludedCounts[cls as keyof typeof excludedCounts]++;
continue;
}
+ if (runCookiePolicy.mode === "defer" && cls === "needs_auth") {
+ deferredAuthCount++;
+ continue;
+ }
}
filteredTofetch.push(url);
}
@@ -464,6 +527,11 @@ async function downloadPlaylistManaged(
`Excluded ${totalExcluded} from this run (members_only=${excludedCounts.members_only}, deleted=${excludedCounts.deleted}, private=${excludedCounts.private}). Clear via Diagnostics > Recheck if a video became public again.\n`,
);
}
+ if (deferredAuthCount > 0) {
+ opts.onLog(
+ `Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`,
+ );
+ }
tofetch.length = 0;
tofetch.push(...filteredTofetch);
@@ -539,7 +607,12 @@ async function downloadPlaylistManaged(
}
const { okCount, failedCount, skippedCount, processedCount, firstFailure } =
- await runManagedDownloads(opts, items, effectiveChannelConfig);
+ await runManagedDownloads(
+ opts,
+ items,
+ effectiveChannelConfig,
+ runCookiePolicy,
+ );
if (firstFailure && !opts.signal.aborted) {
throw firstFailure;
}
@@ -575,9 +648,15 @@ async function runManagedDownloads(
opts: RunYtdlpOpts,
urls: ReadonlyArray<string>,
effectiveChannelConfig: ChannelConfig,
+ // Cookie policy for every managed download in this run. Callers that
+ // already resolved it (for their own prefilter) pass it through so the
+ // forceCookies-without-value warning isn't logged twice.
+ cookiePolicyOverride?: ResolvedCookiePolicy,
): Promise<ManagedRunResult> {
const settings = getSettings();
- const globalCookies = settings.cookiesFromBrowser;
+ const cookiePolicy =
+ cookiePolicyOverride ??
+ resolveRunCookiePolicy(opts, effectiveChannelConfig);
const inlineTranscribeOnFallback = settings.inlineTranscribeOnFallback;
const globalSkipLiveDownloads = settings.skipLiveDownloads;
// Resolve the download format once per run (override > channel > global),
@@ -644,7 +723,7 @@ async function runManagedDownloads(
videoUrl: url,
onLog: task ? task.onLog : opts.onLog,
signal: opts.signal,
- globalCookiesFromBrowser: globalCookies || undefined,
+ cookiePolicy,
appendArchive: !opts.ignoreArchive,
inlineTranscribeOnFallback,
globalSkipLiveDownloads,
@@ -782,7 +861,10 @@ async function downloadSubsForUrl(
"--skip-download",
"--no-write-info-json",
"--no-overwrites",
- ...configArgs(opts.channelConfig),
+ ...configArgs(
+ opts.channelConfig,
+ alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)),
+ ),
"--",
url,
];
@@ -951,7 +1033,10 @@ async function downloadOneAudio(opts: RunYtdlpOpts): Promise<void> {
"--audio-format",
fmt,
...(opts.channelConfig.keepSourceVideo ? ["-k"] : []),
- ...configArgs(opts.channelConfig),
+ ...configArgs(
+ opts.channelConfig,
+ alwaysCookies(resolveRunCookiePolicy(opts, opts.channelConfig)),
+ ),
...(opts.extraYtdlpArgs ?? []),
"--",
opts.singleVideoUrl,
@@ -1013,6 +1098,9 @@ async function sync(opts: RunYtdlpOpts): Promise<void> {
let totalFailed = 0;
let totalSkipped = 0;
let firstFailure: Error | null = null;
+ // Resolved once for the whole sync: drives both the defer-mode needs_auth
+ // exclusion below and the managed downloads themselves.
+ const runCookiePolicy = resolveRunCookiePolicy(opts, opts.channelConfig);
for (
let page = 0;
@@ -1026,20 +1114,44 @@ async function sync(opts: RunYtdlpOpts): Promise<void> {
const newUrls: string[] = [];
let archivedHits = 0;
+ let deferredAuthCount = 0;
for (const url of pageUrls) {
const archiveId = await archiveIdForUrl(url, dataDir);
if (archiveId && archive.ids.has(archiveId)) {
archivedHits++;
continue;
}
+ // Cookie mode "defer": don't re-attempt known auth-gated videos on
+ // every sync — they wait in the Needs-cookies bucket instead.
+ const dirId = extractVideoId(url);
+ if (
+ dirId &&
+ (await isDeferredAuthExcluded(
+ path.join(dataDir, dirId),
+ runCookiePolicy,
+ ))
+ ) {
+ deferredAuthCount++;
+ continue;
+ }
newUrls.push(url);
}
opts.onLog(
`Sync page ${page + 1}: ${pageUrls.length} entries, ${newUrls.length} new, ${archivedHits} already archived.\n`,
);
+ if (deferredAuthCount > 0) {
+ opts.onLog(
+ `Cookie mode is "defer": needs_auth deferred=${deferredAuthCount} — run the "Needs cookies" bucket to download them with cookies.\n`,
+ );
+ }
if (newUrls.length > 0) {
- const res = await runManagedDownloads(opts, newUrls, opts.channelConfig);
+ const res = await runManagedDownloads(
+ opts,
+ newUrls,
+ opts.channelConfig,
+ runCookiePolicy,
+ );
totalNew += res.okCount;
totalFailed += res.failedCount;
totalSkipped += res.skippedCount;
diff --git a/editor/CHANGELOG.md b/editor/CHANGELOG.md
@@ -1,6 +1,10 @@
# Changelog
## [Unreleased]
+- **Channel groups gain a "Show channels as individual chips" checkbox on the Site form.** Toggling it on marks the group `inline` in site.json, so the viewer renders each member channel as its own loose chip in the filter row instead of a collapsible group box. New sites' seeded default group ("All channels") ships with it **on**; groups added manually on the Site form or created inline from a channel form default **off**. See `editor/app/sites/components/SiteForm.tsx`, `common/lib/channelGroups.ts`, and `editor/e2e/sites-crud.spec.ts`.
+- **Cookies-from-browser is now configurable: three cookie modes, per-channel overrides, and a "Needs cookies" bucket.** The single global cookies value used to be hard-wired to one behavior (passed only on the download auth-retry attempt). Settings now carry a **Cookie mode** next to the value — **When required** (default; the old behavior, now also cookie-retrying a failed *metadata prefetch*, so an age-gated video succeeds on its first real attempt and reports `ok-with-cookies` with a `metadata-prefetch-auth-retry` attempt in `download-outcome.json`), **Always** (every yt-dlp invocation carries `--cookies-from-browser`: enumeration, prefetch, availability probes, downloads — for channels whose listing itself needs auth; a clean success still reports plain `ok`), and **Defer** (normal runs never use cookies: known `needs_auth` videos are excluded from sync/download-missing batches — previously they were re-attempted and re-failed on *every* run — and wait for a manual cookie run). Both the value and the mode can be overridden per channel in the channel form's Advanced section (blank/Inherit = use global). Every channel's Download stage gains a **Needs cookies (N)** card listing undownloaded videos whose recorded availability is cookie-recoverable (`needs_auth`, and also `members_only`/`private`, which batches always excluded but a subscribed/owning account's cookies can fetch) with a **Download with cookies** button that re-runs them with cookies forced on every invocation (bookmarkable, like the other bucket jobs; it warns if no cookie value is configured anywhere). Caveats: in defer mode a *manual single-video* download intentionally gets no cookies and no auth retry — the bucket button is the explicit cookie path; availability probes only attach cookies in Always mode, so needs_auth keeps being observed (defer depends on that signal). Back-compat: an existing `settings.json` without `cookieMode` behaves exactly as before (`when-required`). See `common/lib/cookiePolicy.ts`, `common/ytdlp/{downloadOneManaged,runYtdlp}.ts`, `common/controller/{channelSnapshot,checkAvailability}.ts`, and `editor/e2e/cookies-mode.spec.ts`.
+- **Channel forms now edit site membership directly — pick sites, groups, and create groups inline.** A channel's site membership used to be editable only from the Site form (`/sites/<id>`), and creating a channel silently appended it to the active site (a hidden `activeSite` field) with no group choice and no visibility. Both the **New channel** and channel **Configure** forms now carry a **Sites** section: every configured site listed with a membership checkbox and a compact group dropdown — `(default)`, any existing group, or **"+ New group…"**, which reveals a name input and creates the group on that site (id slugified from the name, reused if it already exists) as part of the save. On create, the active site (`?site=`) is pre-checked, reproducing the old behavior but visibly and overridably; on edit, current memberships pre-check with their groups, and unchecking removes the membership. The checked sites serialize into one hidden `siteMembershipsJson` field (the SiteForm hidden-JSON precedent); the server plans all writes up front — validating group ids, preserving other channels' entries and this channel's `order`, erroring clearly on a since-deleted group, skipping since-deleted sites, and rewriting only sites that actually changed. See the new `editor/app/channels/{lib/siteMemberships.ts,components/SiteMembershipsSection.tsx}`, `editor/app/channels/{actions.ts,components/{ChannelForm,ChannelFormClient}.tsx,new/page.tsx,[slug]/page.tsx}`, and `editor/e2e/channel-site-membership.spec.ts`.
+- **The monitor widget can now start a sync and shows sync freshness + scheduler health — each behind its own flag.** The widget was start-a-sync-less and said nothing about how fresh your channels were. Five new opt-in URL flags, all toggleable in the builder and the in-widget gear (all default off, so existing links are unchanged): **(1) a channel-aware Sync button** (`sync=1`) — pinned to one channel (`channel=X`) it runs that channel's streaming sync; otherwise it sweeps all channels (`syncAllChannelsAction`), briefly showing `Queued X · skipped Y`. **(2) A last-sync readout** (`lastsync=1`) — `Last full sync: …` from a new persisted `lastSyncAllAt` marker written at the end of each Sync-all sweep, plus a `Last channel sync: …` line when an individual channel synced more recently. **(3) A scheduler-status strip** (`sched=1`) — `Auto-sync on/off · next … · last run …`, derived from the same `buildScheduleView` the `/scheduler` page uses. **(4) An absolute-time toggle** (`abstime=1`) — readouts show locale timestamps instead of relative "5m ago". **(5) A Sync-all confirm** (`syncask=1`) — a `window.confirm` before a full sweep. The two readouts poll a new lightweight `/api/widget/sync` route (a few scalars, 15s floor) and, with the Sync button/controls, keep the widget from collapsing to "Idle" — sync health is exactly what you check when nothing's running. See `editor/app/widget/lib/config.ts`, `editor/app/widget/components/{MonitorWidget,WidgetControls,WidgetConfigForm}.tsx`, the new `editor/app/api/widget/sync/route.ts`, `common/jobs/syncSchedulerState.ts` (`lastSyncAllAt`), `editor/app/channels/actions.ts`, and `editor/e2e/widget.spec.ts`.
- **The audio-integrity check now tightens its probe interval when a source starts serving corruption, then relaxes as it stabilises.** Previously the integrity probe ran on a fixed cadence (default 60s) for the whole download, so up to ~60s of bytes were downloaded — and discarded — between a corruption event and the checkpoint that caught it. The interval is now adaptive (AIMD, like TCP congestion control, inverted): each **malformed** checkpoint **halves** the live interval (60→30→15→10s, floored at the existing `AUDIO_CHECK_INTERVAL_MIN_SECONDS` of 10s), so a misbehaving source gets probed more aggressively and wastes fewer bytes per rollback; a run of clean checkpoints then **steps it back up** additively (+15s after every 2 clean probes) toward the configured interval. The reduced cadence persists across yt-dlp relaunches for the rest of the download run. Fully backward compatible — a clean download never leaves the configured interval. Tunable via constants in `common/lib/channelConfig.ts` (`AUDIO_CHECK_INTERVAL_BACKOFF_FACTOR_DEFAULT`, `AUDIO_CHECK_INTERVAL_RECOVER_STEP_SECONDS`, `AUDIO_CHECK_INTERVAL_RECOVER_AFTER_CLEAN`) plus test-only env overrides. See `common/ytdlp/audioCheckCadence.ts` (pure AIMD math + `audioCheckCadence.test.ts`) and `common/ytdlp/audioCheckedDownload.ts` (`resolveKnobs`, the watcher loop, and the advance/malformed checkpoint branches).
- **Docker build mode is now real: build every site in parallel, then deploy them serially.** The `Docker` build mode (Settings → Build pipeline) was previously a stub that fell back to the basic build. It now runs a proper pipeline, driven by a new **Build all sites** control on the Deploy page (one job, one log, one Cancel). The shared, corpus-scale work — the search index, the per-site staging, and the downloadable archive zips — runs **once on the host**; then each site's `compose + next build` runs in its **own container in parallel** (capped by the **Max parallel builds** setting), each writing an isolated per-site `out/` under `export/.export-builds/<siteId>/`; then the built sites **deploy serially** on the host (R2 upload + `wrangler pages deploy`), tolerant of a single site failing. Containers are read-only over the shared corpus/index/archive cache and run as your host user so outputs aren't root-owned. The image (`Dockerfile.build`, tag from **Build image**) is built/reused via Docker layer caching; when no container engine is available the action falls back to a serial host build+deploy. New env knobs: `DOCKER_BIN` (e.g. `podman`), `DOCKER_BUILD_MEMORY`/`DOCKER_BUILD_CPUS` (per-container caps). See `editor/app/deploy/buildDeployCore.ts` (`runDockerBuildAllPhase`/`runDockerDeployAllPhase`), `editor/app/build/buildAction.ts` (`buildAndDeployAllSitesAction`/`buildAllSitesAction`), `editor/app/deploy/components/BuildAllSitesButton.tsx`, `Dockerfile.build`, `docker/build-site.sh`, `common/bin/build-archives.ts`, and **[DEPLOY_DOCKER.md](../DEPLOY_DOCKER.md)**.
diff --git a/editor/app/api/widget/sync/route.ts b/editor/app/api/widget/sync/route.ts
@@ -0,0 +1,72 @@
+import { NextResponse } from "next/server";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { listChannels } from "yt-dlp-transcript-common/controller/channels";
+import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { buildScheduleView } from "yt-dlp-transcript-common/jobs/syncScheduler";
+import { readSchedulerState } from "yt-dlp-transcript-common/jobs/syncSchedulerState";
+
+export const dynamic = "force-dynamic";
+
+// Backs the monitor widget's last-sync and scheduler-status strips. Keeps the
+// payload tiny (a handful of scalars) rather than shipping the full
+// scheduler-status payload on every poll — the widget only needs freshness
+// markers and a one-line scheduler summary. Reuses the same helpers the
+// /scheduler status page does so the two never drift.
+export type WidgetSyncPayload = {
+ // Epoch ms of the last manual full "Sync all" sweep. null = never.
+ lastSyncAllAt: number | null;
+ // Newest per-channel lastSyncedAt across all channels. null = none synced.
+ lastIndividualSyncAt: number | null;
+ scheduler: {
+ enabled: boolean;
+ lastRunAt: number | null; // last scheduled sweep (state.runs[0].at)
+ nextRunAt: number | null; // soonest eligible channel's nextDueAt
+ overdue: boolean; // some eligible channel is due now
+ };
+};
+
+export async function GET() {
+ const paths = getPaths();
+ const settings = getSettings();
+ const state = await readSchedulerState(paths);
+ const now = Date.now();
+ const channels = await listChannels(paths);
+
+ let lastIndividualSyncAt: number | null = null;
+ for (const c of channels) {
+ const last = c.config.lastSyncedAt;
+ if (!last) continue;
+ const parsed = Date.parse(last);
+ if (Number.isNaN(parsed)) continue;
+ if (lastIndividualSyncAt === null || parsed > lastIndividualSyncAt) {
+ lastIndividualSyncAt = parsed;
+ }
+ }
+
+ const view = buildScheduleView({
+ channels: channels.map((c) => ({ slug: c.slug, config: c.config })),
+ scheduler: settings.syncScheduler,
+ state,
+ now,
+ });
+ const eligible = view.filter((v) => v.autoSyncEligible);
+ let nextRunAt: number | null = null;
+ let overdue = false;
+ for (const v of eligible) {
+ if (v.overdue) overdue = true;
+ if (v.nextDueAt !== null && (nextRunAt === null || v.nextDueAt < nextRunAt)) {
+ nextRunAt = v.nextDueAt;
+ }
+ }
+
+ return NextResponse.json({
+ lastSyncAllAt: state.lastSyncAllAt,
+ lastIndividualSyncAt,
+ scheduler: {
+ enabled: settings.syncScheduler.enabled,
+ lastRunAt: state.runs[0]?.at ?? null,
+ nextRunAt,
+ overdue,
+ },
+ } satisfies WidgetSyncPayload);
+}
diff --git a/editor/app/channels/[slug]/components/RetryBucketControl.tsx b/editor/app/channels/[slug]/components/RetryBucketControl.tsx
@@ -17,6 +17,9 @@ type Props = {
// The snapshot bucket these ids represent. When set, the launched retry job
// is bookmarkable and re-runs against the current bucket members.
bucketKey?: ReplayBucket;
+ // Needs-cookies bucket: force cookie mode "always" for the run and label the
+ // button "Download with cookies" so the manual cookie path is explicit.
+ forceCookies?: boolean;
};
export function RetryBucketControl({
@@ -26,6 +29,7 @@ export function RetryBucketControl({
defaultQueueKey,
existingQueues,
bucketKey,
+ forceCookies,
}: Props) {
const [queue, setQueue] = useState(defaultQueueKey);
const [handlingOverride, setHandlingOverride] = useState("");
@@ -47,10 +51,15 @@ export function RetryBucketControl({
abortOnError,
handlingOverride || undefined,
bucketKey,
+ forceCookies,
)
}
cancelAction={cancelJobAction}
- buttonLabel={`Retry (${ids.length})`}
+ buttonLabel={
+ forceCookies
+ ? `Download with cookies (${ids.length})`
+ : `Retry (${ids.length})`
+ }
runningLabel="Retrying…"
label={`Retry ${actionLabel}`}
extraControls={
diff --git a/editor/app/channels/[slug]/components/stages/DownloadStage.tsx b/editor/app/channels/[slug]/components/stages/DownloadStage.tsx
@@ -28,6 +28,7 @@ type Props = {
excludedFromDownload: ExcludedFromDownload;
noTranscriptIds: string[];
partialDownloadIds: string[];
+ needsCookiesIds: string[];
missingShard: ShardConfigSummary | null;
};
@@ -40,6 +41,7 @@ export function DownloadStage({
excludedFromDownload,
noTranscriptIds,
partialDownloadIds,
+ needsCookiesIds,
missingShard,
}: Props) {
const [downloadQueue, setDownloadQueue] = useState(defaultQueueKey);
@@ -217,6 +219,12 @@ export function DownloadStage({
defaultQueueKey={defaultQueueKey}
existingQueues={existingQueues}
/>
+ <NeedsCookiesList
+ slug={slug}
+ ids={needsCookiesIds}
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ />
</div>
);
}
@@ -416,6 +424,54 @@ function PartialDownloadsList({
);
}
+function NeedsCookiesList({
+ slug,
+ ids,
+ defaultQueueKey,
+ existingQueues,
+}: {
+ slug: string;
+ ids: string[];
+ defaultQueueKey: string;
+ existingQueues: string[];
+}) {
+ if (ids.length === 0) return null;
+ return (
+ <div className="flex flex-col gap-2 rounded border border-border p-3">
+ <div>
+ <h4 className="text-sm font-semibold">
+ Needs cookies ({ids.length})
+ </h4>
+ <p className="text-xs text-muted-foreground">
+ Undownloaded videos that yt-dlp reported as auth-gated
+ (age-restricted, members-only, or private) — browser cookies from a
+ signed-in account may recover them. The button below re-runs them
+ with <code>--cookies-from-browser</code> forced on every invocation,
+ using the cookie value from Settings (or this channel's
+ override).
+ </p>
+ </div>
+ <VideoIdList
+ slug={slug}
+ ids={ids}
+ ariaLabel="needs cookies list"
+ emptyAriaLabel="needs cookies empty"
+ emptyMessage="None"
+ itemAriaLabel={(id) => `needs cookies ${id}`}
+ />
+ <RetryBucketControl
+ slug={slug}
+ ids={ids}
+ actionLabel="needs cookies"
+ defaultQueueKey={defaultQueueKey}
+ existingQueues={existingQueues}
+ bucketKey="needsCookies"
+ forceCookies
+ />
+ </div>
+ );
+}
+
function Heading({ title, desc }: { title: string; desc: string }) {
return (
<div>
diff --git a/editor/app/channels/[slug]/lib/stageStatus.ts b/editor/app/channels/[slug]/lib/stageStatus.ts
@@ -30,6 +30,7 @@ export function normalizeBuckets(
skippedByFilter: raw?.skippedByFilter ?? [],
incompleteTranscript: raw?.incompleteTranscript ?? [],
shortAudio: raw?.shortAudio ?? [],
+ needsCookies: raw?.needsCookies ?? [],
};
}
diff --git a/editor/app/channels/[slug]/page.tsx b/editor/app/channels/[slug]/page.tsx
@@ -35,7 +35,13 @@ import {
TRANSCRIPTION_QUEUE,
} from "yt-dlp-transcript-common/lib/platform";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
+import { listSites } from "yt-dlp-transcript-common/lib/site";
+import { sortGroups } from "yt-dlp-transcript-common/lib/channelGroups";
import { ChannelFormClient } from "../components/ChannelFormClient";
+import type {
+ InitialMembership,
+ SiteMembershipOption,
+} from "../components/SiteMembershipsSection";
import { DeleteChannelForm } from "../components/DeleteChannelForm";
import { RenameChannelForm } from "../components/RenameChannelForm";
import { RunningJobsList } from "../../jobs/components/RunningJobsList";
@@ -219,6 +225,25 @@ export default async function ChannelDetailPage({
"danger",
];
+ // Sites membership section of the configure form: every configured site plus
+ // this channel's current membership (and group) on each.
+ const allSites = listSites(paths);
+ const siteOptions: SiteMembershipOption[] = allSites.map((s) => ({
+ siteId: s.siteId,
+ siteTitle: s.siteTitle,
+ defaultGroupId: s.defaultGroupId,
+ groups: sortGroups(s.groups).map((g) => ({ id: g.id, name: g.name })),
+ }));
+ const initialMemberships: InitialMembership[] = allSites.flatMap((s) => {
+ const m = s.channels.find((c) => c.slug === slug);
+ if (!m) return [];
+ return [
+ m.groupId
+ ? { siteId: s.siteId, groupId: m.groupId }
+ : { siteId: s.siteId },
+ ];
+ });
+
const panels: Partial<Record<StageId, ReactNode>> = {
configure: (
<ChannelFormClient
@@ -230,6 +255,8 @@ export default async function ChannelDetailPage({
}
initial={{ slug, config }}
submitLabel="Save changes"
+ sites={siteOptions}
+ initialMemberships={initialMemberships}
/>
),
playlist: (
@@ -250,6 +277,7 @@ export default async function ChannelDetailPage({
excludedFromDownload={excludedFromDownload}
noTranscriptIds={actionableNoTranscriptIds}
partialDownloadIds={buckets.partialDownloads}
+ needsCookiesIds={buckets.needsCookies}
missingShard={downloadMissingShard}
/>
),
diff --git a/editor/app/channels/[slug]/pipelineActions.ts b/editor/app/channels/[slug]/pipelineActions.ts
@@ -24,6 +24,7 @@ import {
import { extractVideoId, runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy";
import { checkDiskSpace } from "yt-dlp-transcript-common/lib/diskSpace";
import { formatBytes } from "yt-dlp-transcript-common/lib/format";
import {
@@ -70,6 +71,10 @@ async function runPipelineAction(
shardIndex?: number;
bucketIds?: ReadonlyArray<string>;
handlingOverride?: ChannelHandling;
+ // retry-bucket only (the "Needs cookies" bucket): force cookie mode
+ // "always" for this run so every yt-dlp invocation carries the configured
+ // cookies and the defer-mode exclusion is bypassed.
+ forceCookies?: boolean;
// Per-run persistence overrides (Phase 2), forwarded to runYtdlp →
// downloadOneManaged for every managed download in this run.
keepSourceVideoOverride?: boolean;
@@ -166,6 +171,7 @@ async function runPipelineAction(
shardIndex: options?.shardIndex,
bucketIds: options?.bucketIds,
handlingOverride: options?.handlingOverride,
+ forceCookies: options?.forceCookies,
keepSourceVideoOverride: options?.keepSourceVideoOverride,
extractImmediately: options?.extractImmediately,
audioFormatOverride: options?.audioFormatOverride,
@@ -302,6 +308,8 @@ export async function retryBucketAction(
// against the CURRENT bucket; omitted by ad-hoc checkbox selections, which are
// therefore not bookmarkable.
bucketKey?: ReplayBucket,
+ // Needs-cookies bucket only: run with cookie mode forced to "always".
+ forceCookies?: boolean,
): Promise<StreamActionResult> {
if (!Array.isArray(bucketIds) || bucketIds.length === 0) {
return { ok: false, error: "No video IDs supplied for retry." };
@@ -321,7 +329,7 @@ export async function retryBucketAction(
kind: "retry-bucket",
slug,
bucket: bucketKey,
- params: { queueKey, abortOnError, handlingOverride },
+ params: { queueKey, abortOnError, handlingOverride, forceCookies },
}
: undefined;
return runPipelineAction(
@@ -329,7 +337,13 @@ export async function retryBucketAction(
"retry-bucket",
"retry-bucket",
queueKey,
- { bucketIds, handlingOverride: handling, abortOnError, spec },
+ {
+ bucketIds,
+ handlingOverride: handling,
+ abortOnError,
+ forceCookies,
+ spec,
+ },
);
}
@@ -380,7 +394,7 @@ export async function importVideoAction(
videoUrl,
onLog: task.onLog,
signal,
- globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ cookiePolicy: resolveCookiePolicy(settings, channelConfig),
inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
diff --git a/editor/app/channels/[slug]/videos/[id]/videoActions.ts b/editor/app/channels/[slug]/videos/[id]/videoActions.ts
@@ -36,6 +36,7 @@ import { transcribeWithWorker } from "yt-dlp-transcript-common/controller/transc
import { findVideoSourceUrl } from "yt-dlp-transcript-common/controller/undownloadedVideos";
import { unpersistSavedVideo } from "yt-dlp-transcript-common/lib/savedVideo-server";
import { getSettings } from "yt-dlp-transcript-common/lib/settings";
+import { resolveCookiePolicy } from "yt-dlp-transcript-common/lib/cookiePolicy";
import { downloadOneManaged } from "yt-dlp-transcript-common/ytdlp/downloadOneManaged";
import { runYtdlp } from "yt-dlp-transcript-common/ytdlp/runYtdlp";
import {
@@ -172,7 +173,7 @@ export async function downloadVideoPipelineAction(
videoUrl: url,
onLog: task.onLog,
signal,
- globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ cookiePolicy: resolveCookiePolicy(settings, r.config),
inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
globalSkipLiveDownloads: settings.skipLiveDownloads,
downloadFormatPreset: resolveDownloadFormatPreset({
@@ -237,7 +238,7 @@ export async function redownloadToArchiveAction(
videoUrl: url,
onLog: task.onLog,
signal,
- globalCookiesFromBrowser: settings.cookiesFromBrowser || undefined,
+ cookiePolicy: resolveCookiePolicy(settings, r.config),
inlineTranscribeOnFallback: settings.inlineTranscribeOnFallback,
globalSkipLiveDownloads: settings.skipLiveDownloads,
appendArchive: true,
diff --git a/editor/app/channels/actions.ts b/editor/app/channels/actions.ts
@@ -17,17 +17,21 @@ import { renameChannel } from "yt-dlp-transcript-common/controller/renameChannel
import { generateChannelSnapshot } from "yt-dlp-transcript-common/controller/channelSnapshot";
import { requestChannelSnapshot } from "yt-dlp-transcript-common/jobs/snapshotScheduler";
import { getRegistry } from "yt-dlp-transcript-common/jobs/registry";
-import {
- getSite,
- isValidSiteId,
- listSiteIds,
- writeSite,
-} from "yt-dlp-transcript-common/lib/site";
+import type { Site } from "yt-dlp-transcript-common/lib/site";
import { activeSyncSlugs } from "yt-dlp-transcript-common/jobs/syncJobs";
import {
+ readSchedulerState,
+ writeSchedulerState,
+} from "yt-dlp-transcript-common/jobs/syncSchedulerState";
+import {
CHANNEL_FORM_FIELDS,
parseChannelForm,
} from "./components/parseChannelForm";
+import {
+ applySiteWrites,
+ parseSiteMembershipsField,
+ planSiteMembershipWrites,
+} from "./lib/siteMemberships";
import { syncAction } from "./[slug]/pipelineActions";
export type ActionResult = { error: string } | undefined;
@@ -53,28 +57,35 @@ export async function createChannelAction(
};
}
const paths = getPaths();
+ // Validate + plan the site-membership writes from the form's Sites section
+ // BEFORE the channel exists, so a rejected submit can be corrected and
+ // resubmitted without hitting "already exists". A null parse result (field
+ // absent — zero sites configured) skips membership work entirely.
+ let siteWrites: Site[] = [];
+ try {
+ const requests = parseSiteMembershipsField(
+ formData.get("siteMembershipsJson"),
+ );
+ if (requests) {
+ siteWrites = planSiteMembershipWrites(paths, slug, requests);
+ }
+ } catch (e) {
+ return { error: (e as Error).message };
+ }
if (await channelExists(paths, slug)) {
return { error: `Channel "${slug}" already exists` };
}
await createChannel(paths, slug, config);
- // When created under a specific active site, add the new channel to that
- // site's membership so it's visible in the scoped Channels list. No-op under
- // "all sites" (the hidden field is omitted by the form in that case).
- const activeSite = formData.get("activeSite");
- if (
- typeof activeSite === "string" &&
- isValidSiteId(activeSite) &&
- listSiteIds(paths).includes(activeSite)
- ) {
- const site = getSite(activeSite, paths);
- if (!site.channels.some((c) => c.slug === slug)) {
- await writeSite(
- { ...site, channels: [...site.channels, { slug }] },
- paths,
- );
- }
+ try {
+ await applySiteWrites(siteWrites, paths);
+ } catch (e) {
+ // The channel itself was created; don't redirect as if nothing happened.
+ return {
+ error: `Channel "${slug}" was created, but updating site memberships failed: ${(e as Error).message}. Open its Configure panel to retry.`,
+ };
}
revalidatePath("/channels");
+ revalidatePath("/sites");
revalidatePath("/");
redirect(`/channels/${slug}`);
}
@@ -93,6 +104,20 @@ export async function updateChannelAction(
const paths = getPaths();
const existing = await readChannelConfig(paths, slug);
if (!existing) return { error: `Channel "${slug}" not found` };
+ // Plan the site-membership writes (Sites section) before touching anything,
+ // so bad input errors out with no partial write. Null = field absent
+ // (zero sites configured / legacy submit) → leave memberships alone.
+ let siteWrites: Site[] = [];
+ try {
+ const requests = parseSiteMembershipsField(
+ formData.get("siteMembershipsJson"),
+ );
+ if (requests) {
+ siteWrites = planSiteMembershipWrites(paths, slug, requests);
+ }
+ } catch (e) {
+ return { error: (e as Error).message };
+ }
// The form parser only emits keys whose form value is meaningful, so a
// cleared input is absent from `parsed.config`. A plain spread would keep
// the stale value from `existing`. Clear every form-managed key from the
@@ -103,11 +128,18 @@ export async function updateChannelAction(
for (const key of CHANNEL_FORM_FIELDS) delete merged[key];
Object.assign(merged, parsed.config);
await writeChannelConfig(paths, slug, merged);
+ try {
+ await applySiteWrites(siteWrites, paths);
+ } catch (e) {
+ return { error: (e as Error).message };
+ }
// Config changes (e.g. audioFormat / handling) feed snapshot buckets, so
// refresh the report through the global debounced scheduler.
requestChannelSnapshot(paths, slug);
revalidatePath("/channels");
revalidatePath(`/channels/${slug}`);
+ revalidatePath("/sites");
+ revalidatePath("/");
return undefined;
}
@@ -223,6 +255,12 @@ export async function syncAllChannelsAction(): Promise<SyncAllResult> {
// runManagedFunction's onLog regardless.
void result.stream.cancel();
}
+ // Record the sweep's freshness marker for the monitor widget's last-sync
+ // readout. Read-modify-write right before the write keeps the clobber window
+ // vs. a concurrent scheduler tick minimal (single-user editor — acceptable).
+ const state = await readSchedulerState(paths);
+ state.lastSyncAllAt = Date.now();
+ await writeSchedulerState(paths, state);
return { queued, skipped };
}
diff --git a/editor/app/channels/components/ChannelForm.tsx b/editor/app/channels/components/ChannelForm.tsx
@@ -15,15 +15,23 @@ import {
AUDIO_CHECK_MAX_ROLLBACKS_MIN,
} from "yt-dlp-transcript-common/lib/channelConfig";
import { SYNC_INTERVAL_PRESETS } from "../../scheduler/intervalPresets";
+import {
+ SiteMembershipsSection,
+ type InitialMembership,
+ type SiteMembershipOption,
+} from "./SiteMembershipsSection";
type Props = {
action: string | ((formData: FormData) => void | Promise<void>);
initial?: { slug: string; config: ChannelConfig };
submitLabel: string;
errorMessage?: string;
- // Extra hidden fields submitted with the form (e.g. the active site so a newly
- // created channel can be added to that site's membership).
- hiddenFields?: Record<string, string>;
+ // All configured sites, for the Sites membership section.
+ sites: SiteMembershipOption[];
+ // Edit mode: the channel's current site memberships.
+ initialMemberships?: InitialMembership[];
+ // Create mode: the active site (from ?site=) to pre-check.
+ activeSiteId?: string | null;
};
export function ChannelForm({
@@ -31,16 +39,14 @@ export function ChannelForm({
initial,
submitLabel,
errorMessage,
- hiddenFields,
+ sites,
+ initialMemberships,
+ activeSiteId,
}: Props) {
const isEdit = !!initial;
const c = initial?.config;
return (
<form action={action} className="flex flex-col gap-4 max-w-xl">
- {hiddenFields &&
- Object.entries(hiddenFields).map(([name, value]) => (
- <input key={name} type="hidden" name={name} defaultValue={value} />
- ))}
{errorMessage && (
<div className="rounded border border-destructive/30 bg-destructive-soft px-3 py-2 text-sm text-destructive">
{errorMessage}
@@ -69,13 +75,13 @@ export function ChannelForm({
placeholder="kebab-case (auto-derived from name if blank)"
/>
)}
- <p className="text-xs text-muted-foreground">
- Channel grouping is now configured per site under{" "}
- <a href="/sites" className="underline">
- Sites
- </a>
- , where each site picks its own channels and their groups.
- </p>
+ </Section>
+ <Section title="Sites">
+ <SiteMembershipsSection
+ sites={sites}
+ initialMemberships={initialMemberships}
+ activeSiteId={activeSiteId}
+ />
</Section>
<Section title="Source">
<fieldset className="flex flex-col gap-2">
@@ -245,6 +251,39 @@ export function ChannelForm({
className="rounded border border-border bg-card px-2 py-1 text-sm font-mono"
/>
</label>
+ <Field
+ label="Cookies from browser"
+ name="cookiesFromBrowser"
+ defaultValue={c?.cookiesFromBrowser ?? ""}
+ placeholder="(inherit global)"
+ hint="Per-channel override of the global yt-dlp --cookies-from-browser browser spec (e.g. firefox, chrome:Default). Blank inherits the global value from /settings."
+ />
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Cookie mode</span>
+ <select
+ name="cookieMode"
+ defaultValue={c?.cookieMode ?? ""}
+ aria-label="cookie mode"
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="">Inherit global</option>
+ <option value="when-required">
+ When required — retry auth/age failures with cookies
+ </option>
+ <option value="always">
+ Always — pass cookies on every yt-dlp invocation
+ </option>
+ <option value="defer">
+ Defer — never in normal runs; collect into "Needs cookies"
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ When this channel's downloads use browser cookies. Leave on
+ Inherit to use the global mode from /settings. Defer excludes
+ auth-gated videos from batch runs and gathers them into the
+ “Needs cookies” bucket on the Download stage.
+ </span>
+ </label>
<label className="flex flex-col gap-1 text-sm">
<span className="font-medium">
Sleep between managed downloads
diff --git a/editor/app/channels/components/ChannelFormClient.tsx b/editor/app/channels/components/ChannelFormClient.tsx
@@ -5,6 +5,10 @@ import { ChannelForm } from "./ChannelForm";
import type { ChannelConfig } from "yt-dlp-transcript-common/lib/channelConfig";
import type { ActionResult } from "../actions";
import { ALL_SITES } from "../../lib/activeSite";
+import type {
+ InitialMembership,
+ SiteMembershipOption,
+} from "./SiteMembershipsSection";
type Props = {
action: (
@@ -13,16 +17,26 @@ type Props = {
) => Promise<ActionResult>;
initial?: { slug: string; config: ChannelConfig };
submitLabel: string;
+ sites: SiteMembershipOption[];
+ initialMemberships?: InitialMembership[];
};
-export function ChannelFormClient({ action, initial, submitLabel }: Props) {
+export function ChannelFormClient({
+ action,
+ initial,
+ submitLabel,
+ sites,
+ initialMemberships,
+}: Props) {
const [state, formAction] = useActionState<ActionResult, FormData>(
action,
undefined,
);
// The active site is mirrored into the `?site=` URL param by SiteScopeSelect.
// Read it here (avoiding useSearchParams so this form needn't be wrapped in a
- // Suspense boundary) so a newly created channel joins that site's membership.
+ // Suspense boundary) so a newly created channel pre-checks that site in the
+ // Sites membership section. Only meaningful in create mode — an existing
+ // channel's memberships come from initialMemberships.
const [activeSite, setActiveSite] = useState<string | null>(null);
useEffect(() => {
try {
@@ -31,15 +45,17 @@ export function ChannelFormClient({ action, initial, submitLabel }: Props) {
/* ignore */
}
}, []);
- const hiddenFields =
- activeSite && activeSite !== ALL_SITES ? { activeSite } : undefined;
+ const activeSiteId =
+ !initial && activeSite && activeSite !== ALL_SITES ? activeSite : null;
return (
<ChannelForm
action={formAction}
initial={initial}
submitLabel={submitLabel}
errorMessage={state?.error}
- hiddenFields={hiddenFields}
+ sites={sites}
+ initialMemberships={initialMemberships}
+ activeSiteId={activeSiteId}
/>
);
}
diff --git a/editor/app/channels/components/DeleteChannelForm.tsx b/editor/app/channels/components/DeleteChannelForm.tsx
@@ -27,6 +27,7 @@ export function DeleteChannelForm({ slug, action }: Props) {
name="confirmSlug"
required
placeholder={slug}
+ aria-label="confirm slug to delete"
className="rounded border border-border bg-card px-2 py-1 text-sm font-mono"
/>
<button
diff --git a/editor/app/channels/components/SiteMembershipsSection.tsx b/editor/app/channels/components/SiteMembershipsSection.tsx
@@ -0,0 +1,192 @@
+"use client";
+
+import { useEffect, useState } from "react";
+
+// The channel form's per-site membership picker: every configured site with a
+// checkbox (member or not) plus a group dropdown, including a "+ New group…"
+// option that reveals a name input. Serializes the checked sites into the
+// hidden `siteMembershipsJson` field consumed by lib/siteMemberships.ts —
+// same hidden-JSON-field pattern as SiteForm's groupsJson/channelsJson.
+
+export type SiteMembershipOption = {
+ siteId: string;
+ siteTitle: string;
+ defaultGroupId: string;
+ // Pre-sorted server-side via sortGroups.
+ groups: { id: string; name: string }[];
+};
+
+export type InitialMembership = { siteId: string; groupId?: string };
+
+// Sentinel select value for "+ New group…". Can never collide with a real
+// group id — underscores fail isValidGroupId (same trick as ALL_SITES).
+const NEW_GROUP = "__new__";
+
+type Row = { groupId: string; newGroupName: string };
+
+type Props = {
+ sites: SiteMembershipOption[];
+ // Edit mode: the channel's current memberships (pre-checked).
+ initialMemberships?: InitialMembership[];
+ // Create mode: the active site from `?site=` to pre-check. Arrives via a
+ // mount effect in ChannelFormClient, before any user interaction.
+ activeSiteId?: string | null;
+};
+
+export function SiteMembershipsSection({
+ sites,
+ initialMemberships,
+ activeSiteId,
+}: Props) {
+ // Presence in the map = checked. groupId "" = the site's default group.
+ const [selected, setSelected] = useState<Map<string, Row>>(
+ () =>
+ new Map(
+ (initialMemberships ?? []).map((m) => [
+ m.siteId,
+ { groupId: m.groupId ?? "", newGroupName: "" },
+ ]),
+ ),
+ );
+
+ // Create-mode pre-check of the active site (mirrors the old hidden
+ // `activeSite` field's behavior). Fires before user interaction, so no
+ // clobber guard beyond "already checked" is needed.
+ useEffect(() => {
+ if (!activeSiteId || !sites.some((s) => s.siteId === activeSiteId)) return;
+ setSelected((prev) => {
+ if (prev.has(activeSiteId)) return prev;
+ const next = new Map(prev);
+ next.set(activeSiteId, { groupId: "", newGroupName: "" });
+ return next;
+ });
+ }, [activeSiteId, sites]);
+
+ if (sites.length === 0) {
+ // No hidden field at all: the actions skip membership reconciliation.
+ return (
+ <p className="text-xs text-muted-foreground">
+ No sites configured — create one under{" "}
+ <a href="/sites" className="underline">
+ Sites
+ </a>
+ .
+ </p>
+ );
+ }
+
+ const toggle = (siteId: string, on: boolean) =>
+ setSelected((prev) => {
+ const next = new Map(prev);
+ if (on) next.set(siteId, prev.get(siteId) ?? { groupId: "", newGroupName: "" });
+ else next.delete(siteId);
+ return next;
+ });
+ const setGroup = (siteId: string, groupId: string) =>
+ setSelected((prev) => {
+ const next = new Map(prev);
+ const row = prev.get(siteId) ?? { groupId: "", newGroupName: "" };
+ next.set(siteId, { ...row, groupId });
+ return next;
+ });
+ const setNewGroupName = (siteId: string, newGroupName: string) =>
+ setSelected((prev) => {
+ const next = new Map(prev);
+ const row = prev.get(siteId) ?? { groupId: "", newGroupName: "" };
+ next.set(siteId, { ...row, newGroupName });
+ return next;
+ });
+
+ // One entry per CHECKED site; an unchecked site is simply absent (= remove
+ // membership on edit). See MembershipRequest in lib/siteMemberships.ts.
+ const membershipsJson = JSON.stringify(
+ sites
+ .filter((s) => selected.has(s.siteId))
+ .map((s) => {
+ const row = selected.get(s.siteId)!;
+ if (row.groupId === NEW_GROUP) {
+ const name = row.newGroupName.trim();
+ // The visible name input is `required`, so an empty name never
+ // submits; falling back to default is belt-and-braces only.
+ return name
+ ? { siteId: s.siteId, newGroupName: name }
+ : { siteId: s.siteId };
+ }
+ return row.groupId
+ ? { siteId: s.siteId, groupId: row.groupId }
+ : { siteId: s.siteId };
+ }),
+ );
+
+ return (
+ <div className="flex flex-col gap-1">
+ <input type="hidden" name="siteMembershipsJson" value={membershipsJson} />
+ <p className="text-xs text-muted-foreground">
+ Pick the sites this channel appears on, and the group it belongs to on
+ each.
+ </p>
+ {sites.map((s) => {
+ const row = selected.get(s.siteId);
+ const on = !!row;
+ const label = s.siteTitle || s.siteId;
+ return (
+ <div
+ key={s.siteId}
+ className="flex flex-col gap-1 py-1 border-b border-border last:border-b-0"
+ >
+ <div className="flex items-center gap-2 text-sm">
+ <label className="flex items-center gap-2 flex-1 min-w-0">
+ <input
+ type="checkbox"
+ checked={on}
+ onChange={(e) => toggle(s.siteId, e.target.checked)}
+ aria-label={`Include on ${label}`}
+ />
+ <span className="truncate">
+ {label}{" "}
+ <code className="text-xs text-muted-foreground">
+ {s.siteId}
+ </code>
+ </span>
+ </label>
+ {on && (
+ <select
+ value={row.groupId}
+ onChange={(e) => setGroup(s.siteId, e.target.value)}
+ aria-label={`Group for ${label}`}
+ className="rounded border border-border bg-card px-1.5 py-0.5 text-xs shrink-0"
+ >
+ <option value="">(default)</option>
+ {s.groups.map((g) => (
+ <option key={g.id} value={g.id}>
+ {g.name || g.id}
+ </option>
+ ))}
+ <option value={NEW_GROUP}>+ New group…</option>
+ </select>
+ )}
+ </div>
+ {on && row.groupId === NEW_GROUP && (
+ <input
+ type="text"
+ value={row.newGroupName}
+ onChange={(e) => setNewGroupName(s.siteId, e.target.value)}
+ required
+ placeholder="New group name"
+ aria-label={`New group name for ${label}`}
+ className="ml-6 rounded border border-border bg-card px-2 py-1 text-sm"
+ />
+ )}
+ </div>
+ );
+ })}
+ <p className="text-xs text-muted-foreground">
+ Full group management (rename, reorder, defaults) lives under{" "}
+ <a href="/sites" className="underline">
+ Sites
+ </a>
+ .
+ </p>
+ </div>
+ );
+}
diff --git a/editor/app/channels/components/parseChannelForm.ts b/editor/app/channels/components/parseChannelForm.ts
@@ -17,6 +17,7 @@ import {
detectPlatform,
type Platform,
} from "yt-dlp-transcript-common/lib/platform";
+import { isCookieMode } from "yt-dlp-transcript-common/lib/cookiePolicy";
import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
export type ParsedChannelForm = {
@@ -41,6 +42,8 @@ export const CHANNEL_FORM_FIELDS = [
"ytdlpExtraArgs",
"syncIntervalMinutes",
"sleepBetweenDownloadsSeconds",
+ "cookiesFromBrowser",
+ "cookieMode",
"audioCheck",
] as const satisfies ReadonlyArray<keyof ChannelConfig>;
@@ -143,6 +146,12 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
syncIntervalMinutes = n;
}
+ // Cookie overrides. Blank value / "Inherit global" mode = omit, so a cleared
+ // input actually clears the stored override (see CHANNEL_FORM_FIELDS).
+ const cookiesFromBrowser = stringOrUndef(formData, "cookiesFromBrowser");
+ const cookieModeRaw = String(formData.get("cookieMode") ?? "").trim();
+ const cookieMode = isCookieMode(cookieModeRaw) ? cookieModeRaw : undefined;
+
const audioCheckEnabled = formData.get("audioCheckEnabled") != null;
let audioCheck: AudioCheckConfig | undefined;
if (audioCheckEnabled) {
@@ -221,6 +230,8 @@ export function parseChannelForm(formData: FormData): ParsedChannelForm {
if (sleepBetweenDownloadsSeconds != null) {
config.sleepBetweenDownloadsSeconds = sleepBetweenDownloadsSeconds;
}
+ if (cookiesFromBrowser) config.cookiesFromBrowser = cookiesFromBrowser;
+ if (cookieMode) config.cookieMode = cookieMode;
if (audioCheck) config.audioCheck = audioCheck;
return { name, slug, config };
diff --git a/editor/app/channels/lib/siteMemberships.ts b/editor/app/channels/lib/siteMemberships.ts
@@ -0,0 +1,152 @@
+import slugify from "@sindresorhus/slugify";
+import { isValidGroupId } from "yt-dlp-transcript-common/lib/channelGroups";
+import type { Paths } from "yt-dlp-transcript-common/lib/paths";
+import {
+ getSite,
+ listSiteIds,
+ writeSite,
+ type Site,
+ type SiteChannelMembership,
+} from "yt-dlp-transcript-common/lib/site";
+
+// Server-side half of the channel form's "Sites" membership section (see
+// SiteMembershipsSection.tsx). The form serializes one entry per CHECKED site
+// into the hidden `siteMembershipsJson` field; a site absent from the array
+// means "not a member" (removal on edit). The field being absent entirely
+// (zero sites configured, or a legacy submit) means "don't touch memberships".
+
+export type MembershipRequest = {
+ siteId: string;
+ // Explicit existing group on that site; absent = the site's default group.
+ groupId?: string;
+ // Mutually exclusive with groupId: create this group (id = slugify(name))
+ // on the site and assign the channel to it.
+ newGroupName?: string;
+};
+
+// Parse the raw hidden-field value. Returns null when the field is absent so
+// callers can skip membership reconciliation entirely; throws on a payload
+// that isn't a JSON array (never produced by our form).
+export function parseSiteMembershipsField(
+ raw: FormDataEntryValue | null,
+): MembershipRequest[] | null {
+ if (typeof raw !== "string") return null;
+ let parsed: unknown;
+ try {
+ parsed = JSON.parse(raw);
+ } catch {
+ throw new Error("Malformed site memberships payload");
+ }
+ if (!Array.isArray(parsed)) {
+ throw new Error("Malformed site memberships payload");
+ }
+ const out: MembershipRequest[] = [];
+ const seen = new Set<string>();
+ for (const entry of parsed) {
+ if (!entry || typeof entry !== "object") continue;
+ const r = entry as Record<string, unknown>;
+ if (typeof r.siteId !== "string" || !r.siteId || seen.has(r.siteId)) {
+ continue;
+ }
+ seen.add(r.siteId);
+ const req: MembershipRequest = { siteId: r.siteId };
+ // newGroupName wins when both are present (the form never emits both).
+ if (typeof r.newGroupName === "string" && r.newGroupName.trim()) {
+ req.newGroupName = r.newGroupName.trim();
+ } else if (typeof r.groupId === "string" && r.groupId.trim()) {
+ req.groupId = r.groupId.trim();
+ }
+ out.push(req);
+ }
+ return out;
+}
+
+// Compute the Site objects that need rewriting so `slug`'s membership matches
+// `requests`. Pure planning: reads every configured site, validates the
+// requests (throws a user-facing Error on bad input), and returns ONLY the
+// sites whose groups/membership actually changed — a no-op save rewrites
+// nothing. Requests naming unknown siteIds are silently skipped (the site was
+// deleted since the form loaded).
+export function planSiteMembershipWrites(
+ paths: Paths,
+ slug: string,
+ requests: MembershipRequest[],
+): Site[] {
+ const bySiteId = new Map(requests.map((r) => [r.siteId, r]));
+ const changed: Site[] = [];
+ for (const siteId of listSiteIds(paths)) {
+ const site = getSite(siteId, paths);
+ const request = bySiteId.get(siteId);
+ const existing = site.channels.find((c) => c.slug === slug);
+
+ // Resolve the target group (possibly creating one) for a checked site.
+ let groups = site.groups;
+ let groupId: string | undefined;
+ if (request) {
+ if (request.newGroupName) {
+ const id = slugify(request.newGroupName);
+ if (!isValidGroupId(id)) {
+ throw new Error(
+ `Could not derive a valid group id from "${request.newGroupName}" (site "${siteId}")`,
+ );
+ }
+ // Same slug as an existing group -> reuse it (idempotent resubmits).
+ if (!groups.some((g) => g.id === id)) {
+ // Matches SiteForm's addGroup default: new groups start unselected
+ // and without the inline flag (only the seeded default group renders
+ // its channels as loose chips) — deliberate, not an omission.
+ groups = [
+ ...groups,
+ { id, name: request.newGroupName, selectedByDefault: false },
+ ];
+ }
+ groupId = id;
+ } else if (request.groupId) {
+ // Explicit group must still exist — writeSite would silently drop an
+ // unknown groupId from the membership, so fail loudly instead.
+ if (!groups.some((g) => g.id === request.groupId)) {
+ throw new Error(
+ `Group "${request.groupId}" no longer exists on site "${siteId}"`,
+ );
+ }
+ groupId = request.groupId;
+ }
+ }
+
+ // Next membership list: preserve every other channel's entry untouched and
+ // keep this channel's existing extra fields (notably `order`).
+ let channels: SiteChannelMembership[];
+ if (!request) {
+ channels = existing
+ ? site.channels.filter((c) => c.slug !== slug)
+ : site.channels;
+ } else if (existing) {
+ channels = site.channels.map((c) => {
+ if (c.slug !== slug) return c;
+ if (groupId) return { ...c, groupId };
+ const { groupId: _drop, ...rest } = c;
+ return rest;
+ });
+ } else {
+ channels = [...site.channels, groupId ? { slug, groupId } : { slug }];
+ }
+
+ const groupsChanged = groups !== site.groups;
+ const membershipChanged = request
+ ? !existing || (existing.groupId ?? undefined) !== groupId
+ : !!existing;
+ if (groupsChanged || membershipChanged) {
+ changed.push({ ...site, groups, channels });
+ }
+ }
+ return changed;
+}
+
+export async function applySiteWrites(
+ sites: Site[],
+ paths: Paths,
+): Promise<void> {
+ for (const site of sites) {
+ await writeSite(site, paths);
+ }
+}
diff --git a/editor/app/channels/new/page.tsx b/editor/app/channels/new/page.tsx
@@ -1,11 +1,22 @@
import type { Metadata } from "next";
import Link from "next/link";
+import { getPaths } from "yt-dlp-transcript-common/lib/paths";
+import { listSites } from "yt-dlp-transcript-common/lib/site";
+import { sortGroups } from "yt-dlp-transcript-common/lib/channelGroups";
import { ChannelFormClient } from "../components/ChannelFormClient";
+import type { SiteMembershipOption } from "../components/SiteMembershipsSection";
import { createChannelAction } from "../actions";
+export const dynamic = "force-dynamic";
export const metadata: Metadata = { title: "New channel — Channels" };
-export default function NewChannelPage() {
+export default async function NewChannelPage() {
+ const sites: SiteMembershipOption[] = listSites(getPaths()).map((s) => ({
+ siteId: s.siteId,
+ siteTitle: s.siteTitle,
+ defaultGroupId: s.defaultGroupId,
+ groups: sortGroups(s.groups).map((g) => ({ id: g.id, name: g.name })),
+ }));
return (
<div className="flex flex-col gap-4">
<div className="flex items-center gap-2 text-sm text-muted-foreground">
@@ -19,6 +30,7 @@ export default function NewChannelPage() {
<ChannelFormClient
action={createChannelAction}
submitLabel="Create channel"
+ sites={sites}
/>
</div>
);
diff --git a/editor/app/jobs/jobReplayRegistry.ts b/editor/app/jobs/jobReplayRegistry.ts
@@ -125,6 +125,7 @@ export const JOB_REPLAY_HANDLERS: Record<string, ReplayHandler> = {
bool(p.abortOnError),
str(p.handlingOverride),
spec.bucket,
+ Boolean(p.forceCookies),
);
},
"download-from-playlist": (spec) => {
diff --git a/editor/app/settings/actions.ts b/editor/app/settings/actions.ts
@@ -21,6 +21,10 @@ import {
type SocialLink,
} from "yt-dlp-transcript-common/lib/settings";
import { DEFAULT_TRANSCRIPTION_APP_ID } from "yt-dlp-transcript-common/lib/transcriptionApps";
+import {
+ DEFAULT_COOKIE_MODE,
+ isCookieMode,
+} from "yt-dlp-transcript-common/lib/cookiePolicy";
import { isDownloadFormatPreset } from "yt-dlp-transcript-common/ytdlp/downloadFormat";
import {
sanitizeWorkers,
@@ -40,6 +44,10 @@ export async function saveSettingsAction(
const cookiesFromBrowser = String(
formData.get("cookiesFromBrowser") ?? "",
).trim();
+ const cookieModeRaw = String(formData.get("cookieMode") ?? "").trim();
+ const cookieMode = isCookieMode(cookieModeRaw)
+ ? cookieModeRaw
+ : DEFAULT_COOKIE_MODE;
const sleepRaw = String(
formData.get("sleepBetweenDownloadsSeconds") ?? "",
).trim();
@@ -213,6 +221,7 @@ export async function saveSettingsAction(
transcriptionApps: {},
workers,
cookiesFromBrowser,
+ cookieMode,
sleepBetweenDownloadsSeconds: sleepParsed,
downloadFormat,
minFreeDiskGB: minFreeDiskParsed,
diff --git a/editor/app/settings/components/SettingsForm.tsx b/editor/app/settings/components/SettingsForm.tsx
@@ -74,11 +74,40 @@ export function SettingsForm({ initial, apps }: Props) {
<WorkersField initial={initial.workers} apps={apps} name="workersJson" />
</fieldset>
<Field
- label="Cookies from browser (retry only)"
+ label="Cookies from browser"
name="cookiesFromBrowser"
defaultValue={initial.cookiesFromBrowser}
- hint="Browser spec passed to yt-dlp --cookies-from-browser only when the primary download attempt fails with an auth/age error and the channel doesn't have its own cookies set. e.g. firefox, chrome:Default. Leave blank to disable."
+ hint="Browser spec passed to yt-dlp --cookies-from-browser (e.g. firefox, chrome:Default) for age-restricted, members-only, and private videos. When it is used is set by the cookie mode below. Leave blank for no cookies. Each channel can override both in its Advanced settings."
/>
+ <label className="flex flex-col gap-1 text-sm">
+ <span className="font-medium">Cookie mode</span>
+ <select
+ name="cookieMode"
+ defaultValue={initial.cookieMode}
+ className="rounded border border-border bg-card px-2 py-1 text-sm"
+ >
+ <option value="when-required">
+ When required — retry auth/age failures with cookies
+ </option>
+ <option value="always">
+ Always — pass cookies on every yt-dlp invocation
+ </option>
+ <option value="defer">
+ Defer — never in normal runs; collect into "Needs cookies"
+ </option>
+ </select>
+ <span className="text-xs text-muted-foreground">
+ <strong>When required</strong> (default) runs cookie-free and retries
+ only the attempts that fail with an auth/age error.{" "}
+ <strong>Always</strong> sends cookies with every invocation
+ (enumeration, metadata prefetch, availability probes, downloads) —
+ for channels whose listing itself needs auth.{" "}
+ <strong>Defer</strong> never uses cookies in normal runs: auth-gated
+ videos are excluded from batches and gathered into each
+ channel's “Needs cookies” bucket, downloaded on
+ demand with its “Download with cookies” button.
+ </span>
+ </label>
<Field
label="Sleep between managed downloads (seconds)"
name="sleepBetweenDownloadsSeconds"
diff --git a/editor/app/sites/components/SiteForm.tsx b/editor/app/sites/components/SiteForm.tsx
@@ -28,6 +28,7 @@ type GroupRow = {
name: string;
description: string;
selectedByDefault: boolean;
+ inline: boolean;
};
// A featured related-sites group: a label + the sibling siteIds picked into it,
@@ -40,6 +41,7 @@ function toGroupRow(g: ChannelGroup): GroupRow {
name: g.name,
description: g.description ?? "",
selectedByDefault: g.selectedByDefault,
+ inline: g.inline === true,
};
}
@@ -81,6 +83,7 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
description: g.description.trim() || undefined,
selectedByDefault: g.selectedByDefault,
order: idx,
+ inline: g.inline || undefined,
})),
);
const channelsJson = JSON.stringify(
@@ -119,7 +122,9 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
const addGroup = () =>
setGroups((prev) => [
...prev,
- { id: "", name: "", description: "", selectedByDefault: false },
+ // Manually added groups default off; only the seeded FALLBACK_GROUP
+ // starts with inline on.
+ { id: "", name: "", description: "", selectedByDefault: false, inline: false },
]);
const removeGroup = (idx: number) =>
setGroups((prev) => {
@@ -470,6 +475,25 @@ export function SiteForm({ initial, channels, allSites, isNew }: Props) {
/>
Selected by default
</label>
+ <label className="flex items-center gap-2 text-sm">
+ <input
+ type="checkbox"
+ checked={g.inline}
+ onChange={(e) =>
+ setGroups((prev) =>
+ prev.map((row, i) =>
+ i === idx ? { ...row, inline: e.target.checked } : row,
+ ),
+ )
+ }
+ className="accent-brand"
+ />
+ Show channels as individual chips
+ </label>
+ <p className="text-xs text-muted-foreground">
+ Renders each member channel as its own loose chip in the
+ viewer's filter row instead of a collapsible group box.
+ </p>
</div>
);
})}
diff --git a/editor/app/widget/components/MonitorWidget.tsx b/editor/app/widget/components/MonitorWidget.tsx
@@ -11,6 +11,7 @@ import { jobKindLabel } from "../../jobs/jobKindLabels";
import type { WorkersPayload, WorkerView } from "../../workers/components/WorkersView";
import { InlineActionButton } from "../../actionable/components/InlineActionButton";
import type { WidgetActionablePayload } from "../../api/widget/actionable/route";
+import type { WidgetSyncPayload } from "../../api/widget/sync/route";
import { buildWidgetQuery, type WidgetConfig } from "../lib/config";
import { WidgetConfigForm } from "./WidgetConfigForm";
import { WidgetControls } from "./WidgetControls";
@@ -93,6 +94,8 @@ export function MonitorWidget({
// a reload re-parses the same config.
const [config, setConfig] = useState<WidgetConfig>(initialConfig);
const [settingsOpen, setSettingsOpen] = useState(false);
+ // Drives the relative "5m ago" labels in the sync readouts; null until mounted.
+ const now = useNow();
const pollMs = config.pollSeconds * 1000;
// The controls row needs live `paused` state, so poll workers whenever either
@@ -126,6 +129,16 @@ export function MonitorWidget({
Math.max(pollMs, 15000),
null,
);
+ // Freshness/scheduler markers move on the order of minutes, so poll no faster
+ // than every 15s regardless of the configured cadence. `refetchSync` lets the
+ // Sync button refresh the readouts the moment a sweep is queued.
+ const { data: syncData, refetch: refetchSync } =
+ usePolledPayload<WidgetSyncPayload>(
+ "/api/widget/sync",
+ config.lastSync || config.scheduler,
+ Math.max(pollMs, 15000),
+ null,
+ );
// Apply a config change and mirror it into the address bar so the widget is
// self-describing: a reload re-parses the same config and the link stays
@@ -167,8 +180,12 @@ export function MonitorWidget({
// A low-disk warning is worth showing even when nothing is running, since it
// explains why no downloads start — so it overrides hideIdle. Interactive
- // controls also stay visible when idle (you may want to pause preemptively).
- if (config.hideIdle && nothingActive && !disk?.low && !config.controls) {
+ // controls also stay visible when idle (you may want to pause preemptively),
+ // as do the Sync button and the freshness/scheduler readouts — checking sync
+ // health is exactly what you do when nothing's active.
+ const keepWhenIdle =
+ config.controls || config.sync || config.lastSync || config.scheduler;
+ if (config.hideIdle && nothingActive && !disk?.low && !keepWhenIdle) {
return (
<div
className="relative p-2 text-xs text-muted-foreground"
@@ -183,12 +200,23 @@ export function MonitorWidget({
return (
<div className="relative flex flex-col gap-3 p-2 text-foreground">
{settingsLayer}
- {config.controls && (
+ {(config.controls || config.sync) && (
<WidgetControls
+ controls={config.controls}
+ sync={config.sync}
+ channel={config.channel}
+ confirmSyncAll={config.syncConfirm}
paused={workersPayload?.paused ?? false}
onWorkersChange={refetchWorkers}
+ onSynced={refetchSync}
/>
)}
+ {config.lastSync && syncData && (
+ <LastSyncStrip data={syncData} now={now} absolute={config.syncTimeAbsolute} />
+ )}
+ {config.scheduler && syncData && (
+ <SchedulerStrip data={syncData} now={now} absolute={config.syncTimeAbsolute} />
+ )}
{config.disk && disk?.enabled && <DiskStrip disk={disk} />}
{config.cleanable && cleanablePayload && (
<CleanableStrip bytes={cleanablePayload.bytes} />
@@ -329,6 +357,118 @@ function DiskStrip({ disk }: { disk: DiskStatusView }) {
);
}
+// Compact relative label. `deltaMs = now - target`: positive is past ("5m
+// ago"), negative is future ("in 5m"). Steps s → m → h → d.
+function relativeTime(deltaMs: number): string {
+ const past = deltaMs >= 0;
+ const s = Math.round(Math.abs(deltaMs) / 1000);
+ const val =
+ s < 60
+ ? `${s}s`
+ : s < 3600
+ ? `${Math.round(s / 60)}m`
+ : s < 86400
+ ? `${Math.round(s / 3600)}h`
+ : `${Math.round(s / 86400)}d`;
+ return past ? `${val} ago` : `in ${val}`;
+}
+
+// Format an epoch-ms marker per the widget's abstime flag. Absolute is
+// now-independent (locale string); relative depends on the live clock, so it
+// returns null until `now` is set (post-mount) to avoid an SSR/hydration
+// mismatch and a nonsensical "0s ago" flash. Callers render "…" for null.
+function fmtTime(
+ ms: number,
+ now: number | null,
+ absolute: boolean,
+): string | null {
+ if (absolute) return new Date(ms).toLocaleString();
+ if (now === null) return null;
+ return relativeTime(now - ms);
+}
+
+// Last-sync freshness readout: the last full "Sync all" sweep, plus the newest
+// individual channel sync when it is more recent than that sweep (a per-channel
+// Sync since the last global one).
+function LastSyncStrip({
+ data,
+ now,
+ absolute,
+}: {
+ data: WidgetSyncPayload;
+ now: number | null;
+ absolute: boolean;
+}) {
+ const full =
+ data.lastSyncAllAt === null
+ ? "never"
+ : (fmtTime(data.lastSyncAllAt, now, absolute) ?? "…");
+ const showChannel =
+ data.lastIndividualSyncAt !== null &&
+ (data.lastSyncAllAt === null ||
+ data.lastIndividualSyncAt > data.lastSyncAllAt);
+ return (
+ <section
+ aria-label="Last sync"
+ className="flex flex-col gap-0.5 rounded border border-border bg-card px-2 py-1 text-xs text-muted-foreground"
+ >
+ <span className="tabular-nums">Last full sync: {full}</span>
+ {showChannel && (
+ <span className="tabular-nums">
+ Last channel sync:{" "}
+ {fmtTime(data.lastIndividualSyncAt as number, now, absolute) ?? "…"}
+ </span>
+ )}
+ </section>
+ );
+}
+
+// Auto-sync scheduler status: on/off (colored dot), next due, last run. Dims to
+// a neutral line when the scheduler is disabled.
+function SchedulerStrip({
+ data,
+ now,
+ absolute,
+}: {
+ data: WidgetSyncPayload;
+ now: number | null;
+ absolute: boolean;
+}) {
+ const s = data.scheduler;
+ const nextText = s.overdue
+ ? "due now"
+ : s.nextRunAt !== null
+ ? (fmtTime(s.nextRunAt, now, absolute) ?? "…")
+ : "—";
+ const lastText =
+ s.lastRunAt !== null ? (fmtTime(s.lastRunAt, now, absolute) ?? "…") : "never";
+ return (
+ <section
+ aria-label="Auto-sync scheduler"
+ className={`flex items-center gap-2 rounded border px-2 py-1 text-xs ${
+ s.enabled
+ ? "border-border bg-card text-muted-foreground"
+ : "border-border bg-card text-muted-foreground/60"
+ }`}
+ >
+ <span
+ className={`inline-block h-2 w-2 shrink-0 rounded-full ${
+ s.enabled ? "bg-success" : "bg-muted-foreground/50"
+ }`}
+ />
+ <span className="tabular-nums">
+ Auto-sync {s.enabled ? "on" : "off"}
+ {s.enabled && (
+ <>
+ {" · "}next {nextText}
+ {" · "}last run {lastText}
+ </>
+ )}
+ </span>
+ </section>
+ );
+}
+
type CleanablePayload = { bytes: number };
function CleanableStrip({ bytes }: { bytes: number }) {
diff --git a/editor/app/widget/components/WidgetConfigForm.tsx b/editor/app/widget/components/WidgetConfigForm.tsx
@@ -94,6 +94,35 @@ export function WidgetConfigForm({
/>
</fieldset>
+ <fieldset className="flex flex-col gap-2">
+ <legend className="text-sm font-medium mb-1">Sync</legend>
+ <Check
+ label="Sync button"
+ checked={config.sync}
+ onChange={(v) => onChange({ sync: v })}
+ />
+ <Check
+ label="Confirm before Sync all"
+ checked={config.syncConfirm}
+ onChange={(v) => onChange({ syncConfirm: v })}
+ />
+ <Check
+ label="Last-sync readout"
+ checked={config.lastSync}
+ onChange={(v) => onChange({ lastSync: v })}
+ />
+ <Check
+ label="Scheduler status"
+ checked={config.scheduler}
+ onChange={(v) => onChange({ scheduler: v })}
+ />
+ <Check
+ label="Absolute timestamps"
+ checked={config.syncTimeAbsolute}
+ onChange={(v) => onChange({ syncTimeAbsolute: v })}
+ />
+ </fieldset>
+
<label className="flex flex-col gap-1 text-sm">
<span className="font-medium">Channel filter (slug, optional)</span>
<input
diff --git a/editor/app/widget/components/WidgetControls.tsx b/editor/app/widget/components/WidgetControls.tsx
@@ -1,30 +1,149 @@
"use client";
+import { useState } from "react";
import { PauseTranscriptionsButton } from "../../jobs/components/PauseTranscriptionsButton";
import { DrainAllButton } from "../../jobs/components/DrainAllButton";
import { RetryAllFailedButton } from "../../jobs/components/RetryAllFailedButton";
+import { syncAllChannelsAction, type SyncAllResult } from "../../channels/actions";
+import { syncAction } from "../../channels/[slug]/pipelineActions";
-// Opt-in interactive controls for the monitor widget (enabled with controls=1).
-// Reuses the same actions/buttons as the Workers and Active Jobs pages so a
-// pinned widget can free the GPU (pause), wind work down (drain), or recover
-// failures (retry) without opening the full app. `onWorkersChange` refetches the
-// widget's worker payload so the pause/resume label flips immediately instead of
-// waiting for the next poll.
+// Opt-in interactive controls for the monitor widget. Two independent
+// capabilities, each behind its own flag:
+// - `controls` → Pause/Resume + Drain + Retry (reused verbatim from the
+// Workers / Active Jobs pages), so a pinned widget can free the GPU, wind
+// work down, or recover failures without opening the full app.
+// - `sync` → a channel-aware Sync button: pinned to one channel it syncs just
+// that channel (draining the per-channel stream); otherwise it sweeps all
+// channels.
+// `onWorkersChange` refetches the worker payload so the pause/resume label flips
+// immediately; `onSynced` refetches the last-sync/scheduler readouts so they
+// update as soon as a sweep is queued.
export function WidgetControls({
+ controls,
+ sync,
+ channel,
+ confirmSyncAll,
paused,
onWorkersChange,
+ onSynced,
}: {
+ controls: boolean;
+ sync: boolean;
+ channel?: string;
+ confirmSyncAll: boolean;
paused: boolean;
onWorkersChange: () => void | Promise<void>;
+ onSynced: () => void | Promise<void>;
}) {
return (
<section
aria-label="Controls"
className="flex flex-wrap items-center gap-1.5"
>
- <PauseTranscriptionsButton paused={paused} onChange={onWorkersChange} />
- <DrainAllButton />
- <RetryAllFailedButton />
+ {controls && (
+ <>
+ <PauseTranscriptionsButton paused={paused} onChange={onWorkersChange} />
+ <DrainAllButton />
+ <RetryAllFailedButton />
+ </>
+ )}
+ {sync && (
+ <SyncButton
+ channel={channel}
+ confirmSyncAll={confirmSyncAll}
+ onSynced={onSynced}
+ />
+ )}
</section>
);
}
+
+// The Sync button. Channel-aware: with `channel` set it runs the streaming
+// per-channel sync (draining the stream to completion, exactly as
+// ChannelSyncButton); otherwise it runs the full Sync-all sweep (optionally
+// gated by a confirm) and briefly surfaces the queued/skipped counts.
+function SyncButton({
+ channel,
+ confirmSyncAll,
+ onSynced,
+}: {
+ channel?: string;
+ confirmSyncAll: boolean;
+ onSynced: () => void | Promise<void>;
+}) {
+ const [running, setRunning] = useState(false);
+ const [error, setError] = useState<string | null>(null);
+ const [result, setResult] = useState<SyncAllResult | null>(null);
+
+ async function handleClick() {
+ setError(null);
+ setResult(null);
+ if (channel) {
+ setRunning(true);
+ try {
+ const res = await syncAction(channel);
+ if (!res.ok) {
+ setError(res.error);
+ return;
+ }
+ // Drain the stream to completion; the job keeps running server-side.
+ const reader = res.stream.getReader();
+ while (true) {
+ const { done } = await reader.read();
+ if (done) break;
+ }
+ } catch (e) {
+ setError((e as Error).message);
+ } finally {
+ setRunning(false);
+ await onSynced();
+ }
+ return;
+ }
+ if (confirmSyncAll && !window.confirm("Sync all channels now?")) return;
+ setRunning(true);
+ try {
+ const res = await syncAllChannelsAction();
+ setResult(res);
+ } catch (e) {
+ setError((e as Error).message);
+ } finally {
+ setRunning(false);
+ await onSynced();
+ }
+ }
+
+ const label = channel ? "Sync" : "Sync all";
+ const ariaLabel = channel ? `sync ${channel}` : "sync all channels";
+ return (
+ <div className="flex items-center gap-2">
+ <button
+ type="button"
+ onClick={handleClick}
+ disabled={running}
+ aria-label={ariaLabel}
+ className="px-3 py-2 rounded-md bg-primary text-primary-foreground text-sm font-medium hover:opacity-90 disabled:opacity-50"
+ >
+ {running ? "Syncing…" : label}
+ </button>
+ {result && (
+ <span
+ aria-label="sync all result"
+ className="text-xs text-muted-foreground tabular-nums"
+ title={
+ result.skipped.length === 0
+ ? undefined
+ : result.skipped.map((s) => `${s.slug}: ${s.reason}`).join("\n")
+ }
+ >
+ Queued {result.queued.length} · skipped {result.skipped.length}
+ </span>
+ )}
+ {error && (
+ <span role="alert" className="text-xs text-destructive">
+ {error}
+ </span>
+ )}
+ </div>
+ );
+}
diff --git a/editor/app/widget/lib/config.ts b/editor/app/widget/lib/config.ts
@@ -42,6 +42,21 @@ export type WidgetConfig = {
// Show the in-widget settings gear that opens the config overlay in place. On
// by default; turn off to bake a locked-down shared/embedded link.
settings: boolean;
+ // Show the interactive Sync button. Channel-aware: syncs just `channel` when
+ // the widget is pinned to one, else sweeps all channels. Off by default.
+ sync: boolean;
+ // Show the last-sync readout strip (last full sync-all + last individual
+ // channel sync when newer). Off by default.
+ lastSync: boolean;
+ // Show the auto-sync scheduler status strip (on/off, next due, last run). Off
+ // by default.
+ scheduler: boolean;
+ // Readouts show absolute (locale) time instead of relative "5m ago". Off by
+ // default (relative).
+ syncTimeAbsolute: boolean;
+ // window.confirm before a full Sync-all sweep. Off by default. (A pinned
+ // single-channel Sync is never gated.)
+ syncConfirm: boolean;
};
export const WIDGET_DEFAULTS: WidgetConfig = {
@@ -61,6 +76,11 @@ export const WIDGET_DEFAULTS: WidgetConfig = {
controls: false,
actionable: false,
settings: true,
+ sync: false,
+ lastSync: false,
+ scheduler: false,
+ syncTimeAbsolute: false,
+ syncConfirm: false,
};
// Next's searchParams give each key as string | string[] | undefined.
@@ -106,6 +126,11 @@ export function parseWidgetConfig(params: RawParams): WidgetConfig {
controls: parseBool(params.controls, WIDGET_DEFAULTS.controls),
actionable: parseBool(params.act, WIDGET_DEFAULTS.actionable),
settings: parseBool(params.gear, WIDGET_DEFAULTS.settings),
+ sync: parseBool(params.sync, WIDGET_DEFAULTS.sync),
+ lastSync: parseBool(params.lastsync, WIDGET_DEFAULTS.lastSync),
+ scheduler: parseBool(params.sched, WIDGET_DEFAULTS.scheduler),
+ syncTimeAbsolute: parseBool(params.abstime, WIDGET_DEFAULTS.syncTimeAbsolute),
+ syncConfirm: parseBool(params.syncask, WIDGET_DEFAULTS.syncConfirm),
};
}
@@ -141,5 +166,15 @@ export function buildWidgetQuery(config: WidgetConfig): string {
sp.set("act", config.actionable ? "1" : "0");
if (config.settings !== WIDGET_DEFAULTS.settings)
sp.set("gear", config.settings ? "1" : "0");
+ if (config.sync !== WIDGET_DEFAULTS.sync)
+ sp.set("sync", config.sync ? "1" : "0");
+ if (config.lastSync !== WIDGET_DEFAULTS.lastSync)
+ sp.set("lastsync", config.lastSync ? "1" : "0");
+ if (config.scheduler !== WIDGET_DEFAULTS.scheduler)
+ sp.set("sched", config.scheduler ? "1" : "0");
+ if (config.syncTimeAbsolute !== WIDGET_DEFAULTS.syncTimeAbsolute)
+ sp.set("abstime", config.syncTimeAbsolute ? "1" : "0");
+ if (config.syncConfirm !== WIDGET_DEFAULTS.syncConfirm)
+ sp.set("syncask", config.syncConfirm ? "1" : "0");
return sp.toString();
}
diff --git a/editor/e2e/channel-site-membership.spec.ts b/editor/e2e/channel-site-membership.spec.ts
@@ -0,0 +1,186 @@
+import { test, expect, type Page } from "@playwright/test";
+import { readJson, resetData, writeSite } from "./helpers";
+
+// The channel form's "Sites" membership section: per-site checkbox + group
+// dropdown (+ inline "+ New group…") on both the create and edit screens.
+
+type SiteFile = {
+ groups: { id: string; name: string; selectedByDefault: boolean }[];
+ channels: { slug: string; groupId?: string; order?: number }[];
+};
+
+// Sentinel select value for "+ New group…" (SiteMembershipsSection.tsx).
+const NEW_GROUP = "__new__";
+
+// Two sites over the shared two-channel pool. Alpha has a second group and
+// slow-b sits in it with an explicit order; beta has just the default group.
+async function seed() {
+ await resetData("two-slow-channels");
+ await writeSite("alpha", {
+ siteTitle: "Alpha",
+ groups: [
+ { id: "default", name: "All channels", selectedByDefault: true },
+ { id: "news", name: "News", selectedByDefault: false },
+ ],
+ channels: [
+ { slug: "slow-a" },
+ { slug: "slow-b", groupId: "news", order: 5 },
+ ],
+ });
+ await writeSite("beta", { siteTitle: "Beta" });
+}
+
+const readSite = (id: string) =>
+ readJson<SiteFile>(`test-transcripts/sites/${id}/site.json`);
+
+// The membership inputs are controlled React state that the edit page also
+// server-renders, so a pre-hydration click can be silently swallowed (see the
+// charts-metadata lost-click incident). Both helpers verify each interaction
+// against React-rendered output (the group select / new-group input only
+// exist when state says so) and retry until it sticks.
+async function setSiteChecked(page: Page, site: string, on: boolean) {
+ const groupSelect = page.getByLabel(`Group for ${site}`);
+ await expect(async () => {
+ if (((await groupSelect.count()) > 0) !== on) {
+ await page.getByLabel(`Include on ${site}`).click();
+ }
+ await expect(groupSelect).toHaveCount(on ? 1 : 0, { timeout: 1_000 });
+ }).toPass({ timeout: 15_000 });
+}
+
+async function selectGroup(page: Page, site: string, value: string) {
+ const groupSelect = page.getByLabel(`Group for ${site}`);
+ const nameInput = page.getByLabel(`New group name for ${site}`);
+ // Selecting "+ New group…" reveals a React-rendered input — proof the change
+ // handler ran. Once that sticks, picking the real value is safe.
+ await expect(async () => {
+ await groupSelect.selectOption(NEW_GROUP);
+ await expect(nameInput).toBeVisible({ timeout: 1_000 });
+ }).toPass({ timeout: 15_000 });
+ if (value !== NEW_GROUP) {
+ await groupSelect.selectOption(value);
+ await expect(nameInput).toHaveCount(0);
+ }
+}
+
+test("create: active site pre-checked, explicit group + second site", async ({
+ page,
+}) => {
+ await seed();
+ await page.goto("/channels/new?site=alpha");
+ // The pre-check happens client-side only (mount effect), so this assertion
+ // doubles as a hydration gate for the plain interactions below.
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
+ await expect(page.getByLabel("Include on Beta")).not.toBeChecked();
+
+ await page.getByLabel("Group for Alpha").selectOption("news");
+ await page.getByLabel("Include on Beta").check();
+ await page.getByLabel(/^name/i).fill("Gamma Channel");
+ await page.getByLabel(/^slug/i).fill("gamma");
+ await page.getByRole("button", { name: /create channel/i }).click();
+ await page.waitForURL("**/channels/gamma", { timeout: 10_000 });
+
+ const alpha = await readSite("alpha");
+ expect(alpha.channels).toContainEqual({ slug: "gamma", groupId: "news" });
+ const beta = await readSite("beta");
+ expect(beta.channels).toContainEqual({ slug: "gamma" });
+});
+
+test("create: + New group creates the group and assigns the channel", async ({
+ page,
+}) => {
+ await seed();
+ await page.goto("/channels/new?site=alpha");
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
+
+ await selectGroup(page, "Alpha", NEW_GROUP);
+ await page.getByLabel("New group name for Alpha").fill("Interviews");
+ await page.getByLabel(/^name/i).fill("Delta Channel");
+ await page.getByLabel(/^slug/i).fill("delta");
+ await page.getByRole("button", { name: /create channel/i }).click();
+ await page.waitForURL("**/channels/delta", { timeout: 10_000 });
+
+ const alpha = await readSite("alpha");
+ expect(alpha.groups).toContainEqual({
+ id: "interviews",
+ name: "Interviews",
+ selectedByDefault: false,
+ });
+ expect(alpha.channels).toContainEqual({
+ slug: "delta",
+ groupId: "interviews",
+ });
+});
+
+test("edit: switching news → (default) keeps order, sibling untouched", async ({
+ page,
+}) => {
+ await seed();
+ await page.goto("/channels/slow-b");
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
+ await expect(page.getByLabel("Group for Alpha")).toHaveValue("news");
+
+ await selectGroup(page, "Alpha", "");
+ await page.getByRole("button", { name: /save changes/i }).click();
+
+ await expect
+ .poll(async () =>
+ (await readSite("alpha")).channels.find((c) => c.slug === "slow-b"),
+ )
+ .toEqual({ slug: "slow-b", order: 5 });
+ const alpha = await readSite("alpha");
+ expect(alpha.channels).toContainEqual({ slug: "slow-a" });
+});
+
+test("edit: unchecking removes the membership, sibling intact", async ({
+ page,
+}) => {
+ await seed();
+ await page.goto("/channels/slow-b");
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
+
+ await setSiteChecked(page, "Alpha", false);
+ await page.getByRole("button", { name: /save changes/i }).click();
+
+ await expect
+ .poll(async () => (await readSite("alpha")).channels.map((c) => c.slug))
+ .toEqual(["slow-a"]);
+});
+
+test("edit: + New group on another site", async ({ page }) => {
+ await seed();
+ await page.goto("/channels/slow-a");
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
+ await expect(page.getByLabel("Include on Beta")).not.toBeChecked();
+
+ await setSiteChecked(page, "Beta", true);
+ await selectGroup(page, "Beta", NEW_GROUP);
+ await page.getByLabel("New group name for Beta").fill("Interviews");
+ await page.getByRole("button", { name: /save changes/i }).click();
+
+ await expect
+ .poll(async () => (await readSite("beta")).groups)
+ .toContainEqual({
+ id: "interviews",
+ name: "Interviews",
+ selectedByDefault: false,
+ });
+ const beta = await readSite("beta");
+ expect(beta.channels).toContainEqual({
+ slug: "slow-a",
+ groupId: "interviews",
+ });
+});
+
+test("zero sites: informational note, create still succeeds", async ({
+ page,
+}) => {
+ await resetData("empty");
+ await page.goto("/channels/new");
+ await expect(page.getByText(/no sites configured/i)).toBeVisible();
+
+ await page.getByLabel(/^name/i).fill("Solo Channel");
+ await page.getByLabel(/^slug/i).fill("solo");
+ await page.getByRole("button", { name: /create channel/i }).click();
+ await page.waitForURL("**/channels/solo", { timeout: 10_000 });
+});
diff --git a/editor/e2e/channels.spec.ts b/editor/e2e/channels.spec.ts
@@ -74,12 +74,14 @@ test("requires typed-confirmation to delete", async ({ page }) => {
await resetData("one-youtube-channel");
await page.goto("/channels/test-youtube");
- await page.getByPlaceholder("test-youtube").fill("wrong-slug");
+ // getByPlaceholder would be ambiguous: the rename form's confirm input
+ // shares the slug placeholder.
+ await page.getByLabel("confirm slug to delete").fill("wrong-slug");
await page.getByRole("button", { name: /delete channel/i }).click();
await expect(page.getByText(/type the channel slug/i)).toBeVisible();
expect(await pathExists("test-transcripts/channels/test-youtube")).toBe(true);
- await page.getByPlaceholder("test-youtube").fill("test-youtube");
+ await page.getByLabel("confirm slug to delete").fill("test-youtube");
await page.getByRole("button", { name: /delete channel/i }).click();
await page.waitForURL("**/channels", { timeout: 10_000 });
expect(await pathExists("test-transcripts/channels/test-youtube")).toBe(false);
diff --git a/editor/e2e/cookies-mode.spec.ts b/editor/e2e/cookies-mode.spec.ts
@@ -0,0 +1,331 @@
+import { mkdir, readFile, writeFile } from "node:fs/promises";
+import { fileURLToPath } from "node:url";
+import path from "node:path";
+import { test, expect } from "@playwright/test";
+import {
+ pathExists,
+ readJson,
+ resetData,
+ resolvePath,
+ writeSettings,
+} from "./helpers";
+
+// Configurable cookies-from-browser: global + per-channel value, and a cookie
+// MODE (always / when-required / defer). The fake yt-dlp's `cookiegated`
+// sentinel fails any metadata/download invocation without
+// --cookies-from-browser (yt-dlp's age-gate error) and succeeds with it, and
+// every relevant invocation line logs `cookies=<value>` so flag presence can
+// be asserted per spawn.
+
+const CHANNEL = "test-cookies";
+const FIXTURE = "cookies-mode-channel";
+const ROOT = `test-transcripts/channels/${CHANNEL}`;
+const COOKIES = "firefox:test";
+const here = path.dirname(fileURLToPath(import.meta.url));
+
+type DownloadOutcome = {
+ status: string;
+ attempts: Array<{ kind: string; usedCookies: boolean }>;
+};
+
+type Snapshot = {
+ buckets?: { needsCookies?: string[] };
+};
+
+async function defaultSettings(): Promise<Record<string, unknown>> {
+ const raw = await readFile(
+ path.join(here, "fixtures", "test-settings.default.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as Record<string, unknown>;
+}
+
+async function readInvocations(): Promise<string> {
+ try {
+ return await readFile(
+ path.join(here, "..", ROOT, "fake-ytdlp.invocations"),
+ "utf8",
+ );
+ } catch {
+ return "";
+ }
+}
+
+// The download job arms a debounced (~1s) snapshot regen; poll snapshot.json
+// until the needsCookies bucket reaches the expected size before reloading.
+async function waitForNeedsCookiesBucket(size: number): Promise<void> {
+ await expect
+ .poll(
+ async () => {
+ const snap = await readJson<Snapshot>(`${ROOT}/snapshot.json`).catch(
+ () => null,
+ );
+ return snap?.buckets?.needsCookies?.length ?? -1;
+ },
+ { timeout: 15_000 },
+ )
+ .toBe(size);
+}
+
+test("settings: cookie value + mode round-trip through the form", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await page.goto("/settings");
+ await page.getByLabel(/^cookies from browser/i).fill(COOKIES);
+ // The value field's hint mentions "cookie mode", so a label lookup is
+ // ambiguous — target the select by name.
+ const modeSelect = page.locator('select[name="cookieMode"]');
+ await modeSelect.selectOption("always");
+ await page.getByRole("button", { name: /save settings/i }).click();
+ await expect(
+ page.getByRole("status").filter({ hasText: "Saved" }),
+ ).toBeVisible();
+
+ const saved = await readJson<{
+ cookiesFromBrowser?: string;
+ cookieMode?: string;
+ }>("test-settings.json");
+ expect(saved.cookiesFromBrowser).toBe(COOKIES);
+ expect(saved.cookieMode).toBe("always");
+
+ await page.reload();
+ await expect(page.getByLabel(/^cookies from browser/i)).toHaveValue(COOKIES);
+ await expect(modeSelect).toHaveValue("always");
+});
+
+test("channel form: overrides persist and clear back to inherit", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ await page.getByLabel(/^cookies from browser/i).fill("chrome:Profile 1");
+ await page.getByLabel("cookie mode").selectOption("defer");
+ await page.getByRole("button", { name: /save changes/i }).click();
+
+ await expect
+ .poll(async () =>
+ readJson<{ cookiesFromBrowser?: string; cookieMode?: string }>(
+ `${ROOT}/config.json`,
+ ),
+ )
+ .toMatchObject({
+ cookiesFromBrowser: "chrome:Profile 1",
+ cookieMode: "defer",
+ });
+
+ // Clear both back to inherit. CHANNEL_FORM_FIELDS layering must actually
+ // remove the stored values, not re-layer the stale ones.
+ await page.reload();
+ await page.locator("summary").filter({ hasText: "Advanced" }).click();
+ const cookiesField = page.getByLabel(/^cookies from browser/i);
+ await expect(cookiesField).toHaveValue("chrome:Profile 1");
+ await cookiesField.fill("");
+ await page.getByLabel("cookie mode").selectOption("");
+ await page.getByRole("button", { name: /save changes/i }).click();
+
+ await expect
+ .poll(async () => {
+ const config = await readJson<Record<string, unknown>>(
+ `${ROOT}/config.json`,
+ );
+ return {
+ cookiesFromBrowser: config.cookiesFromBrowser,
+ cookieMode: config.cookieMode,
+ };
+ })
+ .toEqual({ cookiesFromBrowser: undefined, cookieMode: undefined });
+});
+
+test("always mode: every yt-dlp invocation carries the cookie value", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData(FIXTURE);
+ await writeSettings({
+ ...(await defaultSettings()),
+ cookiesFromBrowser: COOKIES,
+ cookieMode: "always",
+ });
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ "Managed download complete: 2 succeeded, 0 failed",
+ { timeout: 60_000 },
+ );
+
+ const invocations = await readInvocations();
+ expect(invocations).toMatch(
+ new RegExp(`prefetch:.*vidcookiegated1 cookies=${COOKIES}`),
+ );
+ expect(invocations).toMatch(
+ new RegExp(`prefetch:.*normalvideo1 cookies=${COOKIES}`),
+ );
+ expect(invocations).toMatch(
+ new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`),
+ );
+ expect(invocations).toMatch(
+ new RegExp(`youtube-single:.*normalvideo1 cookies=${COOKIES}`),
+ );
+ // No cookie-less invocation lines at all in always mode.
+ expect(invocations).not.toMatch(/cookies=$/m);
+
+ // Prophylactic cookies on a clean success stay "ok" — "ok-with-cookies"
+ // keeps meaning "cookies were needed".
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/vidcookiegated1/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("ok");
+ expect(outcome.attempts.every((a) => a.usedCookies)).toBe(true);
+});
+
+test("when-required: failed prefetch is retried with cookies -> ok-with-cookies", async ({
+ page,
+}) => {
+ test.setTimeout(120_000);
+ await resetData(FIXTURE);
+ await writeSettings({
+ ...(await defaultSettings()),
+ cookiesFromBrowser: COOKIES,
+ // cookieMode omitted -> back-compat default "when-required"
+ });
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ `Metadata prefetch auth-required (needs_auth); retrying with --cookies-from-browser ${COOKIES}`,
+ { timeout: 60_000 },
+ );
+ await expect(log).toContainText(
+ "Managed download complete: 2 succeeded, 0 failed",
+ { timeout: 60_000 },
+ );
+
+ // Two prefetch lines for the gated video: cookie-less first, cookies second.
+ const invocations = await readInvocations();
+ const gatedPrefetches = invocations
+ .split("\n")
+ .filter((l) => l.startsWith("prefetch:") && l.includes("vidcookiegated1"));
+ expect(gatedPrefetches).toHaveLength(2);
+ expect(gatedPrefetches[0]).toMatch(/cookies=$/);
+ expect(gatedPrefetches[1]).toContain(`cookies=${COOKIES}`);
+ // The recovered prefetch hands cookies straight to the real download.
+ expect(invocations).toMatch(
+ new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`),
+ );
+ // The normal video never sees cookies in when-required mode.
+ const normalLines = invocations
+ .split("\n")
+ .filter((l) => l.includes("normalvideo1") && l.includes("cookies="));
+ expect(normalLines.length).toBeGreaterThan(0);
+ expect(normalLines.every((l) => l.endsWith("cookies="))).toBe(true);
+
+ const outcome = await readJson<DownloadOutcome>(
+ `${ROOT}/data/vidcookiegated1/download-outcome.json`,
+ );
+ expect(outcome.status).toBe("ok-with-cookies");
+ expect(
+ outcome.attempts.some(
+ (a) => a.kind === "metadata-prefetch-auth-retry" && a.usedCookies,
+ ),
+ ).toBe(true);
+});
+
+test("defer: excluded from batches, surfaced in Needs cookies, downloads via the bucket button", async ({
+ page,
+}) => {
+ test.setTimeout(180_000);
+ await resetData(FIXTURE);
+ await writeSettings({
+ ...(await defaultSettings()),
+ cookiesFromBrowser: COOKIES,
+ cookieMode: "defer",
+ });
+
+ // Run 1: the gated video fails cookie-less (no prefetch retry, no auth
+ // retry — defer never uses cookies in normal runs).
+ await page.goto(`/channels/${CHANNEL}`);
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log = page.getByLabel("Download videos output");
+ await expect(log).toContainText(
+ "Managed download complete: 1 succeeded, 1 failed",
+ { timeout: 60_000 },
+ );
+ let invocations = await readInvocations();
+ expect(invocations).not.toContain(`cookies=${COOKIES}`);
+ expect(
+ invocations
+ .split("\n")
+ .filter((l) => l.startsWith("prefetch:") && l.includes("vidcookiegated1")),
+ ).toHaveLength(1);
+
+ // The failure lands the video in the snapshot's needsCookies bucket.
+ await waitForNeedsCookiesBucket(1);
+
+ // Run 2: the recorded needs_auth video is excluded, not re-attempted.
+ await page.reload();
+ await page.getByRole("button", { name: "Download videos" }).click();
+ const log2 = page.getByLabel("Download videos output");
+ await expect(log2).toContainText("needs_auth deferred=1", {
+ timeout: 60_000,
+ });
+ await expect(log2).toContainText("Nothing to fetch.", { timeout: 60_000 });
+
+ // The Needs-cookies card offers the manual cookie run.
+ const bucket = page.getByLabel("retry needs cookies bucket");
+ await expect(bucket).toBeVisible();
+ await bucket
+ .getByRole("button", { name: /^Download with cookies \(1\)$/ })
+ .click();
+ const retryLog = page.getByLabel("Retry needs cookies output");
+ await expect(retryLog).toContainText(
+ "Managed download complete: 1 succeeded, 0 failed",
+ { timeout: 60_000 },
+ );
+
+ expect(
+ await pathExists(`${ROOT}/data/vidcookiegated1/transcript.en.vtt`),
+ ).toBe(true);
+ invocations = await readInvocations();
+ expect(invocations).toMatch(
+ new RegExp(`youtube-single:.*vidcookiegated1 cookies=${COOKIES}`),
+ );
+
+ // The bucket empties once the snapshot regenerates.
+ await waitForNeedsCookiesBucket(0);
+ await page.reload();
+ await expect(page.getByLabel("retry needs cookies bucket")).toHaveCount(0);
+});
+
+test("members_only videos surface in Needs cookies regardless of mode", async ({
+ page,
+}) => {
+ await resetData(FIXTURE);
+ // Default settings: when-required mode. A members-only playlist video with
+ // no downloaded artifact must still land in the bucket — cookies from a
+ // subscribed account could recover it even though batches exclude it.
+ await writeFile(
+ resolvePath(`${ROOT}/playlist`),
+ "https://www.youtube.com/watch?v=vidmembers1\n",
+ );
+ const videoDir = resolvePath(`${ROOT}/data/vidmembers1`);
+ await mkdir(videoDir, { recursive: true });
+ await writeFile(
+ path.join(videoDir, "availability.json"),
+ JSON.stringify({
+ checkedAt: new Date().toISOString(),
+ availability: "members_only",
+ webpageUrl: "https://www.youtube.com/watch?v=vidmembers1",
+ }),
+ );
+
+ await page.goto(`/channels/${CHANNEL}`);
+ const bucket = page.getByLabel("retry needs cookies bucket");
+ await expect(bucket).toBeVisible();
+ await expect(
+ bucket.getByRole("button", { name: /^Download with cookies \(1\)$/ }),
+ ).toBeVisible();
+ await expect(page.getByLabel("needs cookies vidmembers1")).toBeVisible();
+});
diff --git a/editor/e2e/fixtures/bin/fake-ytdlp.mjs b/editor/e2e/fixtures/bin/fake-ytdlp.mjs
@@ -46,6 +46,27 @@ function lastNonFlag() {
return undefined;
}
+// The `cookiegated` sentinel simulates an age-gated video: any metadata/
+// download invocation for it fails with yt-dlp's age-gate error UNLESS
+// --cookies-from-browser was passed. (Deliberately distinct from the
+// `needsauth` sentinel, which only affects --dump-json probes and still
+// downloads fine — retry-bucket.spec depends on that.)
+function cookieArg() {
+ return arg("--cookies-from-browser") ?? "";
+}
+
+function cookieGateBlocked(url) {
+ const lower = (url ?? "").toLowerCase();
+ return lower.includes("cookiegated") && !cookieArg();
+}
+
+function failCookieGate(url) {
+ process.stderr.write(
+ `ERROR: [youtube] ${url}: Sign in to confirm your age. This video may be inappropriate for some users.\n`,
+ );
+ process.exit(1);
+}
+
function urlIdYouTube(u) {
try {
const parsed = new URL(u);
@@ -214,7 +235,11 @@ async function modeDownloadFromFile(file, opts = {}) {
async function modeDownloadOneUrl(url, opts) {
process.stdout.write(`[fake-ytdlp] single-url download ${url}\n`);
- await appendFile("fake-ytdlp.invocations", `download-one:${url}\n`);
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `download-one:${url} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
// Record the resolved -f selector so format-selection tests can assert it
// (e.g. Odysee "auto" -> original/bestaudio/worst).
await appendFile("fake-ytdlp.invocations", `format:${arg("-f") ?? ""}\n`);
@@ -409,8 +434,9 @@ async function modeYoutubeSingleUrlManaged(url) {
await ensureDir(videoDir);
await appendFile(
"fake-ytdlp.invocations",
- `youtube-single:${url}\n`,
+ `youtube-single:${url} cookies=${cookieArg()}\n`,
);
+ if (cookieGateBlocked(url)) failCookieGate(url);
process.stdout.write(`[fake-ytdlp] managed single-URL ${id}\n`);
// URL sentinels for no-subs-fallback tests. The sentinels live in the
@@ -475,7 +501,11 @@ async function main() {
if (has("--dump-json") && has("--skip-download")) {
const url = lastNonFlag() ?? "";
const lower = url.toLowerCase();
- await appendFile("fake-ytdlp.invocations", `dump-json:${url}\n`);
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `dump-json:${url} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
if (lower.includes("deleted")) {
process.stderr.write(`ERROR: [youtube] ${url}: Video unavailable\n`);
process.exit(1);
@@ -587,7 +617,11 @@ async function main() {
const id = urlIdYouTube(url);
const videoDir = path.join("data", id);
await ensureDir(videoDir);
- await appendFile("fake-ytdlp.invocations", `prefetch:${url}\n`);
+ await appendFile(
+ "fake-ytdlp.invocations",
+ `prefetch:${url} cookies=${cookieArg()}\n`,
+ );
+ if (cookieGateBlocked(url)) failCookieGate(url);
await writeMetadata(videoDir, id, urlSentinels(url));
process.stdout.write(`[fake-ytdlp] metadata prefetch ${id}\n`);
return;
diff --git a/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/config.json b/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/config.json
@@ -0,0 +1,5 @@
+{
+ "handling": "youtube",
+ "name": "Test Cookies",
+ "url": "https://www.youtube.com/@example/videos"
+}
diff --git a/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/playlist b/editor/e2e/fixtures/test-transcripts/cookies-mode-channel/channels/test-cookies/playlist
@@ -0,0 +1,2 @@
+https://www.youtube.com/watch?v=vidcookiegated1
+https://www.youtube.com/watch?v=normalvideo1
diff --git a/editor/e2e/site-scope.spec.ts b/editor/e2e/site-scope.spec.ts
@@ -76,8 +76,9 @@ test("deploy targets the active site and is disabled under all sites", async ({
await expect(
page.getByText(/select a specific site from the sidebar to build & deploy\./i),
).toBeVisible();
+ // exact: the "Build & deploy all sites" batch button also matches otherwise.
await expect(
- page.getByRole("button", { name: "Build & deploy" }),
+ page.getByRole("button", { name: "Build & deploy", exact: true }),
).toBeDisabled();
await page.goto("/deploy?site=alpha");
@@ -104,6 +105,9 @@ test("creating a channel under a site adds it to that site's membership", async
}) => {
await twoSites();
await page.goto("/channels/new?site=alpha");
+ // The Sites section pre-checks the active site — that's what carries the
+ // membership now (the old hidden activeSite field is gone).
+ await expect(page.getByLabel("Include on Alpha")).toBeChecked();
await page.getByLabel(/^name/i).fill("Gamma Channel");
await page.getByLabel(/^slug/i).fill("gamma");
await page.getByLabel(/^url/i).fill("https://www.youtube.com/@gamma/videos");
diff --git a/editor/e2e/sites-crud.spec.ts b/editor/e2e/sites-crud.spec.ts
@@ -241,6 +241,55 @@ test("edit page loads an existing public URL and featured site", async ({
).toBeChecked();
});
+test("inline-chips flag round-trips; seeded default group has it on", async ({
+ page,
+}) => {
+ await resetData("empty");
+
+ await page.goto("/sites/new");
+ await page.getByLabel(/site id/i).fill("chipsite");
+ await page.getByLabel(/site title/i).fill("Chip Site");
+ await page.getByLabel(/header title/i).fill("Chip Site");
+
+ const chipToggles = page.getByRole("checkbox", {
+ name: /show channels as individual chips/i,
+ });
+ // The seeded default group (FALLBACK_GROUP) ships with the flag on.
+ await expect(chipToggles).toHaveCount(1);
+ await expect(chipToggles.first()).toBeChecked();
+
+ // Manually added groups default off.
+ await page.getByRole("button", { name: /\+ add group/i }).click();
+ await expect(chipToggles).toHaveCount(2);
+ await expect(chipToggles.nth(1)).not.toBeChecked();
+ await page.getByLabel("ID", { exact: true }).nth(1).fill("extra");
+ await page.getByLabel("Name", { exact: true }).nth(1).fill("Extra");
+
+ await page.getByRole("button", { name: /create site/i }).click();
+ await expect(
+ page.getByRole("status").filter({ hasText: "Saved" }),
+ ).toBeVisible();
+
+ type GroupsSiteFile = {
+ groups: { id: string; inline?: boolean }[];
+ };
+ // Poll: the "Saved" status and the on-disk write can land slightly apart.
+ await expect(async () => {
+ const site = await readJson<GroupsSiteFile>(
+ "test-transcripts/sites/chipsite/site.json",
+ );
+ expect(site.groups[0].inline).toBe(true);
+ // Flag-off groups serialize without the key at all.
+ expect("inline" in site.groups[1]).toBe(false);
+ }).toPass({ timeout: 10_000 });
+
+ // Reload the edit page — checkbox states restore from disk.
+ await page.goto("/sites/chipsite");
+ await expect(chipToggles).toHaveCount(2);
+ await expect(chipToggles.first()).toBeChecked();
+ await expect(chipToggles.nth(1)).not.toBeChecked();
+});
+
test("first-run migrate button appears only when no sites exist", async ({
page,
}) => {
diff --git a/editor/e2e/widget.spec.ts b/editor/e2e/widget.spec.ts
@@ -99,6 +99,42 @@ test("controls=1 adds interactive Pause/Resume, Drain, and Retry buttons", async
).toBeVisible({ timeout: 10_000 });
});
+test("sync=1 shows a channel-aware Sync button", async ({ page }) => {
+ // No channel pinned → sweeps all channels.
+ await page.goto("/widget?sync=1");
+ const syncAll = page.getByRole("button", { name: "sync all channels" });
+ await expect(syncAll).toBeVisible();
+ await expect(syncAll).toHaveText("Sync all");
+
+ // Empty corpus → the sweep queues nothing and reports its counts.
+ await syncAll.click();
+ await expect(page.getByLabel("sync all result")).toContainText(
+ "Queued 0 · skipped 0",
+ { timeout: 10_000 },
+ );
+
+ // Pinned to one channel → the button targets just that channel.
+ await page.goto("/widget?sync=1&channel=test-youtube");
+ const syncOne = page.getByRole("button", { name: "sync test-youtube" });
+ await expect(syncOne).toBeVisible();
+ await expect(syncOne).toHaveText("Sync");
+});
+
+test("lastsync=1 shows the last-sync readout", async ({ page }) => {
+ await page.goto("/widget?lastsync=1");
+ const strip = page.getByRole("region", { name: "Last sync" });
+ await expect(strip).toBeVisible({ timeout: 10_000 });
+ // Empty corpus with no sweep recorded → never.
+ await expect(strip).toContainText("Last full sync: never");
+});
+
+test("sched=1 shows the auto-sync scheduler strip", async ({ page }) => {
+ await page.goto("/widget?sched=1");
+ const strip = page.getByRole("region", { name: "Auto-sync scheduler" });
+ await expect(strip).toBeVisible({ timeout: 10_000 });
+ await expect(strip).toContainText(/Auto-sync (on|off)/);
+});
+
test("section params gate what renders", async ({ page }) => {
await page.goto("/widget?jobs=0");
await expect(page.getByRole("heading", { name: "Workers" })).toBeVisible();
@@ -210,4 +246,21 @@ test("display toggles serialize into the link and preview", async ({ page }) =>
.check();
await expect(url).toHaveValue(/[?&]controls=1(&|$)/);
await expect(preview).toHaveAttribute("src", /[?&]controls=1(&|$)/);
+
+ // The Sync fieldset flags each serialize under their own key.
+ await page.getByRole("checkbox", { name: "Sync button" }).check();
+ await expect(url).toHaveValue(/[?&]sync=1(&|$)/);
+ await expect(preview).toHaveAttribute("src", /[?&]sync=1(&|$)/);
+
+ await page.getByRole("checkbox", { name: "Confirm before Sync all" }).check();
+ await expect(url).toHaveValue(/[?&]syncask=1(&|$)/);
+
+ await page.getByRole("checkbox", { name: "Last-sync readout" }).check();
+ await expect(url).toHaveValue(/[?&]lastsync=1(&|$)/);
+
+ await page.getByRole("checkbox", { name: "Scheduler status" }).check();
+ await expect(url).toHaveValue(/[?&]sched=1(&|$)/);
+
+ await page.getByRole("checkbox", { name: "Absolute timestamps" }).check();
+ await expect(url).toHaveValue(/[?&]abstime=1(&|$)/);
});
diff --git a/export/CHANGELOG.md b/export/CHANGELOG.md
@@ -2,6 +2,12 @@
## [Unreleased]
- **"Ask AI" — the Report panel's citations are now clickable, precise to the exact moment cited.** The chat's answer bubbles already linked their `[1]`, `[2]`… markers to the cited transcript; the **Report** panel — the persistent document Report mode and whole-corpus sweeps maintain — rendered as plain text, so the citations a user works from most were dead. Now the report mirrors the chat: it renders each citation as an in-page link and lists its **numbered sources** beneath the document, and clicking one **opens the transcript modal seeked to the cited line**. Citations are also upgraded — in the report *and* the chat — to a precise-moment **`[n @ mm:ss]`** form (the source's number plus the specific line's timestamp, e.g. `[3 @ 12:34]`), so a citation jumps to the exact moment instead of the video's first matched snippet. A slightly-off model timestamp is **snapped to the nearest real transcript line**, and it stays graceful: a bare `[n]` still resolves to the first snippet, and an out-of-range number or unparseable time stays plain text rather than becoming a broken link. Making this work in the report (a single accumulated string, unlike a chat message that carries its own sources) required two new pieces: a **persisted report source registry** — every video the report cites, deduped and stored light (no excerpt text) — and **global citation numbering** so a report's `[n]` is stable across every batch/turn folded in, not renumbered per write. The registry is saved with the conversation and each saved chat, so a report's clickable citations survive a reload; New chat clears it; and on the federated hub /ask (no player) citations fall back to scroll-to-source, exactly like the chat. See `export/app/ask/citations.tsx` (new shared renderer — `linkifyCitations` + `CitationLink` + `CitationSources`, with `[n @ mm:ss]` parsing and nearest-snippet snapping), `export/app/ask/{MessageBubble,ReportPanel,AskChat,useAskChat,askChatStorage}.tsx/ts`, `export/app/lib/{askRetrieval,askConversation,searchAgent}.ts` (optional/global-registry numbering + `mergeReportSources` + the `[n @ mm:ss]` prompt updates), and `export/e2e/ask-chat.spec.ts`.
+- **Loose channel chips for flagged groups, plus uniform chip layout.** A channel group can now be marked `inline` (site.json → manifest), which renders each of its member channels as its own loose chip in the filter row — a checkbox + name styled like a collapsed group chip, with no expand affordance — flowing in sort order among the normal group chips; inline groups are also skipped by the multi-group default-collapse heuristic, so their channels are always visible. The out-of-the-box **"All channels" bucket now behaves this way**: a site with no configured groups switches from one open full-width card to loose per-channel chips (behavior change; add an explicit group to keep the old card). The chips themselves also get a layout fix: instead of shrink-to-fit chips that could line-break their own text, all collapsed chips share a **uniform min width** (`sm:min-w-48`), never wrap internally — long names ellipsize with the full name as a tooltip — counts right-align inside the chip, and below the `sm` breakpoint chips stack one per line at full width while larger screens pack them into rows. See `common/lib/channelGroups.ts` (`ChannelGroup.inline`), `common/components/{WorkspaceSearchBar,SearchSessionContext}.tsx`, and `export/e2e/inline-channel-chips.spec.ts`.
+
+## [0.8.1] - 2026-07-22
+- **Channel-group filters are now compact chips.** On a site with many small channel groups the filter panel used to stack every group as its own full-width card, so ten groups meant ten rows before you saw a single channel. Collapsed groups now shrink to labeled chips — a chevron, a tri-state checkbox (checked / mixed / unchecked), the group's accent dot + name, and a `selected/total` count — that wrap several per line. The chip's checkbox toggles the **whole group** on or off in one click *without* expanding it (a mixed group's click selects all), and clicking the name expands the group into the familiar full-width card with per-channel checkboxes and All/None. Groups start collapsed on multi-group sites (your own expand/collapse choices still stick); a collapsed chip shows the group's description as a tooltip. Selection stays flat per-channel — profiles, share links, and saved filters are unchanged. See `common/components/{WorkspaceSearchBar,SearchSessionContext}.tsx`, `common/components/ui/checkbox.tsx`, and `export/e2e/channel-group-chips.spec.ts`.
+
+## [0.8.0] - 2026-07-20
- **MCP server: switch which corpus you're reading on the fly — and it sticks across reconnects.** The MCP server was pinned to a single corpus at launch (`--hub` / `--remote` / `--local` or the `TRANSCRIPT_*` env), so re-aiming it at a different site or a local build meant editing the MCP config and restarting. Now, connected to (say) the **archilyzer hub**, you can retarget it from inside a session with three new **read-only** tools: **`list_sources`** shows the active corpus and, in a hub context, its member sites (`siteId · title · url`); **`use_source`** switches the active corpus — to a single hub member (`site:` — which becomes a plain single-site source and so regains **full channel-group + alias** support), a **federated subset** of the hub (`sites:[…]`, unknown tokens reported not dropped), or an arbitrary `remote:` URL / `local:` dir / `hub:` URL; and **`reset_source`** returns to the startup source. The selection is **persisted** to a small state file (under `$XDG_STATE_HOME/yt-dlp-transcript-mcp`, overridable with `TRANSCRIPT_MCP_STATE_DIR`) keyed by the **startup** source, so it survives a reconnect and two differently-configured servers keep independent selections. Everything the sweep/search tools do reads the current active source. Still strictly **read-only** — this only changes *which* already-published static shards are read, the same capability the startup flags already grant this locally-run tool; nothing is written to any corpus. See `mcp/src/{sources,sourceController,source,server,index}.ts`, `mcp/src/sourceController.test.ts`, and `mcp/README.md`.
- **MCP server: run the corpus sweep through Claude Code itself — on plan usage, no API key.** The in-browser "corpus sweep" (fold every matching transcript into a running report) needs a BYO AI key, and a free key conks out fast on a real 30k-video corpus. The `mcp/` server now lets **Claude Code be the sweep engine** instead, via a first-class **`sweep` prompt** (a slash command, `/mcp__<name>__sweep query="k cups" channel="chrissie-mayr"`) plus the tools to drive it. `search_transcripts` gains **paging** — it reports the full `total` and `has_more`, so the whole match set can be enumerated with `offset` (and `include_snippets:false` for a cheap worklist) — and is now **alias-aware**: a plain query that matches a curated search alias also searches the alias regex (e.g. `k cups` → also `cake cup`), with the footer naming which aliases fired; reaching the scan cap is surfaced as **PARTIAL** coverage rather than hidden. A new **`get_transcripts`** tool batch-reads up to 20 videos in one call as bounded, timestamped **excerpt windows** around the matches (alias-correct) — or full transcripts without a query — so a sweep stays token-bounded. The `sweep` prompt walks Claude through search → enumerate → plan `ceil(N/batch)` batches → per-batch windowed read + cross-referenced upsert of cited findings (*title + [mm:ss]*) into a markdown report it maintains with its own Write/Edit tools. The sweep is now **group-aware and multi-channel**: `search_transcripts` scope is additive over one-or-more channels (`channel`/`channels`) and/or channel **groups** (`group`/`groups`, matched by group **id or display name** — `group="other"` ≡ `group="Extended Universe"` — and expanded to the group's channels), with the footer naming the resolved scope and flagging any channel/group token that matched nothing so a typo isn't silently a whole-corpus scan; `get_transcripts` takes matching `channels` lookup hints; and `list_channels` now organizes channels under their groups with a compact `id · name · N channels` cheat-sheet. Crucially the `sweep` prompt is now **pick-first**: invoked with **no** scope it makes Claude list the groups/channels and **ask which to sweep — or to confirm the whole corpus — before enumerating**, rather than silently scanning everything (an explicit `search_transcripts` call with no selector still means "all"). (Hub mode defers per-site group resolution — multi-channel scoping still works there.) The MCP stays strictly **read-only**; only the report file is written, in Claude's own working directory. See `mcp/src/{search,source,server}.ts`, `mcp/src/search.test.ts`, and `mcp/README.md`.
- **"Ask AI" now paces itself to your key's rate limit instead of failing.** A whole-corpus sweep on a free-tier key used to fire provider calls as fast as the loop could produce them, blow straight through the per-minute request cap, and stop dead on the first HTTP 429 (and a rate-limit *mid-answer* surfaced as a hard error, because only the sweep path ever handled 429). Now a single client-side limiter sits behind **every** AI call: it **spaces requests** to a conservative, free-tier-safe **requests-per-minute** default chosen per model (e.g. Gemini `*-pro` → 5/min, `*flash` → 10/min; Claude/OpenAI higher), so a paid key runs fast and a free key just runs *slowly* rather than erroring. When a 429 does land, the limiter **honours the provider's own retry hint** (the `Retry-After` header, or Gemini's `RetryInfo` retry-delay) — or an exponential backoff — and **re-issues the request** (safe for streaming: the retry happens before any answer text is emitted), and it **self-lowers** the rate after a 429 so later calls pace slower. Only a 429 that outlasts the retries falls through to the existing **pause/checkpoint** (the right home for a daily-quota cap you resume tomorrow). Provider settings gain an editable **Requests per minute** field (with the effective spacing, e.g. *"≈12s between requests"*, and a reset-to-default), and a paced sweep shows a **"Rate limited — retrying in Ns"** line in its progress strip so it never looks frozen. See `export/app/lib/rateLimit.ts` (the whole mechanism), `export/app/lib/askProvider.ts` (`PausableError` + `acquire`/retry in `askStream`, 429 in `ensureOk`), `export/app/lib/nativeTools/shared.ts` (`acquire`/retry in `postJson`), `export/app/ask/{useAskChat,ProviderSettings,PinnedResultsPanel,AskChat}.tsx`, and `export/e2e/ask-workspace.spec.ts`.
diff --git a/export/e2e/channel-group-chips.spec.ts b/export/e2e/channel-group-chips.spec.ts
@@ -0,0 +1,256 @@
+import { expect, test, type Page, type Route } from "@playwright/test";
+
+// Self-contained fixtures exercising the channel-group filter chips
+// (common/components/WorkspaceSearchBar.tsx): several small groups collapse to
+// wrapping chips with a tri-state group checkbox; expanding breaks a group out
+// to a full-width card with per-channel checkboxes and All/None.
+
+type Chan = { name: string; slug: string; groupId: string };
+
+const CHANNELS: Chan[] = [
+ { name: "Alpha", slug: "alpha", groupId: "g1" },
+ { name: "Beta", slug: "beta", groupId: "g1" },
+ { name: "Gamma", slug: "gamma", groupId: "g2" },
+ { name: "Delta", slug: "delta", groupId: "g3" },
+];
+
+const GROUPS = [
+ {
+ id: "g1",
+ name: "Group One",
+ description: "First fixture group",
+ selectedByDefault: true,
+ order: 1,
+ },
+ {
+ id: "g2",
+ name: "Group Two",
+ description: "Second fixture group",
+ selectedByDefault: true,
+ order: 2,
+ },
+ {
+ id: "g3",
+ name: "Group Three",
+ description: "Third fixture group",
+ selectedByDefault: true,
+ order: 3,
+ },
+];
+
+function summary(c: Chan) {
+ return {
+ slug: `${c.slug}/vid-${c.slug}`,
+ id: `vid-${c.slug}`,
+ channelSlug: c.slug,
+ title: `Video from ${c.name}`,
+ uploadDate: "20250101",
+ date: "2025-01-01",
+ duration: "5:00",
+ channel: c.name,
+ isLivestream: false,
+ ageRestricted: false,
+ isDeleted: false,
+ isUnlisted: false,
+ platform: "youtube" as const,
+ webpageUrl: `https://example.com/vid-${c.slug}`,
+ };
+}
+
+async function fulfillJson(route: Route, body: unknown) {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(body),
+ });
+}
+
+async function installRoutes(page: Page) {
+ const list = CHANNELS.map(summary);
+ await page.route("**/summaries/manifest.json", async (route) => {
+ await fulfillJson(route, {
+ version: 3,
+ totalCount: list.length,
+ pageSize: 1000,
+ pageCount: 1,
+ generatedAt: new Date().toISOString(),
+ channels: CHANNELS.map((c) => ({
+ name: c.name,
+ count: 1,
+ groupId: c.groupId,
+ })),
+ groups: GROUPS,
+ defaultGroupId: "g1",
+ });
+ });
+ await page.route("**/summaries/page-*.json", async (route) => {
+ await fulfillJson(route, list);
+ });
+ // Transcripts are never fetched by these tests, but stub the routes so
+ // nothing falls through to the dev server.
+ await page.route(/\/transcripts\/[^/]+\/manifest\.json$/, async (route) => {
+ await fulfillJson(route, {
+ version: 1,
+ channelSlug: "unused",
+ pageCount: 0,
+ maxPageBytes: 8388608,
+ generatedAt: new Date().toISOString(),
+ slugToPage: {},
+ });
+ });
+ await page.route(/\/transcripts\/[^/]+\/page-\d+\.json$/, async (route) => {
+ await fulfillJson(route, []);
+ });
+ // No live chat.
+ await page.route("**/subs/manifest.json", async (route) => {
+ await fulfillJson(route, {
+ version: 4,
+ channels: [],
+ totalCount: 0,
+ liveChatTotalCount: 0,
+ generatedAt: new Date().toISOString(),
+ });
+ });
+}
+
+async function waitForHydration(page: Page) {
+ await page.getByTestId("query-builder").waitFor();
+}
+
+// Group cards are <details> nested inside the outer Filters <details>.
+function groupCard(page: Page, name: string) {
+ return page.locator("details details").filter({ hasText: name });
+}
+
+function groupCheckbox(page: Page, name: string) {
+ return page.getByRole("checkbox", { name: `Select all in ${name}` });
+}
+
+// A member-channel checkbox inside its group card. Scoped + exact so the
+// per-result "Select "Video from Alpha" for AI" checkboxes never collide.
+function memberCheckbox(page: Page, group: string, channel: string) {
+ return groupCard(page, group).getByRole("checkbox", {
+ name: channel,
+ exact: true,
+ });
+}
+
+// Expand/collapse by clicking the group's name inside the chip summary —
+// the tri-state checkbox and All/None isolate their own clicks.
+async function clickGroupName(page: Page, name: string) {
+ await groupCard(page, name)
+ .locator("summary")
+ .getByText(name, { exact: true })
+ .click();
+}
+
+test.describe("channel-group filter chips", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+ });
+
+ test("groups start collapsed as chips that wrap onto one row", async ({
+ page,
+ }) => {
+ // All three chips render with their selected counts.
+ await expect(groupCard(page, "Group One")).toContainText("2/2");
+ await expect(groupCard(page, "Group Two")).toContainText("1/1");
+ await expect(groupCard(page, "Group Three")).toContainText("1/1");
+
+ // Collapsed by default on a multi-group site: member checkboxes hidden.
+ await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden();
+ await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden();
+
+ // Uniform min-width chips pack onto a shared row.
+ const one = await groupCard(page, "Group One")
+ .locator("summary")
+ .boundingBox();
+ const two = await groupCard(page, "Group Two")
+ .locator("summary")
+ .boundingBox();
+ expect(one).not.toBeNull();
+ expect(two).not.toBeNull();
+ expect(one!.y).toBe(two!.y);
+ });
+
+ test("chip checkbox toggles the whole group without expanding", async ({
+ page,
+ }) => {
+ const box = groupCheckbox(page, "Group One");
+ await expect(box).toHaveAttribute("aria-checked", "true");
+
+ // Fully selected → click deselects the whole group.
+ await box.click();
+ await expect(groupCard(page, "Group One")).toContainText("0/2");
+ await expect(box).toHaveAttribute("aria-checked", "false");
+ // ...and the chip did not expand.
+ await expect(groupCard(page, "Group One")).not.toHaveAttribute("open");
+ await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden();
+
+ // Unselected → click selects the whole group.
+ await box.click();
+ await expect(groupCard(page, "Group One")).toContainText("2/2");
+ await expect(box).toHaveAttribute("aria-checked", "true");
+ await expect(groupCard(page, "Group One")).not.toHaveAttribute("open");
+ });
+
+ test("partial selection shows indeterminate; clicking selects all", async ({
+ page,
+ }) => {
+ // Expand Group One and deselect one member.
+ await clickGroupName(page, "Group One");
+ await memberCheckbox(page, "Group One", "Alpha").click();
+ await expect(groupCard(page, "Group One")).toContainText("1/2");
+
+ // Collapse — the chip checkbox reads mixed.
+ await clickGroupName(page, "Group One");
+ await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden();
+ const box = groupCheckbox(page, "Group One");
+ await expect(box).toHaveAttribute("aria-checked", "mixed");
+
+ // Radix cycles indeterminate → checked, i.e. "select all".
+ await box.click();
+ await expect(groupCard(page, "Group One")).toContainText("2/2");
+ await expect(box).toHaveAttribute("aria-checked", "true");
+ });
+
+ test("expanding shows member checkboxes and All/None at full width", async ({
+ page,
+ }) => {
+ await clickGroupName(page, "Group One");
+
+ const card = groupCard(page, "Group One");
+ await expect(card).toHaveAttribute("open", "");
+ await expect(memberCheckbox(page, "Group One", "Alpha")).toBeVisible();
+ await expect(memberCheckbox(page, "Group One", "Beta")).toBeVisible();
+ await expect(card.getByRole("button", { name: "All" })).toBeVisible();
+ await expect(card.getByRole("button", { name: "None" })).toBeVisible();
+
+ // The expanded card breaks out to (roughly) the container's full width.
+ const containerBox = await card.locator("xpath=..").boundingBox();
+ const cardBox = await card.boundingBox();
+ expect(containerBox).not.toBeNull();
+ expect(cardBox).not.toBeNull();
+ expect(cardBox!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9);
+
+ // Other groups stay collapsed chips.
+ await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden();
+ });
+
+ test("expanded/collapsed state persists across a reload", async ({
+ page,
+ }) => {
+ await clickGroupName(page, "Group Two");
+ await expect(groupCard(page, "Group Two")).toHaveAttribute("open", "");
+
+ await page.reload();
+ await waitForHydration(page);
+
+ await expect(groupCard(page, "Group Two")).toHaveAttribute("open", "");
+ await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeVisible();
+ await expect(groupCard(page, "Group One")).not.toHaveAttribute("open");
+ await expect(groupCard(page, "Group Three")).not.toHaveAttribute("open");
+ });
+});
diff --git a/export/e2e/inline-channel-chips.spec.ts b/export/e2e/inline-channel-chips.spec.ts
@@ -0,0 +1,240 @@
+import { expect, test, type Page, type Route } from "@playwright/test";
+
+// Self-contained fixtures exercising inline channel groups
+// (common/components/WorkspaceSearchBar.tsx): a group flagged `inline: true`
+// renders its member channels as loose individual chips — checkbox + name,
+// no collapsible box — flowing among the normal group chips.
+
+type Chan = { name: string; slug: string; groupId: string };
+
+const LONG_NAME =
+ "Extremely Long Channel Name That Certainly Overflows A Uniform Chip";
+
+const CHANNELS: Chan[] = [
+ { name: "Epsilon", slug: "epsilon", groupId: "solo" },
+ { name: LONG_NAME, slug: "longname", groupId: "solo" },
+ { name: "Alpha", slug: "alpha", groupId: "g1" },
+ { name: "Beta", slug: "beta", groupId: "g1" },
+ { name: "Gamma", slug: "gamma", groupId: "g2" },
+];
+
+const GROUPS = [
+ {
+ id: "solo",
+ name: "Solo",
+ selectedByDefault: true,
+ order: 0,
+ inline: true,
+ },
+ {
+ id: "g1",
+ name: "Group One",
+ description: "First fixture group",
+ selectedByDefault: true,
+ order: 1,
+ },
+ {
+ id: "g2",
+ name: "Group Two",
+ description: "Second fixture group",
+ selectedByDefault: true,
+ order: 2,
+ },
+];
+
+function summary(c: Chan) {
+ return {
+ slug: `${c.slug}/vid-${c.slug}`,
+ id: `vid-${c.slug}`,
+ channelSlug: c.slug,
+ title: `Video from ${c.name}`,
+ uploadDate: "20250101",
+ date: "2025-01-01",
+ duration: "5:00",
+ channel: c.name,
+ isLivestream: false,
+ ageRestricted: false,
+ isDeleted: false,
+ isUnlisted: false,
+ platform: "youtube" as const,
+ webpageUrl: `https://example.com/vid-${c.slug}`,
+ };
+}
+
+async function fulfillJson(route: Route, body: unknown) {
+ await route.fulfill({
+ status: 200,
+ contentType: "application/json",
+ body: JSON.stringify(body),
+ });
+}
+
+async function installRoutes(page: Page) {
+ const list = CHANNELS.map(summary);
+ await page.route("**/summaries/manifest.json", async (route) => {
+ await fulfillJson(route, {
+ version: 3,
+ totalCount: list.length,
+ pageSize: 1000,
+ pageCount: 1,
+ generatedAt: new Date().toISOString(),
+ channels: CHANNELS.map((c) => ({
+ name: c.name,
+ count: 1,
+ groupId: c.groupId,
+ })),
+ groups: GROUPS,
+ defaultGroupId: "solo",
+ });
+ });
+ await page.route("**/summaries/page-*.json", async (route) => {
+ await fulfillJson(route, list);
+ });
+ // Transcripts are never fetched by these tests, but stub the routes so
+ // nothing falls through to the dev server.
+ await page.route(/\/transcripts\/[^/]+\/manifest\.json$/, async (route) => {
+ await fulfillJson(route, {
+ version: 1,
+ channelSlug: "unused",
+ pageCount: 0,
+ maxPageBytes: 8388608,
+ generatedAt: new Date().toISOString(),
+ slugToPage: {},
+ });
+ });
+ await page.route(/\/transcripts\/[^/]+\/page-\d+\.json$/, async (route) => {
+ await fulfillJson(route, []);
+ });
+ // No live chat.
+ await page.route("**/subs/manifest.json", async (route) => {
+ await fulfillJson(route, {
+ version: 4,
+ channels: [],
+ totalCount: 0,
+ liveChatTotalCount: 0,
+ generatedAt: new Date().toISOString(),
+ });
+ });
+}
+
+async function waitForHydration(page: Page) {
+ await page.getByTestId("query-builder").waitFor();
+}
+
+// Group cards are <details> nested inside the outer Filters <details>.
+function groupCard(page: Page, name: string) {
+ return page.locator("details details").filter({ hasText: name });
+}
+
+// A member-channel checkbox inside its group card.
+function memberCheckbox(page: Page, group: string, channel: string) {
+ return groupCard(page, group).getByRole("checkbox", {
+ name: channel,
+ exact: true,
+ });
+}
+
+// A loose per-channel chip from an inline group.
+function inlineChip(page: Page, channel: string) {
+ return page
+ .getByTestId("inline-channel-chip")
+ .filter({ hasText: channel })
+ .first();
+}
+
+test.describe("inline channel chips", () => {
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+ });
+
+ test("inline group renders loose chips instead of a group chip", async ({
+ page,
+ }) => {
+ // Two loose chips, one per Solo member — no <details> card for "Solo"
+ // and no group bulk-toggle checkbox.
+ await expect(page.getByTestId("inline-channel-chip")).toHaveCount(2);
+ await expect(groupCard(page, "Solo")).toHaveCount(0);
+ await expect(
+ page.getByRole("checkbox", { name: "Select all in Solo" }),
+ ).toHaveCount(0);
+
+ // Inline members are visible on first load: the default-collapse
+ // heuristic only counts collapsible groups (still >1 here, so g1/g2
+ // members stay hidden behind collapsed chips).
+ await expect(
+ inlineChip(page, "Epsilon").getByRole("checkbox"),
+ ).toBeVisible();
+ await expect(memberCheckbox(page, "Group One", "Alpha")).toBeHidden();
+ await expect(memberCheckbox(page, "Group Two", "Gamma")).toBeHidden();
+ });
+
+ test("inline chip toggles its channel without expanding anything", async ({
+ page,
+ }) => {
+ await expect(page.getByText("5 of 5 channels")).toBeVisible();
+
+ const box = inlineChip(page, "Epsilon").getByRole("checkbox");
+ await expect(box).toHaveAttribute("aria-checked", "true");
+
+ await box.click();
+ await expect(page.getByText("4 of 5 channels")).toBeVisible();
+ await expect(box).toHaveAttribute("aria-checked", "false");
+ // No chip expanded as a side effect of the toggle.
+ await expect(page.locator("details details[open]")).toHaveCount(0);
+
+ await box.click();
+ await expect(page.getByText("5 of 5 channels")).toBeVisible();
+ await expect(box).toHaveAttribute("aria-checked", "true");
+ await expect(page.locator("details details[open]")).toHaveCount(0);
+ });
+
+ test("uniform chip widths with long names ellipsized", async ({ page }) => {
+ // Short inline chip and short collapsed group chip share the uniform
+ // min width.
+ const epsilon = await inlineChip(page, "Epsilon").boundingBox();
+ const g2 = await groupCard(page, "Group Two").boundingBox();
+ expect(epsilon).not.toBeNull();
+ expect(g2).not.toBeNull();
+ expect(epsilon!.width).toBe(g2!.width);
+
+ // Long-named chip carries the full name as a tooltip, never wraps the
+ // text (same height as a short chip), and stays inside the container.
+ const long = inlineChip(page, LONG_NAME);
+ await expect(long).toHaveAttribute("title", LONG_NAME);
+ const longBox = await long.boundingBox();
+ const containerBox = await long.locator("xpath=..").boundingBox();
+ expect(longBox).not.toBeNull();
+ expect(containerBox).not.toBeNull();
+ expect(longBox!.height).toBe(epsilon!.height);
+ expect(longBox!.width).toBeLessThanOrEqual(containerBox!.width);
+ });
+});
+
+test.describe("inline channel chips on mobile", () => {
+ test.use({ viewport: { width: 390, height: 844 } });
+
+ test.beforeEach(async ({ page }) => {
+ await installRoutes(page);
+ await page.goto("/");
+ await waitForHydration(page);
+ });
+
+ test("chips stack one per line at phone widths", async ({ page }) => {
+ const epsilon = await inlineChip(page, "Epsilon").boundingBox();
+ const long = await inlineChip(page, LONG_NAME).boundingBox();
+ const containerBox = await inlineChip(page, "Epsilon")
+ .locator("xpath=..")
+ .boundingBox();
+ expect(epsilon).not.toBeNull();
+ expect(long).not.toBeNull();
+ expect(containerBox).not.toBeNull();
+
+ // Stacked: same left edge, different rows, each near full width.
+ expect(epsilon!.x).toBe(long!.x);
+ expect(epsilon!.y).not.toBe(long!.y);
+ expect(epsilon!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9);
+ expect(long!.width).toBeGreaterThanOrEqual(containerBox!.width * 0.9);
+ });
+});
diff --git a/mcp/README.md b/mcp/README.md
@@ -13,14 +13,35 @@ reads the site's already-published static JSON shards (`corpus.json` +
| Tool | What it does |
|------|--------------|
| `list_channels` | List channels **organized under their channel groups** (name, slug, video count; site in hub mode), with a compact group cheat-sheet (`id · name · N channels`) for scoping. |
-| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets. Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
-| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches, or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
-| `get_transcript` | One video's full transcript as clean markdown (metadata + timestamped captions). |
+| `search_transcripts` | Search captions for a term/phrase (or regex); returns matching videos with timestamped snippets — **each `[mm:ss]` is a clickable link to that exact moment** (or a compact `[mm:ss\|sec]` with `link_style:"base"`). Alias-aware and **pageable** (`total` + `offset`). Scope by one or more channels (`channel`/`channels`) and/or channel groups (`group`/`groups`). |
+| `get_transcripts` | Batch-read up to 20 videos in one call — bounded, timestamped **excerpt windows** around a query's matches (linked `[mm:ss]`, or compact `link_style:"base"` stamps; `max_lines` caps the excerpt), or full transcripts without a query. `channel`/`channels` hints speed the lookup. |
+| `get_transcript` | One video's full transcript as clean markdown (metadata + **linked** timestamped captions). |
| `get_video_metadata` | One video's metadata (title, channel, date, duration, description, tags, source URL) without the transcript body. |
+| `open_link` | Paste an archilyzer viewer **share link** to re-run that exact search here (query tree + every filter, at full fidelity). **Previews** a plan by default; **applies** it (switch source + search, linked results) on `apply:true`. Adjust in natural language via `overrides`. |
| `list_sources` | Show the **active** corpus (label + kind + target) and, with a hub context, its member sites (`siteId · title · url`, marking the current subset) so you can pick one to switch to. |
| `use_source` | **Switch which corpus is read**, on the fly — a hub member (`site`), a hub subset (`sites`), or an arbitrary `remote` URL / `local` dir / `hub` URL. Persists across reconnects. |
| `reset_source` | Return to the source the server was started with and clear the persisted selection. |
+**Clickable moment links.** Every timestamp the read tools emit is a Markdown link
+to the exact second — an **archilyzer viewer** deep link (`…/?v=<slug>&t=<sec>`,
+opening the transcript modal at the moment) when the source has a public origin,
+otherwise the video's platform watch page with a per-platform time param
+(YouTube `&t=<sec>s`, Odysee/Twitch equivalents). A `--local` source with no
+origin falls back to platform links.
+
+**Compact base links (`link_style:"base"`).** Inline links are ~70–90 chars *per
+line* — bulk an agent pipeline shouldn't pay for. `search_transcripts` and
+`get_transcripts` accept `link_style:"base"`: each video gets **one**
+`- moment_base:` header line (a URL ending in `t=`) and every stamp becomes the
+compact `[mm:ss|<seconds>]`. The expansion rule: **full moment link =
+`<moment_base><seconds>`** — append the integer after the `|`, e.g.
+`[title @ 2:36](<moment_base>156)`. The seconds are floored exactly like the
+inline links', so both styles cite the identical second. A video whose base
+can't be built (no viewer origin and a platform whose time param doesn't take
+raw seconds — Twitch — or doesn't exist — Rumble/Kick) omits the line: cite its
+`- source:` URL plain instead. `open_link` results and `get_transcript` stay
+inline-linked (future work).
+
### `search_transcripts`
Beyond `query`, `regex`, and `limit`:
@@ -42,8 +63,11 @@ Beyond `query`, `regex`, and `limit`:
full `total` and `has_more`, so you can enumerate a query's *entire* match set:
page with `offset += limit` until `has_more` is `no`.
- **`include_snippets`** (default true) — set `false` for a cheap worklist
- (id / title / channel / date / match count, no cue text). Ideal for the
- planning pass of a sweep.
+ (id / title / channel / date / match count, no cue text — the per-video
+ `- source:` line is dropped too). Ideal for the planning pass of a sweep.
+- **`link_style`** (default `"inline"`) — `"base"` switches to the compact
+ agent-pipeline form: a `- moment_base:` line per hit and `[mm:ss|<seconds>]`
+ snippet stamps (see *Compact base links* above).
- **`use_aliases`** (default true) — expand the query through the site's curated
search aliases. A plain query that matches a curated trigger also searches the
alias's regex, so mis-transcribed spellings are caught (e.g. `k cups` also
@@ -66,6 +90,46 @@ video's full transcript comes back as markdown. Missing ids are reported inline.
lookup when a batch spans several channels (the ids are already scoped by the
search that produced them).
+- **`link_style`** (default `"inline"`) — `"base"` emits one `- moment_base:`
+ line per video and compact `[mm:ss|<seconds>]` stamps instead of a full
+ Markdown link per line (see *Compact base links* above).
+- **`max_lines`** (default 200) — cap on merged excerpt lines per video when a
+ query is given (the earliest lines are kept). Ignored without a query.
+
+### `open_link` — paste a viewer share link, at full fidelity
+
+The archilyzer viewer's **Share** button produces a URL that encodes the whole
+search: the origin, a `qt=` composite **query tree** (any mix of scopes —
+transcripts, live-chat, title/channel, description, tags — combined with
+AND/OR/NOT), and the filter block (`fc` channels, `ft` type, `fa` audience,
+`fav` availability, `fdf`/`fdt` upload-date range). `open_link` re-runs that exact
+search here — no manual source-switching or query reconstruction, nothing lost.
+
+It is a **preview → adjust → apply** flow:
+
+- **Preview** (default, `apply:false`) — decode the link and return a *plan*: the
+ resolved source (origin, hub vs single-site, auto-probed from `corpus.json`),
+ the query tree rendered readably, every active filter, the channel scope
+ validated against the live corpus, and anything ignored or warned (the `fk`
+ subtitle-track token is vestigial in the composite share model — decoded and
+ reported as ignored). **No source switch, no search yet.**
+- **Adjust** — re-call with structured `overrides` to honor a natural-language
+ edit: `clear_availability` (drop the availability filter), `clear_type`,
+ `clear_age`, `clear_dates`, `clear_filters`, `clear_channels`,
+ `channels:[…]` (re-scope), `date_from`/`date_to`, or `query`/`regex`/
+ `query_scope` (replace the search). The plan updates in place.
+- **Apply** (`apply:true`) — switch the active source to the link's origin (a
+ single-site remote, or a hub when the origin federates), run the search at full
+ fidelity (the query tree + filters, honoring the chat and availability scopes),
+ and return the first page of **linked** results. The applied plan is echoed at
+ the top for transparency. Pageable via `limit`/`offset`. Still read-only.
+
+```
+open_link link="https://rekietalyzer.pages.dev/?qt=…&fv=1&fc=Rekieta%20Law&ft=v&fav=a"
+open_link link="…" overrides={ "clear_availability": true } # "drop the availability filter"
+open_link link="…" apply=true # switch source + search
+```
+
## The `sweep` prompt
A first-class slash command that turns **Claude Code itself** into the corpus
@@ -74,9 +138,12 @@ browser does with a BYO AI key, but driven by your Claude **plan usage** (no API
key) and with the report written to a file.
Invoke it in Claude Code as `/mcp__<server-name>__sweep`. Arguments: `query`
-(required), `channel?`, `channels?` (comma-separated slugs/names), `group?` (a
-channel group id or name), `directive?` (default *"key claims &
-contradictions"*), `batch_size?` (default 8), `report_path?` (default
+(required *unless* a `link` is given), `link?` (an archilyzer viewer share URL to
+seed the sweep from), `channel?`, `channels?` (comma-separated slugs/names),
+`group?` (a channel group id or name), `directive?` (default *"key claims &
+contradictions"*), `batch_size?` (default 8), `parse_model?` (default
+`haiku` — the model requested for the per-batch extractor subagents; they only
+quote verbatim, so the cheapest model wins), `report_path?` (default
`./sweep-report.md`).
```
@@ -84,8 +151,15 @@ contradictions"*), `batch_size?` (default 8), `report_path?` (default
/mcp__rekietalyzer__sweep query="k cups" group="other" # a whole group
/mcp__rekietalyzer__sweep query="k cups" group="Extended Universe" # …by name
/mcp__rekietalyzer__sweep query="k cups" # pick-first (see below)
+/mcp__rekietalyzer__sweep link="https://…/?qt=…&fv=1&fc=…" # seed from a share link
```
+**Seed from a share link (`link=`).** Pass a viewer share URL instead of a
+`query` and the sweep starts by driving `open_link`: it previews the decoded plan
+(source, query tree, filters, scope, ignored bits), lets you confirm or adjust in
+natural language, then applies it — switching source and enumerating the full
+tree+filter match set — before batching. Everything the link encodes is honored.
+
**Pick-first when no scope is given.** Because a whole-corpus sweep can be a lot
of work, invoking `sweep` **without** a `channel`/`channels`/`group` makes Claude
FIRST call `list_channels`, present the groups and their channels, and ask you
@@ -97,13 +171,32 @@ name (`group="other"` ≡ `group="Extended Universe"`).
Once scoped, the prompt instructs Claude Code to: **search** (aliases
auto-expand; the footer names the resolved scope) → **enumerate** the full
worklist by paging with `include_snippets:false` until `has_more` is false →
-**plan** `ceil(N / batch_size)` batches → **per batch** `get_transcripts` for
-windowed context, cross-reference against the report so far, and upsert findings
-(claims, contradictions, sources cited as *title + [mm:ss]*) into well-titled
-`##` sections of the report file with its own Write/Edit tools → finish with a
-short summary. The MCP stays read-only; only the report file is written, in
+**plan** `ceil(N / batch_size)` batches → **per batch, map-reduce** → finish with
+a short summary. The MCP stays read-only; only the report file is written, in
Claude's working directory.
+**Dumb extractors on the cheap model.** Each batch is handed to a **subagent**
+(Claude Code's Task tool) spawned as a *verbatim extractor* on the cheapest
+model — the prompt requests `parse_model` (default `haiku`) via the Task tool's
+model override, and gracefully spawns on the default when no override exists.
+The extractor calls `get_transcripts` with `link_style:"base"` and returns
+*only* the directive-relevant lines, **verbatim** — grouped per video under its
+`moment_base:` line, in their compact `[mm:ss|sec]` form, under a hard budget of
+**≤40 lines (~600 words) per batch**. No analysis, no summarizing: the heavy
+transcript bulk lives and dies inside the cheap subagent, and nothing irrelevant
+is ever reprocessed. The **orchestrator does all the synthesis** — claims,
+contradictions, cross-referencing — expanding each kept citation to a full link
+by appending the seconds to the video's `moment_base`, then discards the
+fragment. Batches are independent, so several can run in parallel. If no
+subagent tool is available, the batch is processed inline (still
+`link_style:"base"`) and the raw excerpt text dropped after folding.
+
+**Linked citations.** Every source is cited as a clickable
+`[title @ mm:ss](<moment url>)` link — built by appending the cited integer
+seconds to the video's `moment_base` from the tool output, so a click seeks to
+the exact second (see *Compact base links* above; a video with no base is cited
+by its `source:` URL). Note `open_link` results stay inline-linked.
+
## Data source (pick one)
Resolved from flags or env — precedence hub > remote > local:
diff --git a/mcp/src/momentUrl.test.ts b/mcp/src/momentUrl.test.ts
@@ -0,0 +1,219 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ momentUrl,
+ viewerMomentUrl,
+ platformMomentUrl,
+ momentBaseUrl,
+ viewerMomentBaseUrl,
+ platformMomentBaseUrl,
+} from "yt-dlp-transcript-common/lib/momentUrl";
+
+// ─── Archilyzer viewer link (preferred) ───
+
+test("viewer link: origin + slug + seconds → ?v=<slug>&t=<sec>", () => {
+ const url = viewerMomentUrl(
+ "https://rekietalyzer.pages.dev",
+ "rekietalaw/0eYLR6CbR78",
+ 754,
+ );
+ assert.ok(url);
+ const u = new URL(url!);
+ assert.equal(u.origin, "https://rekietalyzer.pages.dev");
+ assert.equal(u.pathname, "/");
+ // The slug round-trips through URL decoding to exactly the modal's ?v= value.
+ assert.equal(u.searchParams.get("v"), "rekietalaw/0eYLR6CbR78");
+ assert.equal(u.searchParams.get("t"), "754");
+});
+
+test("viewer link: an origin with its own path/query is normalized to /", () => {
+ const url = viewerMomentUrl("https://site.example/search?q=x", "ch/vid", 10);
+ const u = new URL(url!);
+ assert.equal(u.pathname, "/");
+ assert.equal(u.searchParams.get("q"), null);
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("viewer link: t is omitted at second 0", () => {
+ const url = viewerMomentUrl("https://site.example", "ch/vid", 0);
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), null);
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("viewer link: null without an origin or a slug", () => {
+ assert.equal(viewerMomentUrl(null, "ch/vid", 10), null);
+ assert.equal(viewerMomentUrl("https://site.example", null, 10), null);
+});
+
+// ─── Platform fallback ───
+
+test("platform link: YouTube gets &t=<sec>s appended", () => {
+ const url = platformMomentUrl(
+ "https://www.youtube.com/watch?v=abc123",
+ "youtube",
+ 90,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("v"), "abc123");
+ assert.equal(u.searchParams.get("t"), "90s");
+});
+
+test("platform link: platform is detected from the URL when not given", () => {
+ const url = platformMomentUrl("https://www.youtube.com/watch?v=abc123", null, 5);
+ assert.match(url!, /t=5s/);
+});
+
+test("platform link: Odysee gets ?t=<sec> (raw seconds)", () => {
+ const url = platformMomentUrl(
+ "https://odysee.com/@chan/video",
+ "odysee",
+ 42,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), "42");
+});
+
+test("platform link: Twitch gets ?t=<XhYmZs>", () => {
+ const url = platformMomentUrl(
+ "https://www.twitch.tv/videos/12345",
+ "twitch",
+ 3661,
+ );
+ const u = new URL(url!);
+ assert.equal(u.searchParams.get("t"), "1h1m1s");
+});
+
+test("platform link: Rumble has no reliable start param → bare webpage url", () => {
+ const url = platformMomentUrl("https://rumble.com/v123-title.html", "rumble", 60);
+ assert.equal(url, "https://rumble.com/v123-title.html");
+});
+
+test("platform link: second 0 → bare webpage url; no webpage url → null", () => {
+ assert.equal(
+ platformMomentUrl("https://www.youtube.com/watch?v=abc", "youtube", 0),
+ "https://www.youtube.com/watch?v=abc",
+ );
+ assert.equal(platformMomentUrl(null, "youtube", 30), null);
+});
+
+// ─── momentUrl: prefers the viewer link, falls back to the platform link ───
+
+test("momentUrl: viewer link wins when an origin is present", () => {
+ const url = momentUrl({
+ siteOrigin: "https://site.example",
+ slug: "ch/vid",
+ seconds: 12,
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ const u = new URL(url!);
+ assert.equal(u.origin, "https://site.example");
+ assert.equal(u.searchParams.get("v"), "ch/vid");
+});
+
+test("momentUrl: falls back to the platform link with no origin (local source)", () => {
+ const url = momentUrl({
+ siteOrigin: null,
+ slug: "ch/vid",
+ seconds: 12,
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.match(url!, /youtube\.com/);
+ assert.match(url!, /t=12s/);
+});
+
+test("momentUrl: null when neither a viewer nor a platform link can be built", () => {
+ const url = momentUrl({ siteOrigin: null, slug: "ch/vid", seconds: 12 });
+ assert.equal(url, null);
+});
+
+// ─── Base (appendable) forms — the link_style:"base" building blocks ───
+
+test("viewer base: ends in t= with the slug encoded; appending seconds parses", () => {
+ const base = viewerMomentBaseUrl(
+ "https://rekietalyzer.pages.dev",
+ "rekietalaw/uamorPe6hSc",
+ );
+ assert.ok(base);
+ assert.ok(base!.endsWith("&t="), "the base ends in t= so seconds append cleanly");
+ assert.ok(
+ base!.includes("v=rekietalaw%2FuamorPe6hSc"),
+ "the slug's / is %2F-encoded by URLSearchParams",
+ );
+ // Appending integer seconds yields exactly the viewer deep-link params.
+ const u = new URL(base! + "754");
+ assert.equal(u.searchParams.get("v"), "rekietalaw/uamorPe6hSc");
+ assert.equal(u.searchParams.get("t"), "754");
+});
+
+test("viewer base: null without an origin/slug or with an unparseable origin", () => {
+ assert.equal(viewerMomentBaseUrl(null, "ch/vid"), null);
+ assert.equal(viewerMomentBaseUrl("https://site.example", null), null);
+ assert.equal(viewerMomentBaseUrl("not a url", "ch/vid"), null);
+});
+
+test("platform base: YouTube ends in t= (existing query → &t=, none → ?t=)", () => {
+ const withQuery = platformMomentBaseUrl(
+ "https://www.youtube.com/watch?v=abc123",
+ "youtube",
+ );
+ assert.ok(withQuery!.endsWith("&t="));
+ const noQuery = platformMomentBaseUrl("https://youtu.be/abc123", "youtube");
+ assert.ok(noQuery!.endsWith("?t="));
+});
+
+test("platform base: a pre-existing t param is stripped and re-appended LAST", () => {
+ const base = platformMomentBaseUrl(
+ "https://www.youtube.com/watch?t=99s&v=abc",
+ "youtube",
+ );
+ assert.ok(base!.endsWith("&t="), "t= is the last param");
+ assert.ok(!base!.includes("99s"), "the old seek value is gone");
+ const u = new URL(base! + "42");
+ assert.equal(u.searchParams.get("v"), "abc");
+ assert.equal(u.searchParams.get("t"), "42");
+});
+
+test("platform base: Odysee ends in t= (raw-seconds param)", () => {
+ const base = platformMomentBaseUrl("https://odysee.com/@chan/video", "odysee");
+ assert.ok(base!.endsWith("?t="));
+});
+
+test("platform base: non-appendable cases → null (never an invalid seek)", () => {
+ // Twitch's t takes an XhYmZs token — appending raw seconds would mis-seek.
+ assert.equal(
+ platformMomentBaseUrl("https://www.twitch.tv/videos/12345", "twitch"),
+ null,
+ );
+ // Rumble has no dependable start param at all.
+ assert.equal(
+ platformMomentBaseUrl("https://rumble.com/v123-title.html", "rumble"),
+ null,
+ );
+ assert.equal(platformMomentBaseUrl(null, "youtube"), null);
+ assert.equal(platformMomentBaseUrl("not a url", "youtube"), null);
+});
+
+test("momentBaseUrl: viewer base wins, platform fallback, else null", () => {
+ const viewer = momentBaseUrl({
+ siteOrigin: "https://site.example",
+ slug: "ch/vid",
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.ok(viewer!.startsWith("https://site.example/?v="));
+ assert.ok(viewer!.endsWith("&t="));
+
+ const platform = momentBaseUrl({
+ siteOrigin: null,
+ slug: "ch/vid",
+ webpageUrl: "https://www.youtube.com/watch?v=vid",
+ platform: "youtube",
+ });
+ assert.ok(platform!.startsWith("https://www.youtube.com/watch?v=vid"));
+ assert.ok(platform!.endsWith("&t="));
+
+ assert.equal(momentBaseUrl({ siteOrigin: null, slug: "ch/vid" }), null);
+});
diff --git a/mcp/src/search.test.ts b/mcp/src/search.test.ts
@@ -2,29 +2,60 @@ import { test } from "node:test";
import assert from "node:assert/strict";
import { Client } from "@modelcontextprotocol/sdk/client/index.js";
import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
-import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
+import type {
+ ChannelTranscriptsManifest,
+ ChannelSubsManifest,
+} from "yt-dlp-transcript-common/lib/manifest";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
import type { ChannelGroup } from "yt-dlp-transcript-common/lib/channelGroups";
-import type { ChannelGroups, ChannelRef, ShardSource } from "./source";
+import { newGroup, newLeaf } from "yt-dlp-transcript-common/lib/searchQuery";
+import type {
+ ChannelGroups,
+ ChannelRef,
+ ShardSource,
+ VideoAvailability,
+} from "./source";
import {
searchTranscripts,
getWindowedTranscript,
buildMatcher,
resolveSelectedChannels,
+ runSearchSpec,
+ type SearchFilters,
} from "./search";
import { createServer } from "./server";
+// Every filter facet kept (the "no-op" filter) — override fields per test.
+const KEEP_ALL: SearchFilters = {
+ videos: true,
+ livestreams: true,
+ allAges: true,
+ restricted: true,
+ available: true,
+ unlisted: true,
+ deleted: true,
+};
+
// ─── A tiny in-memory ShardSource for the tests ───
function cues(...pairs: [number, string][]): Cue[] {
return pairs.map(([start, text]) => ({ start, end: start + 3, text }));
}
-function vid(id: string, title: string, channelSlug: string, cs: Cue[]): TranscriptDetail {
+function vid(
+ id: string,
+ title: string,
+ channelSlug: string,
+ cs: Cue[],
+ extra: Partial<TranscriptDetail> = {},
+): TranscriptDetail {
return {
- slug: id,
+ // Real records carry a channel-prefixed slug (`<channelSlug>/<id>`); mirror
+ // that so viewer-link + availability joins are exercised.
+ slug: `${channelSlug}/${id}`,
id,
channelSlug,
title,
@@ -38,6 +69,7 @@ function vid(id: string, title: string, channelSlug: string, cs: Cue[]): Transcr
platform: "youtube",
webpageUrl: `https://example.test/${id}`,
cues: cs,
+ ...extra,
};
}
@@ -51,16 +83,40 @@ const K_CUPS_ALIAS: SearchAlias = {
};
// Channel A: five "coffee" videos (paging) + one literal "k cups" + one "cake cup".
+// Extra metadata on a1/a3/a4/a5 exercises the description/tags scopes and the
+// type/age/date filters without changing what matches the "coffee" cue search.
const CHAN_A: TranscriptDetail[] = [
- vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"])),
+ vid("a1", "Coffee one", "chan-a", cues([10, "i love coffee"]), {
+ description: "a pour over brewing guide",
+ tags: ["espresso", "beans"],
+ }),
vid("a2", "Coffee two", "chan-a", cues([10, "more coffee here"])),
- vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"])),
- vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"])),
- vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"])),
+ vid("a3", "Coffee three", "chan-a", cues([10, "coffee coffee coffee"]), {
+ isLivestream: true,
+ }),
+ vid("a4", "Coffee four", "chan-a", cues([10, "cold brew coffee"]), {
+ ageRestricted: true,
+ }),
+ vid("a5", "Coffee five", "chan-a", cues([10, "the last coffee"]), {
+ uploadDate: "20250601",
+ }),
vid("a6", "Kcup talk", "chan-a", cues([10, "i buy k cups weekly"])),
vid("a7", "Cake cup talk", "chan-a", cues([10, "she said cake cup on air"])),
];
+// Live-chat cues, keyed by video slug — served via the subs shard methods so a
+// `chat`-scope leaf has something to match. Only b1 has a chat track.
+const CHAT: Record<string, Cue[]> = {
+ "chan-b/b1": cues([12, "@fan: banana bread anyone"], [15, "@mod: stay on topic"]),
+};
+
+// Availability overrides, keyed by video slug (defaults: available). a1 deleted,
+// a2 unlisted — the rest available.
+const AVAILABILITY: Record<string, VideoAvailability> = {
+ "chan-a/a1": { isDeleted: true, isUnlisted: false },
+ "chan-a/a2": { isDeleted: false, isUnlisted: true },
+};
+
// Channel B: one long transcript for windowing (a single match at 100s).
const CHAN_B: TranscriptDetail[] = [
vid(
@@ -136,6 +192,52 @@ class StubSource implements ShardSource {
async transcriptPage(ch: ChannelRef, page: number): Promise<TranscriptDetail[]> {
return this.pages(ch)[page] ?? [];
}
+
+ publicOrigin(): string | null {
+ return null; // a stub has no viewer origin
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ // Only chan-b ships a (single-page) subs shard, with b1 on page 0.
+ if (ch.slug !== "chan-b") return null;
+ return {
+ version: 1,
+ channelSlug: ch.slug,
+ pageCount: 1,
+ maxPageBytes: 0,
+ generatedAt: "",
+ tracks: ["live_chat"],
+ slugToPage: { b1: 0 },
+ };
+ }
+
+ async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ if (ch.slug !== "chan-b" || page !== 0) return [];
+ const slug = "chan-b/b1";
+ return [
+ {
+ slug,
+ id: "b1",
+ channelSlug: "chan-b",
+ title: "Long one",
+ uploadDate: "20240101",
+ date: "2024-01-01",
+ duration: "5:00",
+ channel: "Channel B",
+ isLivestream: false,
+ ageRestricted: false,
+ isDeleted: false,
+ isUnlisted: false,
+ platform: "youtube",
+ webpageUrl: "https://example.test/b1",
+ tracks: { live_chat: CHAT[slug] },
+ },
+ ];
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map(Object.entries(AVAILABILITY));
+ }
}
// ─── (a) Paging: offset / limit / total / hasMore ───
@@ -302,6 +404,202 @@ test("getWindowedTranscript: no matches yields no lines", async () => {
assert.equal(lines.length, 0);
});
+test("getWindowedTranscript: a stamp formatter renders the bracket contents", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "coffee" });
+ const { lines } = getWindowedTranscript(rec, match, {
+ before: 30,
+ after: 30,
+ stamp: (clock, seconds) => `${clock}|${Math.floor(seconds)}`,
+ });
+ assert.ok(
+ lines.includes("[1:40|100] here is the coffee moment"),
+ lines.join("\n"),
+ );
+});
+
+test("getWindowedTranscript: maxLines keeps the earliest merged lines", async () => {
+ const rec = CHAN_B[0];
+ const { match } = buildMatcher({ query: "coffee" });
+ const { lines, matchCount } = getWindowedTranscript(rec, match, {
+ before: 30,
+ after: 30,
+ maxLines: 2,
+ });
+ assert.equal(matchCount, 1, "matchCount is unaffected by the line cap");
+ assert.equal(lines.length, 2);
+ assert.ok(lines[0].includes("right before the moment"));
+ assert.ok(lines[1].includes("here is the coffee moment"));
+ assert.ok(!lines.join("\n").includes("right after the moment"));
+});
+
+// ─── runSearchSpec: per-scope leaves, tree algebra, filters, paging ───
+
+async function allChannels(src: StubSource): Promise<ChannelRef[]> {
+ return (await resolveSelectedChannels(src, {})).channels;
+}
+
+function leafTree(
+ query: string,
+ scope: "transcripts" | "chat" | "metadata" | "description" | "tags",
+ extra: Record<string, unknown> = {},
+) {
+ return newGroup({ children: [newLeaf({ query, scope, ...extra })] });
+}
+
+async function specIds(
+ src: StubSource,
+ tree: ReturnType<typeof newGroup>,
+ spec: { filters?: SearchFilters | null; aliases?: SearchAlias[] } = {},
+): Promise<string[]> {
+ const res = await runSearchSpec(src, await allChannels(src), { tree, ...spec }, { limit: 50 });
+ return res.hits.map((h) => h.videoId).sort();
+}
+
+test("spec transcripts leaf: matches cue text (same set as the plain scanner)", async () => {
+ const src = new StubSource();
+ assert.deepEqual(await specIds(src, leafTree("coffee", "transcripts")), [
+ "a1", "a2", "a3", "a4", "a5", "b1",
+ ]);
+});
+
+test("spec metadata leaf: matches title/channel, not cues", async () => {
+ const src = new StubSource();
+ // "kcup" is in a6's title but no cue (cue says "k cups" with a space).
+ assert.deepEqual(await specIds(src, leafTree("kcup", "metadata")), ["a6"]);
+});
+
+test("spec description leaf: matches the description only", async () => {
+ const src = new StubSource();
+ // "brewing" is only in a1's description, in no cue.
+ assert.deepEqual(await specIds(src, leafTree("brewing", "description")), ["a1"]);
+});
+
+test("spec tags leaf: matches the joined tags", async () => {
+ const src = new StubSource();
+ assert.deepEqual(await specIds(src, leafTree("espresso", "tags")), ["a1"]);
+});
+
+test("spec chat leaf: matches live_chat cues fetched from subs shards", async () => {
+ const src = new StubSource();
+ // "banana" only appears in b1's live chat, in no transcript cue.
+ const res = await runSearchSpec(src, await allChannels(src), {
+ tree: leafTree("banana", "chat"),
+ }, { limit: 50 });
+ assert.deepEqual(res.hits.map((h) => h.videoId), ["b1"]);
+ assert.equal(res.hits[0].snippets[0].scope, "chat");
+ assert.equal(res.hits[0].snippets[0].track, "live_chat");
+});
+
+test("spec AND: transcripts AND tags narrows to the intersection", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags" }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a1"]);
+});
+
+test("spec OR: description OR chat unions the two", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "OR",
+ children: [
+ newLeaf({ query: "brewing", scope: "description" }),
+ newLeaf({ query: "banana", scope: "chat" }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a1", "b1"]);
+});
+
+test("spec negate: coffee AND NOT (espresso tag) drops a1", async () => {
+ const src = new StubSource();
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags", negate: true }),
+ ],
+ });
+ assert.deepEqual(await specIds(src, tree), ["a2", "a3", "a4", "a5", "b1"]);
+});
+
+test("spec filter ft: livestreams:false drops the livestream (a3)", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, livestreams: false },
+ });
+ assert.deepEqual(ids, ["a1", "a2", "a4", "a5", "b1"]);
+});
+
+test("spec filter fa: keep only age-restricted → a4", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, allAges: false },
+ });
+ assert.deepEqual(ids, ["a4"]);
+});
+
+test("spec filter fav: available-only drops deleted a1 + unlisted a2", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, unlisted: false, deleted: false },
+ });
+ assert.deepEqual(ids, ["a3", "a4", "a5", "b1"]);
+});
+
+test("spec filter fav: deleted+unlisted-only keeps a1 + a2", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, available: false },
+ });
+ assert.deepEqual(ids, ["a1", "a2"]);
+});
+
+test("spec filter dates: fdf bound keeps only the later upload (a5)", async () => {
+ const src = new StubSource();
+ const ids = await specIds(src, leafTree("coffee", "transcripts"), {
+ filters: { ...KEEP_ALL, dateFrom: "20250101" },
+ });
+ assert.deepEqual(ids, ["a5"]);
+});
+
+test("spec paging/total: stable total with an offset/limit slice", async () => {
+ const src = new StubSource();
+ const chans = await allChannels(src);
+ const tree = leafTree("coffee", "transcripts");
+ const page = await runSearchSpec(src, chans, { tree }, { limit: 2, offset: 2 });
+ assert.equal(page.total, 6);
+ assert.deepEqual(page.hits.map((h) => h.videoId), ["a3", "a4"]);
+ assert.equal(page.hasMore, true);
+});
+
+test("spec aliases: a transcripts leaf expands via curated aliases", async () => {
+ const src = new StubSource();
+ const res = await runSearchSpec(
+ src,
+ await allChannels(src),
+ { tree: leafTree("k cups", "transcripts"), aliases: await src.loadAliases() },
+ { limit: 50 },
+ );
+ assert.deepEqual(res.hits.map((h) => h.videoId).sort(), ["a6", "a7"]);
+ assert.equal(res.firedAliases.length, 1);
+ assert.equal(res.firedAliases[0].id, "k-cups");
+});
+
+test("spec hits carry slug + platform for moment links", async () => {
+ const src = new StubSource();
+ const res = await runSearchSpec(src, await allChannels(src), {
+ tree: leafTree("coffee", "transcripts"),
+ }, { limit: 1 });
+ assert.equal(res.hits[0].slug, "chan-a/a1");
+ assert.equal(res.hits[0].platform, "youtube");
+ assert.ok(res.hits[0].snippets[0].seconds >= 0);
+});
+
// ─── End-to-end through the MCP server (tools + prompts + missing ids) ───
async function connectClient(source: ShardSource): Promise<Client> {
@@ -410,6 +708,8 @@ test("server: the sweep prompt lists with its arguments and renders the query",
"channels",
"directive",
"group",
+ "link",
+ "parse_model",
"query",
"report_path",
]);
@@ -449,3 +749,159 @@ test("server: the sweep prompt with no scope instructs a group/channel pick firs
assert.match(text, /confirm \*\*all\*\*/);
await client.close();
});
+
+test("server: the sweep prompt mandates linked citations and subagent batches", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ // Feature 1: linked citation form (not the old "title + [mm:ss]").
+ assert.match(text, /\[title @ mm:ss\]\(<moment url>\)/);
+ assert.ok(!/title \+ \[mm:ss\]/.test(text), "old bare citation form is gone");
+ // Feature 2: subagent map-reduce + inline fallback.
+ assert.match(text, /Spawn a subagent/);
+ assert.match(text, /Task tool/);
+ assert.match(text, /Fallback/);
+ await client.close();
+});
+
+test("server: the sweep prompt accepts a link= seed and drives open_link", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { link: "https://site.example/?q=coffee" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /open_link/);
+ assert.match(text, /apply:true/);
+ assert.match(text, /https:\/\/site\.example\/\?q=coffee/);
+ // A link-only sweep needs no query argument.
+ assert.match(text, /share link/);
+ await client.close();
+});
+
+// ─── link_style:"base" — compact stamps, moment_base, max_lines, worklist trim ───
+
+test("server: get_transcripts link_style base emits moment_base + compact stamps", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee", link_style: "base" },
+ });
+ const out = firstText(res);
+ // StubSource has no viewer origin → the platform (YouTube-param) base.
+ assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m);
+ assert.match(out, /\[1:40\|100\] here is the coffee moment/);
+ assert.ok(!out.includes("](http"), "no full inline links in base excerpts");
+ assert.match(out, /<moment_base><seconds>/, "the expansion note is present");
+ await client.close();
+});
+
+test("server: get_transcripts default output stays inline-linked (regression pin)", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee" },
+ });
+ const out = firstText(res);
+ assert.ok(out.includes("](https://example.test/b1?t=100s)"), out);
+ assert.ok(!out.includes("moment_base"), "no base artifacts in inline mode");
+ await client.close();
+});
+
+test("server: get_transcripts base full transcript uses compact stamps + header base", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], link_style: "base" },
+ });
+ const out = firstText(res);
+ assert.match(out, /\[0:00\|0\] intro chatter/);
+ assert.match(out, /- moment_base: https:\/\/example\.test\/b1\?t=$/m);
+ assert.ok(!out.includes("](http"), "no inline links in base full transcripts");
+ await client.close();
+});
+
+test("server: get_transcripts max_lines caps the merged excerpt lines", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "get_transcripts",
+ arguments: { video_ids: ["b1"], query: "coffee", max_lines: 1 },
+ });
+ const out = firstText(res);
+ assert.match(out, /1 matching line\(s\), windowed/);
+ assert.ok(out.includes("right before the moment"), "the earliest line is kept");
+ assert.ok(!out.includes("right after the moment"), "later lines are dropped");
+ await client.close();
+});
+
+test("server: search_transcripts link_style base emits moment_base + compact snippets", async () => {
+ const client = await connectClient(new StubSource());
+ const res = await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", link_style: "base", limit: 20 },
+ });
+ const out = firstText(res);
+ assert.match(out, /- moment_base: https:\/\/example\.test\/a1\?t=$/m);
+ assert.ok(out.includes(" - [0:10|10] i love coffee"), out);
+ assert.ok(!out.includes("](http"), "no full inline links in base snippets");
+ assert.match(out, /<moment_base><seconds>/, "the expansion note is present");
+ await client.close();
+});
+
+test("server: search_transcripts worklist trims the source line (and no moment_base)", async () => {
+ const client = await connectClient(new StubSource());
+ const withSnips = firstText(
+ await client.callTool({
+ name: "search_transcripts",
+ arguments: { query: "coffee", limit: 5 },
+ }),
+ );
+ assert.match(withSnips, /- source: /, "source line present by default");
+
+ const worklist = firstText(
+ await client.callTool({
+ name: "search_transcripts",
+ arguments: {
+ query: "coffee",
+ limit: 5,
+ include_snippets: false,
+ link_style: "base",
+ },
+ }),
+ );
+ assert.ok(!worklist.includes("- source:"), "worklist mode drops the source line");
+ assert.ok(!worklist.includes("moment_base"), "…and emits no moment_base either");
+ await client.close();
+});
+
+test("server: the sweep prompt drives dumb extractors on the cheap model", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /haiku/, "the default parse model is requested");
+ assert.match(text, /DUMB EXTRACTOR/);
+ assert.match(text, /link_style/);
+ assert.match(text, /moment_base/);
+ assert.match(text, /\[mm:ss\|seconds\]/);
+ assert.match(text, /40 lines/, "the extractor's hard line budget");
+ assert.match(text, /<moment_base><seconds>/, "the expansion rule is spelled out");
+ await client.close();
+});
+
+test("server: the sweep prompt honors parse_model", async () => {
+ const client = await connectClient(new StubSource());
+ const got = await client.getPrompt({
+ name: "sweep",
+ arguments: { query: "k cups", channel: "chan-a", parse_model: "sonnet" },
+ });
+ const text = (got.messages[0].content as { text: string }).text;
+ assert.match(text, /sonnet/);
+ assert.ok(!/haiku/.test(text), "the default model name is fully replaced");
+ await client.close();
+});
diff --git a/mcp/src/search.ts b/mcp/src/search.ts
@@ -1,5 +1,16 @@
import { formatDuration } from "yt-dlp-transcript-common/lib/format";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type { Cue } from "yt-dlp-transcript-common/lib/vtt";
+import type { Platform } from "yt-dlp-transcript-common/lib/platform";
+import {
+ forEachLeaf,
+ isLeaf,
+ isLeafActive,
+ isNodeActive,
+ type GroupNode,
+ type LayerScope,
+ type QueryNode,
+} from "yt-dlp-transcript-common/lib/searchQuery";
import {
matchAliases,
type SearchAlias,
@@ -14,7 +25,7 @@ import {
resolveChannelGroupId,
type ChannelGroup,
} from "yt-dlp-transcript-common/lib/channelGroups";
-import type { ChannelRef, ShardSource } from "./source";
+import type { ChannelRef, ShardSource, VideoAvailability } from "./source";
export type Snippet = { clock: string; seconds: number; text: string };
@@ -126,9 +137,17 @@ export async function resolveSelectedChannels(
export type SearchHit = {
videoId: string;
+ // The video's slug (`<channelSlug>/<id>`) — the archilyzer viewer's `?v=`
+ // value, used to build a moment deep link (momentUrl).
+ slug: string;
channelSlug: string;
channelName: string;
siteTitle?: string;
+ // The public origin of the site that owns this video (a member url in hub
+ // mode, the site base for a single remote). Used with `slug` for a viewer
+ // link; absent for a local source (→ platform-link fallback).
+ siteUrl?: string;
+ platform?: Platform;
title: string;
uploadDate: string;
webpageUrl?: string;
@@ -325,9 +344,12 @@ export async function searchTranscripts(
if (matches === 0 && !titleHit) continue;
all.push({
videoId: rec.id,
+ slug: rec.slug,
channelSlug: ch.slug,
channelName: ch.name,
...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}),
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(rec.platform ? { platform: rec.platform } : {}),
title: rec.title,
uploadDate: rec.uploadDate,
webpageUrl: rec.webpageUrl,
@@ -376,6 +398,14 @@ export function getWindowedTranscript(
after?: number;
maxCues?: number;
timestamps?: boolean;
+ // Optional formatter for the bracketed stamp's CONTENTS (e.g. an inline
+ // Markdown link "[m:ss](url)" or the compact base form "m:ss|156"), given
+ // the line's clock + start seconds. Bare clock when omitted.
+ stamp?: (clock: string, seconds: number) => string;
+ // Cap on merged excerpt lines emitted for this video (default
+ // WINDOW_LINE_CAP). The earliest lines are kept; `maxCues` above stays the
+ // per-window bound.
+ maxLines?: number;
} = {},
): { lines: string[]; matchCount: number } {
const cues = record.cues ?? [];
@@ -392,11 +422,13 @@ export function getWindowedTranscript(
maxCues: opts.maxCues,
}),
);
- merged = mergeSnippets(merged, win, WINDOW_LINE_CAP);
+ merged = mergeSnippets(merged, win, opts.maxLines ?? WINDOW_LINE_CAP);
}
- const lines = merged.map((s) =>
- timestamps ? `[${s.clock}] ${s.text}` : s.text,
- );
+ const lines = merged.map((s) => {
+ if (!timestamps) return s.text;
+ const stamp = opts.stamp ? opts.stamp(s.clock, s.seconds) : s.clock;
+ return `[${stamp}] ${s.text}`;
+ });
return { lines, matchCount };
}
@@ -445,3 +477,419 @@ export async function findVideo(
}
return null;
}
+
+// ─── Full-fidelity spec engine: query tree + filter predicate ───
+//
+// `runSearchSpec` is the per-record evaluator that backs the `open_link` /
+// `sweep link=` flows: it honors everything an archilyzer share link can carry
+// (a composite `qt=` query tree over any mix of scopes, plus the `fc/ft/fa/fav/
+// fdf/fdt` filters), while keeping the same paging / total / truncation
+// contract as `searchTranscripts`. The plain `search_transcripts` path stays on
+// the simpler `searchTranscripts` scanner above (a trivial single-leaf query),
+// so today's callers/tests are unaffected.
+//
+// It mirrors the browser's tree algebra (searchEval.ts) and filter predicate
+// (SearchSessionContext.tsx `passesFilter`) — evaluated per record to a boolean
+// instead of over slug sets — and the per-scope text extraction of
+// searchPipeline.ts.
+
+// The positive share-filter selection (parseShareV1), as a predicate input.
+export type SearchFilters = {
+ // ft — video / livestream types kept.
+ videos: boolean;
+ livestreams: boolean;
+ // fa — all-ages / age-restricted kept.
+ allAges: boolean;
+ restricted: boolean;
+ // fav — availability three-way (available / unlisted / deleted) kept.
+ available: boolean;
+ unlisted: boolean;
+ deleted: boolean;
+ // fdf / fdt — inclusive upload-date bounds, "YYYYMMDD".
+ dateFrom?: string;
+ dateTo?: string;
+};
+
+export type SearchSpec = {
+ // The root of the composite query (a `qt=` tree, or a synthesized single leaf
+ // for a legacy `q`/`re` link).
+ tree: GroupNode;
+ // The decoded share filters, or null for an unfiltered search.
+ filters?: SearchFilters | null;
+ // Curated aliases to expand transcripts-scope leaves through (as in the plain
+ // path). Empty/omitted → no expansion.
+ aliases?: SearchAlias[];
+ // Expand transcripts leaves via aliases (default true; skipped per-leaf for a
+ // regex leaf).
+ useAliases?: boolean;
+};
+
+// A snippet tagged with the scope it came from, carrying the seconds needed to
+// build a moment link. `seconds` is 0 for the non-timed scopes (metadata /
+// description / tags).
+export type ScopedSnippet = {
+ scope: LayerScope;
+ track?: string;
+ clock: string;
+ seconds: number;
+ text: string;
+};
+
+export type SpecHit = {
+ videoId: string;
+ slug: string;
+ channelSlug: string;
+ channelName: string;
+ siteTitle?: string;
+ siteUrl?: string;
+ platform?: Platform;
+ title: string;
+ uploadDate: string;
+ webpageUrl?: string;
+ matches: number;
+ snippets: ScopedSnippet[];
+};
+
+export type SpecResult = {
+ hits: SpecHit[];
+ total: number;
+ offset: number;
+ limit: number;
+ hasMore: boolean;
+ firedAliases: SearchAlias[];
+ scanned: { channels: number; pages: number };
+ truncated: boolean;
+};
+
+type LeafMatcher = { scope: LayerScope; test: Matcher };
+
+// Compile a matcher per active leaf. Transcripts leaves are alias-aware (unless
+// they're regex); every other scope matches plain-substring / regex only. Fired
+// aliases are unioned for the caller to report.
+function buildLeafMatchers(
+ root: QueryNode,
+ aliases: SearchAlias[],
+ useAliases: boolean,
+): { matchers: Map<string, LeafMatcher>; fired: SearchAlias[] } {
+ const matchers = new Map<string, LeafMatcher>();
+ const fired: SearchAlias[] = [];
+ forEachLeaf(root, (leaf) => {
+ if (!isLeafActive(leaf)) return;
+ const aliasAware =
+ leaf.scope === "transcripts" && !leaf.useRegex && useAliases;
+ const built = buildMatcher({
+ query: leaf.query,
+ regex: leaf.useRegex,
+ useAliases: aliasAware,
+ aliases: aliasAware ? aliases : [],
+ });
+ matchers.set(leaf.id, { scope: leaf.scope, test: built.match });
+ for (const a of built.firedAliases) {
+ if (!fired.some((x) => x.id === a.id)) fired.push(a);
+ }
+ });
+ return { matchers, fired };
+}
+
+// Per-record context the tree evaluates against.
+type RecordCtx = {
+ title: string;
+ channel: string;
+ description: string;
+ tags: string;
+ cues: Cue[];
+ chatCues: Cue[];
+ snippetsPerVideo: number;
+ includeSnippets: boolean;
+};
+
+type LeafOutcome = { matched: boolean; count: number; hits: ScopedSnippet[] };
+
+function evalLeaf(leaf: QueryNode, m: LeafMatcher, ctx: RecordCtx): LeafOutcome {
+ if (!isLeaf(leaf)) return { matched: false, count: 0, hits: [] };
+ const hits: ScopedSnippet[] = [];
+ const push = (s: ScopedSnippet): void => {
+ if (ctx.includeSnippets && hits.length < ctx.snippetsPerVideo) hits.push(s);
+ };
+ let count = 0;
+ switch (m.scope) {
+ case "transcripts":
+ for (const cue of ctx.cues) {
+ if (!m.test(cue.text)) continue;
+ count++;
+ push({
+ scope: "transcripts",
+ clock: clock(cue.start),
+ seconds: cue.start,
+ text: truncate(cue.text),
+ });
+ }
+ break;
+ case "chat":
+ for (const cue of ctx.chatCues) {
+ if (!m.test(cue.text)) continue;
+ count++;
+ push({
+ scope: "chat",
+ track: "live_chat",
+ clock: clock(cue.start),
+ seconds: cue.start,
+ text: truncate(cue.text),
+ });
+ }
+ break;
+ case "metadata": {
+ const titleHit = m.test(ctx.title);
+ const channelHit = m.test(ctx.channel);
+ if (titleHit || channelHit) {
+ count++;
+ if (titleHit) {
+ push({ scope: "metadata", clock: clock(0), seconds: 0, text: truncate(ctx.title) });
+ } else {
+ push({ scope: "metadata", clock: clock(0), seconds: 0, text: `Channel: ${ctx.channel}` });
+ }
+ }
+ break;
+ }
+ case "description":
+ if (ctx.description && m.test(ctx.description)) {
+ count++;
+ push({ scope: "description", clock: clock(0), seconds: 0, text: truncate(ctx.description) });
+ }
+ break;
+ case "tags":
+ if (ctx.tags && m.test(ctx.tags)) {
+ count++;
+ push({ scope: "tags", clock: clock(0), seconds: 0, text: truncate(ctx.tags) });
+ }
+ break;
+ }
+ return { matched: count > 0, count, hits };
+}
+
+type NodeOutcome = { match: boolean; count: number; hits: ScopedSnippet[] };
+
+// Evaluate the tree against one record — mirrors searchEval's AND/OR/negate,
+// per record. A negated node contributes no hits (like the browser's `diff`).
+// An inactive (empty) subtree is identity (matches, no hits).
+function evalNode(
+ node: QueryNode,
+ matchers: Map<string, LeafMatcher>,
+ ctx: RecordCtx,
+): NodeOutcome {
+ if (!isNodeActive(node)) return { match: true, count: 0, hits: [] };
+ if (isLeaf(node)) {
+ const m = matchers.get(node.id);
+ if (!m) return { match: true, count: 0, hits: [] };
+ const r = evalLeaf(node, m, ctx);
+ if (node.negate) return { match: !r.matched, count: 0, hits: [] };
+ return {
+ match: r.matched,
+ count: node.contributeHits ? r.count : 0,
+ hits: node.contributeHits ? r.hits : [],
+ };
+ }
+ const active = node.children.filter(isNodeActive);
+ if (active.length === 0) return { match: true, count: 0, hits: [] };
+ if (node.op === "AND") {
+ let allMatch = true;
+ let count = 0;
+ const hits: ScopedSnippet[] = [];
+ for (const c of active) {
+ const r = evalNode(c, matchers, ctx);
+ if (!r.match) {
+ allMatch = false;
+ break;
+ }
+ count += r.count;
+ hits.push(...r.hits);
+ }
+ const match = node.negate ? !allMatch : allMatch;
+ return match && !node.negate
+ ? { match, count, hits }
+ : { match, count: 0, hits: [] };
+ }
+ // OR
+ let any = false;
+ let count = 0;
+ const hits: ScopedSnippet[] = [];
+ for (const c of active) {
+ const r = evalNode(c, matchers, ctx);
+ if (r.match) {
+ any = true;
+ count += r.count;
+ hits.push(...r.hits);
+ }
+ }
+ const match = node.negate ? !any : any;
+ return match && !node.negate
+ ? { match, count, hits }
+ : { match, count: 0, hits: [] };
+}
+
+// The share-filter predicate over a transcript record + its availability.
+// Mirrors SearchSessionContext.tsx `passesFilter` exactly.
+function passesFilters(
+ rec: TranscriptDetail,
+ f: SearchFilters,
+ avail: VideoAvailability | undefined,
+): boolean {
+ // ft — type
+ if (rec.isLivestream ? !f.livestreams : !f.videos) return false;
+ // fa — audience
+ if (rec.ageRestricted ? !f.restricted : !f.allAges) return false;
+ // fav — availability three-way
+ if (avail?.isDeleted) {
+ if (!f.deleted) return false;
+ } else if (avail?.isUnlisted) {
+ if (!f.unlisted) return false;
+ } else if (!f.available) {
+ return false;
+ }
+ // fdf / fdt — upload-date range (lexicographic on YYYYMMDD)
+ if (f.dateFrom && rec.uploadDate < f.dateFrom) return false;
+ if (f.dateTo && rec.uploadDate > f.dateTo) return false;
+ return true;
+}
+
+// True when the fav filter could exclude something (so availability must be
+// fetched). If every availability bucket is kept, there's nothing to look up.
+function needsAvailability(f: SearchFilters | null | undefined): boolean {
+ return !!f && (!f.available || !f.unlisted || !f.deleted);
+}
+
+// A small lazy live-chat fetcher: per-channel subs manifest + page caches, so a
+// chat-scope leaf only pulls the subs shards it actually touches.
+function makeChatFetcher(source: ShardSource) {
+ const manifests = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsManifest"]>>>>();
+ const pages = new Map<string, Promise<Awaited<ReturnType<ShardSource["subsPage"]>>>>();
+ return async function chatCues(ch: ChannelRef, rec: TranscriptDetail): Promise<Cue[]> {
+ let mp = manifests.get(ch.key);
+ if (!mp) {
+ mp = source.subsManifest(ch).catch(() => null);
+ manifests.set(ch.key, mp);
+ }
+ const manifest = await mp;
+ if (!manifest) return [];
+ const page = manifest.slugToPage[rec.id];
+ if (page === undefined) return [];
+ const pk = `${ch.key}:${page}`;
+ let pp = pages.get(pk);
+ if (!pp) {
+ pp = source.subsPage(ch, page).catch(() => []);
+ pages.set(pk, pp);
+ }
+ const records = await pp;
+ const found = records.find((r) => r.slug === rec.slug || r.id === rec.id);
+ const lc = found?.tracks?.live_chat;
+ return Array.isArray(lc) ? lc : [];
+ };
+}
+
+// Run a full-fidelity spec over the given (already-resolved) channel set. Same
+// paging/total/truncation semantics as searchTranscripts.
+export async function runSearchSpec(
+ source: ShardSource,
+ channels: ChannelRef[],
+ spec: SearchSpec,
+ opts: {
+ offset?: number;
+ limit?: number;
+ includeSnippets?: boolean;
+ maxPages?: number;
+ snippetsPerVideo?: number;
+ } = {},
+): Promise<SpecResult> {
+ const limit = opts.limit ?? 20;
+ const offset = Math.max(0, opts.offset ?? 0);
+ const maxPages = opts.maxPages ?? MAX_PAGES;
+ const includeSnippets = opts.includeSnippets !== false;
+ const snippetsPerVideo = opts.snippetsPerVideo ?? 4;
+ const filters = spec.filters ?? null;
+
+ const { matchers, fired } = buildLeafMatchers(
+ spec.tree,
+ spec.aliases ?? [],
+ spec.useAliases !== false,
+ );
+ const wantsChat = [...matchers.values()].some((m) => m.scope === "chat");
+ const chatCuesFor = wantsChat ? makeChatFetcher(source) : null;
+ const availability = needsAvailability(filters)
+ ? await source.availabilityMap()
+ : null;
+
+ const all: SpecHit[] = [];
+ let pagesScanned = 0;
+ let channelsScanned = 0;
+ let truncated = false;
+
+ outer: for (const ch of channels) {
+ let manifest;
+ try {
+ manifest = await source.transcriptsManifest(ch);
+ } catch {
+ continue;
+ }
+ channelsScanned++;
+ for (let page = 0; page < manifest.pageCount; page++) {
+ if (pagesScanned >= maxPages) {
+ truncated = true;
+ break outer;
+ }
+ let records: TranscriptDetail[];
+ try {
+ records = await source.transcriptPage(ch, page);
+ } catch {
+ continue;
+ }
+ pagesScanned++;
+ for (const rec of records) {
+ if (filters && !passesFilters(rec, filters, availability?.get(rec.slug))) {
+ continue;
+ }
+ const ctx: RecordCtx = {
+ title: rec.title ?? "",
+ channel: rec.channel ?? ch.name,
+ description: rec.description ?? "",
+ tags: (rec.tags ?? []).join(", "),
+ cues: rec.cues ?? [],
+ chatCues: chatCuesFor ? await chatCuesFor(ch, rec) : [],
+ snippetsPerVideo,
+ includeSnippets,
+ };
+ const r = evalNode(spec.tree, matchers, ctx);
+ if (!r.match) continue;
+ all.push({
+ videoId: rec.id,
+ slug: rec.slug,
+ channelSlug: ch.slug,
+ channelName: ch.name,
+ ...(ch.siteTitle ? { siteTitle: ch.siteTitle } : {}),
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(rec.platform ? { platform: rec.platform } : {}),
+ title: rec.title,
+ uploadDate: rec.uploadDate,
+ webpageUrl: rec.webpageUrl,
+ matches: r.count || 1,
+ snippets: r.hits,
+ });
+ if (all.length >= HARD_VIDEO_CAP) {
+ truncated = true;
+ break outer;
+ }
+ }
+ }
+ }
+
+ const total = all.length;
+ return {
+ hits: all.slice(offset, offset + limit),
+ total,
+ offset,
+ limit,
+ hasMore: offset + limit < total,
+ firedAliases: fired,
+ scanned: { channels: channelsScanned, pages: pagesScanned },
+ truncated,
+ };
+}
diff --git a/mcp/src/server.ts b/mcp/src/server.ts
@@ -7,6 +7,8 @@ import {
} from "@modelcontextprotocol/sdk/types.js";
import { transcriptToMarkdown } from "yt-dlp-transcript-common/lib/transcriptToMarkdown";
import { formatDate } from "yt-dlp-transcript-common/lib/format";
+import { momentUrl, momentBaseUrl } from "yt-dlp-transcript-common/lib/momentUrl";
+import type { Platform } from "yt-dlp-transcript-common/lib/platform";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
import {
sortGroups,
@@ -14,7 +16,13 @@ import {
FALLBACK_GROUP,
type ChannelGroup,
} from "yt-dlp-transcript-common/lib/channelGroups";
-import { HubSource, type ChannelRef, type HubSite, type ShardSource } from "./source";
+import {
+ HubSource,
+ RemoteSource,
+ type ChannelRef,
+ type HubSite,
+ type ShardSource,
+} from "./source";
import type { SourceSpec } from "./sources";
import type { SourceController } from "./sourceController";
import {
@@ -22,8 +30,89 @@ import {
findVideo,
buildMatcher,
getWindowedTranscript,
+ runSearchSpec,
type SearchResult,
+ type SpecHit,
+ type ScopedSnippet,
} from "./search";
+import {
+ decodeShareLink,
+ applyLinkOverrides,
+ renderQueryTree,
+ describeFilters,
+ type DecodedLink,
+ type LinkOverrides,
+} from "./shareLink";
+
+// The building blocks momentUrl needs from a hit (viewer origin comes from the
+// per-video siteUrl, else the source's single public origin).
+type LinkableHit = {
+ slug: string;
+ siteUrl?: string;
+ webpageUrl?: string;
+ platform?: Platform;
+};
+
+// A moment deep link for one cited second of a hit, or null when neither a
+// viewer nor a platform link can be built (e.g. a local source + no webpage url).
+function momentLinkFor(
+ source: ShardSource,
+ h: LinkableHit,
+ seconds: number,
+): string | null {
+ return momentUrl({
+ siteOrigin: h.siteUrl ?? source.publicOrigin(),
+ slug: h.slug,
+ seconds,
+ webpageUrl: h.webpageUrl,
+ platform: h.platform,
+ });
+}
+
+// A `[m:ss]` timestamp rendered as a Markdown link when we can build one, else
+// bare. Used inside a `- [ … ] text` snippet line → `- [[m:ss](url)] text`.
+function stampMarkup(
+ source: ShardSource,
+ h: LinkableHit,
+ clock: string,
+ seconds: number,
+): string {
+ const url = momentLinkFor(source, h, seconds);
+ return url ? `[${clock}](${url})` : clock;
+}
+
+// The requested timestamp-link form: "base" is the compact agent-pipeline form
+// (one `moment_base` per video + `[m:ss|seconds]` lines); anything else — the
+// default — is the full inline Markdown links.
+function linkStyleOf(args: Record<string, unknown>): "inline" | "base" {
+ return args.link_style === "base" ? "base" : "inline";
+}
+
+// The compact stamp contents for link_style:"base": `m:ss|156`. The integer is
+// floored exactly like momentUrl floors its seconds, so appending it to the
+// video's moment_base cites the identical second the inline link would.
+function baseStamp(clock: string, seconds: number): string {
+ return `${clock}|${Math.floor(seconds)}`;
+}
+
+// The appendable moment base for a hit (mirrors momentLinkFor), or null when
+// none can be built — no viewer origin AND the platform's time param doesn't
+// take raw seconds (Twitch) or doesn't exist (Rumble/Kick/no URL).
+function momentBaseFor(source: ShardSource, h: LinkableHit): string | null {
+ return momentBaseUrl({
+ siteOrigin: h.siteUrl ?? source.publicOrigin(),
+ slug: h.slug,
+ webpageUrl: h.webpageUrl,
+ platform: h.platform,
+ });
+}
+
+// Footer note stating the expansion rule for link_style:"base" output.
+const BASE_EXPANSION_NOTE =
+ "link_style base: full moment link = `<moment_base><seconds>` — append the " +
+ "integer after the `|` to that video's moment_base, e.g. `[title @ 2:36]" +
+ "(<moment_base>156)`; a video with no moment_base line → cite its source " +
+ "url plain";
type ToolResult = {
content: { type: "text"; text: string }[];
@@ -127,6 +216,18 @@ const TOOLS = [
"Max shard pages to scan before stopping (default 400). Reaching " +
"it marks coverage partial.",
},
+ link_style: {
+ type: "string",
+ enum: ["inline", "base"],
+ description:
+ "Timestamp link form (default 'inline': every [m:ss] is a full " +
+ "Markdown moment link). 'base' is a compact agent-pipeline form: " +
+ "each video gets one '- moment_base:' header line (a URL ending " +
+ "in 't=') and snippet stamps become [m:ss|<seconds>]; expand to a " +
+ "full link by appending the integer seconds to the moment_base " +
+ "([title @ m:ss](<moment_base><seconds>)). A video with no " +
+ "buildable base omits the line — cite its source url plain.",
+ },
},
required: ["query"],
additionalProperties: false,
@@ -209,6 +310,25 @@ const TOOLS = [
type: "number",
description: "Seconds of context after each match (default 30).",
},
+ link_style: {
+ type: "string",
+ enum: ["inline", "base"],
+ description:
+ "Timestamp link form (default 'inline': every [m:ss] is a full " +
+ "Markdown moment link). 'base' is a compact agent-pipeline form: " +
+ "each video gets one '- moment_base:' header line (a URL ending " +
+ "in 't=') and line stamps become [m:ss|<seconds>]; expand to a " +
+ "full link by appending the integer seconds to the moment_base " +
+ "([title @ m:ss](<moment_base><seconds>)). A video with no " +
+ "buildable base omits the line — cite its source url plain.",
+ },
+ max_lines: {
+ type: "number",
+ description:
+ "Cap on merged excerpt lines per video when a query is given " +
+ "(default 200; the earliest lines are kept). Ignored without a " +
+ "query.",
+ },
},
required: ["video_ids"],
additionalProperties: false,
@@ -291,6 +411,103 @@ const TOOLS = [
"--local flag or env) and clear the persisted selection.",
inputSchema: { type: "object", properties: {}, additionalProperties: false },
},
+ {
+ name: "open_link",
+ description:
+ "Paste an archilyzer viewer **share link** (origin + a `qt=` query tree + " +
+ "`fc/ft/fa/fav/fdf/fdt` filters) to re-run that exact search here — at full " +
+ "fidelity, with clickable timestamped result links. Two-phase: by default " +
+ "(apply:false) it PREVIEWS — decodes the link and returns a plan (resolved " +
+ "source, the query tree rendered readably, every active filter, the channel " +
+ "scope validated against the live corpus, and anything ignored, e.g. the " +
+ "vestigial `fk` tracks) WITHOUT switching source or searching. Adjust in " +
+ "natural language by re-calling with `overrides` (e.g. clear_availability " +
+ "to drop the availability filter, channels to re-scope, query to replace " +
+ "the search). When the plan looks right, call again with apply:true: it " +
+ "switches the active source to the link's origin (hub or single-site, " +
+ "auto-detected) and returns the first page of linked results. Read-only.",
+ inputSchema: {
+ type: "object",
+ properties: {
+ link: {
+ type: "string",
+ description: "The archilyzer viewer share URL to decode and run.",
+ },
+ apply: {
+ type: "boolean",
+ description:
+ "false (default) previews the plan only; true switches source and " +
+ "runs the search.",
+ },
+ overrides: {
+ type: "object",
+ description:
+ "Structured edits to the decoded search, so a natural-language " +
+ "adjustment can be honored by re-calling with the changed facet.",
+ properties: {
+ clear_filters: { type: "boolean", description: "Drop all filters." },
+ clear_availability: {
+ type: "boolean",
+ description: "Remove the availability (deleted/unlisted) filter.",
+ },
+ clear_type: {
+ type: "boolean",
+ description: "Remove the video/livestream type filter.",
+ },
+ clear_age: {
+ type: "boolean",
+ description: "Remove the all-ages/age-restricted filter.",
+ },
+ clear_dates: {
+ type: "boolean",
+ description: "Remove the upload-date range filter.",
+ },
+ clear_channels: {
+ type: "boolean",
+ description: "Widen the channel scope to the whole corpus.",
+ },
+ channels: {
+ type: "array",
+ items: { type: "string" },
+ description: "Replace the channel scope with these channel names.",
+ },
+ date_from: {
+ type: "string",
+ description: "Set the inclusive lower upload-date bound (YYYYMMDD).",
+ },
+ date_to: {
+ type: "string",
+ description: "Set the inclusive upper upload-date bound (YYYYMMDD).",
+ },
+ query: {
+ type: "string",
+ description: "Replace the whole query with this single term.",
+ },
+ regex: {
+ type: "boolean",
+ description: "Treat the override query as a regex.",
+ },
+ query_scope: {
+ type: "string",
+ enum: ["transcripts", "chat", "metadata", "description", "tags"],
+ description: "Scope for the override query (default transcripts).",
+ },
+ },
+ additionalProperties: false,
+ },
+ limit: {
+ type: "number",
+ description: "Results per page when apply:true (default 20).",
+ },
+ offset: {
+ type: "number",
+ description: "Skip this many matches when apply:true (default 0).",
+ },
+ },
+ required: ["link"],
+ additionalProperties: false,
+ },
+ },
];
// The slice of SourceController the server needs. A bare ShardSource is wrapped
@@ -378,6 +595,8 @@ export function createServer(sourceOrController: ShardSource | SourceController)
return await handleUseSource(controller, args);
case "reset_source":
return await handleResetSource(controller);
+ case "open_link":
+ return await handleOpenLink(controller, args);
default:
return errorText(`unknown tool: ${name}`);
}
@@ -465,6 +684,8 @@ async function handleSearch(
): Promise<ToolResult> {
const query = String(args.query ?? "").trim();
if (!query) return errorText("query is required");
+ const includeSnippets = args.include_snippets !== false;
+ const base = linkStyleOf(args) === "base";
const result = await searchTranscripts(source, {
query,
channel: typeof args.channel === "string" ? args.channel : undefined,
@@ -474,7 +695,7 @@ async function handleSearch(
regex: args.regex === true,
limit: typeof args.limit === "number" ? args.limit : undefined,
offset: typeof args.offset === "number" ? args.offset : undefined,
- includeSnippets: args.include_snippets !== false,
+ includeSnippets,
useAliases: args.use_aliases !== false,
maxPages: typeof args.max_pages === "number" ? args.max_pages : undefined,
});
@@ -502,17 +723,29 @@ async function handleSearch(
return text(`${head}${footer}`);
}
const blocks = result.hits.map((h) => {
+ // Worklist mode (include_snippets:false) trims the source line too — the
+ // caller only wants ids/titles/counts, so no per-video URLs at all.
+ const baseUrl = base && includeSnippets ? momentBaseFor(source, h) : null;
const head =
`### ${h.title}\n` +
`- video_id: ${h.videoId} | channel: ${h.channelName}` +
(h.siteTitle ? ` | site: ${h.siteTitle}` : "") +
` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` +
- (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "");
- const snips = h.snippets.map((s) => ` - [${s.clock}] ${s.text}`).join("\n");
+ (includeSnippets && h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "") +
+ (baseUrl ? `\n- moment_base: ${baseUrl}` : "");
+ const snips = h.snippets
+ .map((s) =>
+ base
+ ? ` - [${baseStamp(s.clock, s.seconds)}] ${s.text}`
+ : ` - [${stampMarkup(source, h, s.clock, s.seconds)}] ${s.text}`,
+ )
+ .join("\n");
return snips ? `${head}\n${snips}` : head;
});
+ // Worklist mode has no stamps (or bases) to expand — skip the note too.
+ const baseNote = base && includeSnippets ? `\n\n(${BASE_EXPANSION_NOTE})` : "";
return text(
- `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}`,
+ `${result.total} video(s) matching "${query}":\n\n${blocks.join("\n\n")}${footer}${baseNote}`,
);
}
@@ -594,6 +827,11 @@ async function handleGetTranscripts(
const before = typeof args.before === "number" ? args.before : 30;
const after = typeof args.after === "number" ? args.after : 30;
+ const base = linkStyleOf(args) === "base";
+ const maxLinesArg =
+ typeof args.max_lines === "number" ? Math.floor(args.max_lines) : undefined;
+ const maxLines =
+ maxLinesArg !== undefined && maxLinesArg >= 1 ? maxLinesArg : undefined;
const blocks: string[] = [];
const missing: string[] = [];
@@ -604,18 +842,36 @@ async function handleGetTranscripts(
continue;
}
const { record, ch } = found;
+ const link: LinkableHit = {
+ slug: record.slug,
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
+ ...(record.platform ? { platform: record.platform } : {}),
+ };
+ const baseUrl = base ? momentBaseFor(source, link) : null;
+ // The bracketed stamp's contents: the full inline Markdown link (byte-
+ // identical to the pre-link_style output), or the compact base form.
+ const stamp = base
+ ? baseStamp
+ : (clock: string, seconds: number): string => {
+ const url = momentLinkFor(source, link, seconds);
+ return url ? `[${clock}](${url})` : clock;
+ };
const head =
`## ${record.title || id}\n` +
`- video_id: ${id} | channel: ${ch.name}` +
(ch.siteTitle ? ` | site: ${ch.siteTitle}` : "") +
` | uploaded: ${formatDate(record.uploadDate)}` +
- (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "");
+ (record.webpageUrl ? `\n- source: ${record.webpageUrl}` : "") +
+ (baseUrl ? `\n- moment_base: ${baseUrl}` : "");
if (matcher) {
const { lines, matchCount } = getWindowedTranscript(record, matcher.match, {
before,
after,
timestamps,
+ stamp,
+ maxLines,
});
const body =
matchCount === 0
@@ -623,10 +879,21 @@ async function handleGetTranscripts(
: `_(${matchCount} matching line(s), windowed)_\n${lines.join("\n")}`;
blocks.push(`${head}\n\n${body}`);
} else {
- const md = transcriptToMarkdown(record, {
- timestamps,
- includeTags: true,
- });
+ const md = transcriptToMarkdown(
+ record,
+ base
+ ? {
+ timestamps,
+ includeTags: true,
+ stampForCue: baseStamp,
+ extraMeta: baseUrl ? [`moment_base: ${baseUrl}`] : undefined,
+ }
+ : {
+ timestamps,
+ includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
+ },
+ );
blocks.push(md.trim());
}
}
@@ -638,6 +905,7 @@ async function handleGetTranscripts(
}
if (missing.length > 0) notes.push(`not found: ${missing.join(", ")}`);
if (dropped > 0) notes.push(`${dropped} extra id(s) beyond the 20-cap dropped`);
+ if (base) notes.push(BASE_EXPANSION_NOTE);
const footer = notes.length > 0 ? `\n\n(${notes.join("; ")})` : "";
if (blocks.length === 0) {
@@ -658,9 +926,17 @@ async function handleGetTranscript(
typeof args.channel === "string" ? args.channel : undefined,
);
if (!found) return errorText(`video not found: ${videoId}`);
- const md = transcriptToMarkdown(found.record, {
+ const { record, ch } = found;
+ const link: LinkableHit = {
+ slug: record.slug,
+ ...(ch.siteUrl ? { siteUrl: ch.siteUrl } : {}),
+ ...(record.webpageUrl ? { webpageUrl: record.webpageUrl } : {}),
+ ...(record.platform ? { platform: record.platform } : {}),
+ };
+ const md = transcriptToMarkdown(record, {
timestamps: args.timestamps !== false,
includeTags: true,
+ linkForCue: (seconds) => momentLinkFor(source, link, seconds),
});
return text(md);
}
@@ -896,6 +1172,273 @@ async function handleResetSource(
);
}
+// ─── open_link: paste a share link → preview, adjust, apply ───
+
+// Map the snake_cased tool `overrides` object onto the LinkOverrides shape.
+function parseOverrides(raw: unknown): LinkOverrides | undefined {
+ if (!raw || typeof raw !== "object") return undefined;
+ const o = raw as Record<string, unknown>;
+ const bool = (k: string): boolean | undefined =>
+ o[k] === true ? true : undefined;
+ const str = (k: string): string | undefined =>
+ typeof o[k] === "string" ? (o[k] as string) : undefined;
+ const scope = str("query_scope");
+ return {
+ clearFilters: bool("clear_filters"),
+ clearAvailability: bool("clear_availability"),
+ clearType: bool("clear_type"),
+ clearAge: bool("clear_age"),
+ clearDates: bool("clear_dates"),
+ clearChannels: bool("clear_channels"),
+ channels: strArray(o.channels),
+ dateFrom: str("date_from"),
+ dateTo: str("date_to"),
+ query: str("query"),
+ regex: o.regex === true ? true : undefined,
+ queryScope:
+ scope === "transcripts" ||
+ scope === "chat" ||
+ scope === "metadata" ||
+ scope === "description" ||
+ scope === "tags"
+ ? scope
+ : undefined,
+ };
+}
+
+type OriginProbe = { kind: "hub" | "remote"; siteCount?: number; channelCount?: number };
+
+// Probe an origin's corpus.json to classify it as a federated hub (has a
+// `sites[]` array) or a single site (has `channels[]`). Transient — no source is
+// committed. Defaults to single-site when the probe can't be read.
+async function probeOrigin(origin: string): Promise<OriginProbe> {
+ try {
+ const res = await fetch(`${origin}/corpus.json`);
+ if (res.ok) {
+ const j = (await res.json()) as {
+ sites?: unknown[];
+ channels?: unknown[];
+ };
+ if (Array.isArray(j.sites)) return { kind: "hub", siteCount: j.sites.length };
+ if (Array.isArray(j.channels)) {
+ return { kind: "remote", channelCount: j.channels.length };
+ }
+ }
+ } catch {
+ // unreachable — fall through to the single-site default
+ }
+ return { kind: "remote" };
+}
+
+type ChannelScope = {
+ channels: ChannelRef[];
+ all: boolean;
+ matched: string[];
+ unknown: string[];
+};
+
+// Resolve a decoded link's `fc` channel NAMES against a live source's channels
+// (by display name or slug, case-insensitive). No channel filter → whole corpus.
+async function resolveLinkChannels(
+ source: ShardSource,
+ decoded: DecodedLink,
+): Promise<ChannelScope> {
+ const all = await source.listChannels();
+ if (!decoded.hasChannelFilter) {
+ return { channels: all, all: true, matched: [], unknown: [] };
+ }
+ const want = decoded.channelNames.map((n) => n.toLowerCase());
+ const matches = (c: ChannelRef): boolean =>
+ want.includes(c.name.toLowerCase()) || want.includes(c.slug.toLowerCase());
+ const channels = all.filter(matches);
+ const matched = channels.map((c) => c.name);
+ const unknown = decoded.channelNames.filter(
+ (n) =>
+ !all.some(
+ (c) =>
+ c.name.toLowerCase() === n.toLowerCase() ||
+ c.slug.toLowerCase() === n.toLowerCase(),
+ ),
+ );
+ return { channels, all: false, matched, unknown };
+}
+
+// Render the human-readable preview/echo plan for a decoded link.
+function buildLinkPlan(
+ decoded: DecodedLink,
+ probe: OriginProbe,
+ scope: ChannelScope,
+ applied: boolean,
+): string {
+ const lines: string[] = [];
+ lines.push(applied ? "## Applied share link" : "## Share-link preview");
+ const kindNote =
+ probe.kind === "hub"
+ ? `hub (${probe.siteCount ?? "?"} member site(s))`
+ : `single site${probe.channelCount != null ? ` (${probe.channelCount} channel(s))` : ""}`;
+ lines.push(`- Source: ${decoded.origin} — ${kindNote}`);
+ lines.push(`- Query: ${renderQueryTree(decoded.tree)} _(from ${decoded.querySource})_`);
+
+ if (scope.all) {
+ lines.push(`- Scope: whole corpus`);
+ } else {
+ const names = scope.matched.length ? scope.matched.join(", ") : "(none)";
+ lines.push(`- Scope: ${scope.channels.length} channel(s): ${names}`);
+ if (scope.unknown.length > 0) {
+ lines.push(` - ⚠ channel name(s) not found in this corpus: ${scope.unknown.join(", ")}`);
+ }
+ if (scope.channels.length === 0) {
+ lines.push(
+ ` - ⚠ the link selects 0 channels here — nothing will match. Re-call ` +
+ `with overrides.clear_channels to search the whole corpus.`,
+ );
+ }
+ }
+
+ const filterLines = describeFilters(decoded.filters);
+ if (filterLines.length > 0) {
+ lines.push(`- Filters:`);
+ for (const f of filterLines) lines.push(` - ${f}`);
+ } else {
+ lines.push(`- Filters: none`);
+ }
+
+ for (const w of decoded.warnings) lines.push(`- Note: ${w}`);
+
+ if (!applied) {
+ lines.push(
+ ``,
+ `No search run yet. Call again with apply:true to switch source and ` +
+ `search, or pass overrides to adjust first.`,
+ );
+ }
+ return lines.join("\n");
+}
+
+// A ScopedSnippet line: a linked `[m:ss]` for timed scopes, or a scope-tagged
+// note for the non-timed scopes (metadata / description / tags).
+function renderScopedSnippet(
+ source: ShardSource,
+ hit: SpecHit,
+ s: ScopedSnippet,
+): string {
+ if (s.seconds > 0) {
+ const tag = s.scope === "transcripts" ? "" : `${s.track ?? s.scope} `;
+ return ` - [${tag}${stampMarkup(source, hit, s.clock, s.seconds)}] ${s.text}`;
+ }
+ return ` - [${s.scope}] ${s.text}`;
+}
+
+// open_link results (and get_transcript) stay inline-linked for now — a
+// link_style:"base" form for them is future work.
+function renderSpecResults(source: ShardSource, hits: SpecHit[]): string {
+ return hits
+ .map((h) => {
+ const head =
+ `### ${h.title}\n` +
+ `- video_id: ${h.videoId} | channel: ${h.channelName}` +
+ (h.siteTitle ? ` | site: ${h.siteTitle}` : "") +
+ ` | uploaded: ${formatDate(h.uploadDate)} | matches: ${h.matches}` +
+ (h.webpageUrl ? `\n- source: ${h.webpageUrl}` : "");
+ const snips = h.snippets
+ .map((s) => renderScopedSnippet(source, h, s))
+ .join("\n");
+ return snips ? `${head}\n${snips}` : head;
+ })
+ .join("\n\n");
+}
+
+async function handleOpenLink(
+ controller: SourceControllerLike,
+ args: Record<string, unknown>,
+): Promise<ToolResult> {
+ const linkStr = typeof args.link === "string" ? args.link.trim() : "";
+ if (!linkStr) return errorText("link is required");
+
+ let decoded: DecodedLink;
+ try {
+ decoded = decodeShareLink(linkStr);
+ } catch (e) {
+ return errorText(`could not parse link as a URL: ${(e as Error).message}`);
+ }
+ decoded = applyLinkOverrides(decoded, parseOverrides(args.overrides));
+
+ const apply = args.apply === true;
+ const probe = await probeOrigin(decoded.origin);
+
+ if (!apply) {
+ // Preview: resolve channels against a transient source; do not commit.
+ const preview: ShardSource =
+ probe.kind === "hub"
+ ? new HubSource(decoded.origin)
+ : new RemoteSource(decoded.origin);
+ let scope: ChannelScope;
+ try {
+ scope = await resolveLinkChannels(preview, decoded);
+ } catch (e) {
+ return errorText(
+ `could not read the corpus at ${decoded.origin}: ${(e as Error).message}`,
+ );
+ }
+ return text(buildLinkPlan(decoded, probe, scope, false));
+ }
+
+ // Apply: commit the source (reusing the controller), then search.
+ const spec: SourceSpec =
+ probe.kind === "hub"
+ ? { kind: "hub", url: decoded.origin }
+ : { kind: "remote", url: decoded.origin };
+ try {
+ await controller.switchTo(spec);
+ } catch (e) {
+ return errorText(`open_link could not switch source: ${(e as Error).message}`);
+ }
+ const source = controller.current;
+
+ let scope: ChannelScope;
+ try {
+ scope = await resolveLinkChannels(source, decoded);
+ } catch (e) {
+ return errorText(
+ `switched to ${source.label} but could not read its corpus: ${(e as Error).message}`,
+ );
+ }
+ const plan = buildLinkPlan(decoded, probe, scope, true);
+
+ const limit = typeof args.limit === "number" ? args.limit : 20;
+ const offset = typeof args.offset === "number" ? args.offset : 0;
+ const result = await runSearchSpec(
+ source,
+ scope.channels,
+ {
+ tree: decoded.tree,
+ filters: decoded.filters,
+ aliases: await source.loadAliases(),
+ },
+ { limit, offset },
+ );
+
+ const aliasNote = describeFiredAliases(result.firedAliases);
+ const rangeStart = result.total === 0 ? 0 : result.offset + 1;
+ const rangeEnd = result.offset + result.hits.length;
+ const footer =
+ `\n\n(total ${result.total} match(es); showing ${rangeStart}–${rangeEnd}; ` +
+ `has_more: ${result.hasMore ? "yes" : "no"}; scanned ${result.scanned.pages} ` +
+ `page(s) across ${result.scanned.channels} channel(s)` +
+ (result.truncated ? "; coverage PARTIAL — scan hit the page/video cap" : "") +
+ (aliasNote ? `; ${aliasNote}` : "") +
+ ")";
+
+ const body =
+ result.hits.length === 0
+ ? result.total === 0
+ ? "No matches for this link's search."
+ : `No matches in this page (offset ${result.offset} is past the ${result.total} total).`
+ : `${result.total} video(s) matched:\n\n${renderSpecResults(source, result.hits)}`;
+
+ return text(`${plan}\n\n---\n\n${body}${footer}`);
+}
+
// ─── Prompts: the first-class `sweep` entry point ───
// A single slash command that drives Claude Code to run the browser's
// "corpus sweep" on plan usage: enumerate a query's full match set, batch the
@@ -912,7 +1455,21 @@ const PROMPTS = [
"on plan usage, no API key. With no scope arg it lists the channel groups " +
"and asks you to pick a group/channels (or confirm 'all') before sweeping.",
arguments: [
- { name: "query", description: "Term or phrase to sweep for.", required: true },
+ {
+ name: "query",
+ description:
+ "Term or phrase to sweep for. Optional when a `link` is given (the " +
+ "link supplies the query).",
+ required: false,
+ },
+ {
+ name: "link",
+ description:
+ "An archilyzer viewer share URL to seed the sweep from — its origin " +
+ "(source), query tree, and filters are decoded and applied via " +
+ "open_link before enumerating.",
+ required: false,
+ },
{
name: "channel",
description: "Optional channel slug/name to restrict the sweep to.",
@@ -943,6 +1500,14 @@ const PROMPTS = [
required: false,
},
{
+ name: "parse_model",
+ description:
+ "Model to request for the per-batch extractor subagents (default " +
+ "'haiku'). The extractors only quote verbatim, so the cheapest " +
+ "model wins; all synthesis stays with the orchestrator.",
+ required: false,
+ },
+ {
name: "report_path",
description: "Report file to write (default ./sweep-report.md).",
required: false,
@@ -958,7 +1523,10 @@ function argStr(args: Record<string, unknown>, key: string): string | undefined
function buildSweepPrompt(args: Record<string, unknown>) {
const query = argStr(args, "query");
- if (!query) throw new Error("sweep requires a query argument");
+ const link = argStr(args, "link");
+ if (!query && !link) {
+ throw new Error("sweep requires a query argument (or a link)");
+ }
const channel = argStr(args, "channel");
const group = argStr(args, "group");
const channelsRaw = argStr(args, "channels");
@@ -967,11 +1535,13 @@ function buildSweepPrompt(args: Record<string, unknown>) {
: [];
const directive = argStr(args, "directive") ?? "key claims & contradictions";
const batchSize = argStr(args, "batch_size") ?? "8";
+ const parseModel = argStr(args, "parse_model") ?? "haiku";
const reportPath = argStr(args, "report_path") ?? "./sweep-report.md";
+ const subject = query ? `"${query}"` : "the share link's search";
// Human-readable scope clauses + the literal search_transcripts scope args to
- // pass. When none is given, the sweep must pick-first (list + ask) rather than
- // silently scanning the whole corpus.
+ // pass. When none is given (and there's no link), the sweep must pick-first
+ // (list + ask) rather than silently scanning the whole corpus.
const scopeClauses: string[] = [];
if (channel) scopeClauses.push(`channel "${channel}"`);
if (channels.length > 0)
@@ -980,70 +1550,117 @@ function buildSweepPrompt(args: Record<string, unknown>) {
const hasScope = scopeClauses.length > 0;
const scopeArgsText = scopeClauses.join(" and ");
- const introScope = hasScope
- ? ` scoped to ${scopeArgsText}`
- : " over a scope you will confirm with me first (see step 1)";
+ const introScope = link
+ ? ` seeded from a share link (its source, query tree, and filters — see step 1)`
+ : hasScope
+ ? ` scoped to ${scopeArgsText}`
+ : " over a scope you will confirm with me first (see step 1)";
- // Build the numbered steps. With an explicit scope we search directly; with
- // none we insert a pick-first step and the Search step uses the chosen scope.
const steps: string[] = [];
- if (!hasScope) {
+ // ── Discovery / enumeration differs for a link-seeded sweep vs a query one ──
+ if (link) {
+ steps.push(
+ `**Decode & confirm the link.** Call \`open_link\` with ` +
+ `link="${link}" (preview mode — no apply). It returns a plan: the ` +
+ `resolved source (origin, hub or single-site), the query tree, every ` +
+ `active filter, the channel scope validated against the corpus, and any ` +
+ `ignored bits (e.g. the vestigial \`fk\` tracks). **Show me the plan and ` +
+ `confirm it captures what I want.** If I ask for a change ("drop the ` +
+ `availability filter", "only channel X", "search Y instead"), re-call ` +
+ `\`open_link\` with the matching \`overrides\` until the plan is right.`,
+ );
+ steps.push(
+ `**Apply & enumerate.** Call \`open_link\` again with the confirmed ` +
+ `arguments plus apply:true — this switches the active source to the ` +
+ `link's origin and runs the search (the full query tree + filters, at ` +
+ `fidelity). Page it with a rising \`offset\` (offset += limit) until ` +
+ `\`has_more\` is no to collect the whole worklist of video ids; note the ` +
+ `\`total\`. Results already carry linked \`[mm:ss](url)\` timestamps and ` +
+ `the applied plan is echoed at the top — record the source, query, and ` +
+ `filters in the report. If coverage is PARTIAL, say so.`,
+ );
+ } else {
+ if (!hasScope) {
+ steps.push(
+ `**Choose the scope first — do NOT default to the whole corpus.** No ` +
+ `channel/channels/group was supplied. Call \`list_channels\`, present ` +
+ `the channel groups and their channels to me, and ask which group(s) ` +
+ `or channel(s) to sweep — or to confirm **all** for the whole corpus. ` +
+ `Wait for my choice before enumerating anything. Only sweep everything ` +
+ `if I explicitly choose "all". Use my choice as the ` +
+ `\`channel\`/\`channels\`/\`group\` scope in every ` +
+ `\`search_transcripts\` call below.`,
+ );
+ }
steps.push(
- `**Choose the scope first — do NOT default to the whole corpus.** No ` +
- `channel/channels/group was supplied. Call \`list_channels\`, present ` +
- `the channel groups and their channels to me, and ask which group(s) or ` +
- `channel(s) to sweep — or to confirm **all** for the whole corpus. Wait ` +
- `for my choice before enumerating anything. Only sweep everything if I ` +
- `explicitly choose "all". Use my choice as the \`channel\`/\`channels\`/` +
- `\`group\` scope in every \`search_transcripts\` call below.`,
+ `**Search.** Call \`search_transcripts\` with query "${query}"` +
+ (hasScope
+ ? ` and ${scopeArgsText}`
+ : ` and the scope I chose in step 1`) +
+ `. Curated aliases auto-expand the query — the footer reports which ` +
+ `fired (e.g. mis-transcribed spellings) and names the resolved scope ` +
+ `(and warns about any channel/group token that matched nothing — fix a ` +
+ `typo before continuing). Treat the *expanded* match set as your target ` +
+ `and mention the expansion and the scope in the report.`,
+ );
+ steps.push(
+ `**Enumerate the full worklist.** Page the complete set with ` +
+ `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` +
+ `until \`has_more\` is false — this gives you every id/title/channel/date ` +
+ `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` +
+ `(page/video cap), say so in the report — the sweep is then a sample, ` +
+ `not exhaustive.`,
);
}
steps.push(
- `**Search.** Call \`search_transcripts\` with query "${query}"` +
- (hasScope
- ? ` and ${scopeArgsText}`
- : ` and the scope I chose in step 1`) +
- `. Curated aliases auto-expand the query — the footer reports which fired ` +
- `(e.g. mis-transcribed spellings) and names the resolved scope (and warns ` +
- `about any channel/group token that matched nothing — fix a typo before ` +
- `continuing). Treat the *expanded* match set as your target and mention ` +
- `the expansion and the scope in the report.`,
- );
-
- steps.push(
- `**Enumerate the full worklist.** Page the complete set with ` +
- `\`include_snippets: false\` and a rising \`offset\` (offset += limit) ` +
- `until \`has_more\` is false — this gives you every id/title/channel/date ` +
- `cheaply. Note the \`total\`. If the footer says coverage is PARTIAL ` +
- `(page/video cap), say so in the report — the sweep is then a sample, not ` +
- `exhaustive.`,
- );
-
- steps.push(
`**Plan.** With N total matches and a batch size of ${batchSize}, that is ` +
`\`ceil(N / ${batchSize})\` batches. State the plan (N and the batch ` +
`count) before you start.`,
);
+ // ── Feature 2: dumb-extractor-per-batch map-reduce on the cheapest model —
+ // the extractor only quotes verbatim base-form excerpts, so the heavy
+ // transcript text never reaches the orchestrator (and never costs the big
+ // model). Feature 1: linked citations, expanded from moment_base + seconds.
steps.push(
- `**Per batch**, for each group of up to ${batchSize} video ids:\n` +
- ` - Call \`get_transcripts\` with those ids **and the query** so each ` +
- `transcript comes back as bounded, timestamped excerpt windows around the ` +
- `matches (alias-correct, high-signal).\n` +
- ` - **Cross-reference** the batch against the report so far. Upsert ` +
+ `**Per batch (map-reduce), for each group of up to ${batchSize} video ids:**\n` +
+ ` - **Spawn a subagent as a DUMB EXTRACTOR on the cheapest model** — ` +
+ `use the Task tool and request model "${parseModel}" (its model ` +
+ `parameter, or a "${parseModel}"-backed agent type); if no model ` +
+ `override is available, spawn it anyway on the default. Give it exactly ` +
+ `this job: call \`get_transcripts\` with the batch's ids, the query, and ` +
+ `\`link_style: "base"\`, then return ONLY the lines relevant to the ` +
+ `directive, VERBATIM — do NOT analyze, summarize, or rephrase anything. ` +
+ `Group the kept lines per video as \`### <title>\` + that video's ` +
+ `\`moment_base:\` line copied exactly (or its \`source:\` line when ` +
+ `there is no moment_base) + the kept lines in their \`[mm:ss|seconds]\` ` +
+ `form. Hard budget: at most 40 lines (~600 words) per batch — if more ` +
+ `match, keep the strongest and end with \`(+N more matching lines)\`. ` +
+ `Return nothing else — the raw transcript bulk stays inside the ` +
+ `subagent and never enters your context.\n` +
+ ` - **Merge — you (the orchestrator) do ALL the synthesis.** Cross-` +
+ `reference the returned fragment against the report so far and upsert ` +
`findings — claims, and contradictions with earlier claims — into ` +
- `well-titled \`## sections\`. Cite every source as *title + [mm:ss]*.\n` +
- ` - Keep \`${reportPath}\` the single source of truth (Write/Edit it ` +
- `each batch), then **drop the raw transcript text** once folded — don't ` +
- `carry it forward.`,
+ `well-titled \`## sections\` of \`${reportPath}\` (Write/Edit). Cite ` +
+ `every finding as **\`[title @ mm:ss](<moment url>)\`**, expanding each ` +
+ `kept \`[mm:ss|seconds]\` stamp by appending the integer after the ` +
+ `\`|\` to that video's moment_base (full link = ` +
+ `\`<moment_base><seconds>\`; no moment_base → link the \`source:\` URL ` +
+ `instead). Then discard the fragment. Batches are independent, so you ` +
+ `may dispatch several subagents in parallel.\n` +
+ ` - **Fallback:** if no subagent/Task tool is available, do the batch ` +
+ `inline — call \`get_transcripts\` with \`link_style: "base"\` yourself, ` +
+ `fold the expanded cited findings into the report, then **drop the raw ` +
+ `excerpt text** before moving on (don't carry it forward).`,
);
steps.push(
`**Finish.** Repeat to the end of the worklist, then write a short summary ` +
`section (the scope swept, how many videos covered, headline findings, any ` +
- `partial-coverage caveat) and tell me the report path.`,
+ `partial-coverage caveat) and tell me the report path. Keep every citation ` +
+ `a clickable \`[title @ mm:ss](url)\` link.`,
);
const numbered = steps
@@ -1051,16 +1668,19 @@ function buildSweepPrompt(args: Record<string, unknown>) {
.join("\n\n");
const text =
- `Run a **corpus sweep** for the query **"${query}"**${introScope}, ` +
- `extracting **${directive}**, and maintain a running markdown report at ` +
+ `Run a **corpus sweep** for ${subject}${introScope}, extracting ` +
+ `**${directive}**, and maintain a running markdown report at ` +
`\`${reportPath}\`. You are the sweep engine — work through the whole match ` +
`set methodically, using the transcript MCP tools for evidence and your own ` +
`Write/Edit tools for the report. The MCP is read-only; never try to change ` +
- `the archive.\n\n` +
+ `the archive. **Cite every finding as a clickable ` +
+ `\`[title @ mm:ss](<moment url>)\` link** (build each moment URL by ` +
+ `appending the cited integer seconds to that video's \`moment_base\` from ` +
+ `the tool output).\n\n` +
`Follow these steps:\n\n${numbered}`;
return {
- description: `Corpus sweep for "${query}" → ${reportPath}`,
+ description: `Corpus sweep for ${subject} → ${reportPath}`,
messages: [
{
role: "user" as const,
diff --git a/mcp/src/shareLink.test.ts b/mcp/src/shareLink.test.ts
@@ -0,0 +1,168 @@
+import { test } from "node:test";
+import assert from "node:assert/strict";
+import {
+ newGroup,
+ newLeaf,
+ stringifyRoot,
+} from "yt-dlp-transcript-common/lib/searchQuery";
+import {
+ decodeShareLink,
+ applyLinkOverrides,
+ renderQueryTree,
+ describeFilters,
+} from "./shareLink";
+
+const ORIGIN = "https://rekietalyzer.pages.dev";
+
+// A realistic composite qt tree: transcript:"coffee" AND NOT tags:"espresso".
+function qtParam(): string {
+ const tree = newGroup({
+ op: "AND",
+ children: [
+ newLeaf({ query: "coffee", scope: "transcripts" }),
+ newLeaf({ query: "espresso", scope: "tags", negate: true }),
+ ],
+ });
+ return encodeURIComponent(stringifyRoot(tree));
+}
+
+// A full-fidelity share link: qt tree + every filter facet + a vestigial fk.
+function fullLink(): string {
+ const p = new URLSearchParams();
+ p.set("qt", "__QT__");
+ p.set("fv", "1");
+ p.append("fc", "Rekieta Law");
+ p.append("fc", "Rekieta Law (Rumble)");
+ p.append("ft", "v"); // videos only
+ p.append("fa", "a"); // all-ages only
+ p.append("fav", "a"); // available only
+ p.append("fk", "live_chat"); // vestigial
+ p.set("fdf", "20250101");
+ p.set("fdt", "20251231");
+ // qt must not be URL-double-encoded by URLSearchParams — splice it in raw.
+ return `${ORIGIN}/?${p.toString().replace("__QT__", qtParam())}`;
+}
+
+test("decode: origin is extracted from the pasted URL", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=x`);
+ assert.equal(d.origin, ORIGIN);
+});
+
+test("decode: qt tree parses to a composite query", () => {
+ const d = decodeShareLink(fullLink());
+ assert.equal(d.querySource, "qt");
+ const rendered = renderQueryTree(d.tree);
+ assert.match(rendered, /transcript:"coffee"/);
+ assert.match(rendered, /NOT tags:"espresso"/);
+ assert.match(rendered, /AND/);
+});
+
+test("decode: every v1 filter facet is decoded", () => {
+ const d = decodeShareLink(fullLink());
+ assert.ok(d.filters);
+ const f = d.filters!;
+ // ft=v → videos kept, livestreams dropped.
+ assert.equal(f.videos, true);
+ assert.equal(f.livestreams, false);
+ // fa=a → all-ages kept, restricted dropped.
+ assert.equal(f.allAges, true);
+ assert.equal(f.restricted, false);
+ // fav=a → available kept, unlisted/deleted dropped.
+ assert.equal(f.available, true);
+ assert.equal(f.unlisted, false);
+ assert.equal(f.deleted, false);
+ // dates.
+ assert.equal(f.dateFrom, "20250101");
+ assert.equal(f.dateTo, "20251231");
+});
+
+test("decode: fc channel names are captured; fk is decoded but flagged ignored", () => {
+ const d = decodeShareLink(fullLink());
+ assert.deepEqual(d.channelNames.sort(), [
+ "Rekieta Law",
+ "Rekieta Law (Rumble)",
+ ]);
+ assert.equal(d.hasChannelFilter, true);
+ assert.deepEqual(d.ignoredTracks, ["live_chat"]);
+ assert.ok(d.warnings.some((w) => /fk/.test(w) && /ignored/.test(w)));
+});
+
+test("decode: legacy q/re → a single regex transcripts leaf, no filters", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=coffee&re=1`);
+ assert.equal(d.querySource, "legacy");
+ assert.equal(d.filters, null);
+ assert.equal(d.hasChannelFilter, false);
+ const leaf = d.tree.children[0];
+ assert.equal(leaf.kind, "leaf");
+ assert.match(renderQueryTree(d.tree), /transcript:\/coffee\//);
+});
+
+test("decode: legacy m=subs → a live-chat leaf", () => {
+ const d = decodeShareLink(`${ORIGIN}/?q=hello&m=subs`);
+ assert.match(renderQueryTree(d.tree), /live-chat:"hello"/);
+});
+
+test("decode: no fv block → unfiltered, whole-corpus scope", () => {
+ const d = decodeShareLink(`${ORIGIN}/?fc=Some%20Channel`);
+ assert.equal(d.filters, null);
+ assert.equal(d.hasChannelFilter, false);
+ assert.deepEqual(d.channelNames, []);
+});
+
+test("decode: malformed qt falls back to legacy/empty with a warning", () => {
+ const d = decodeShareLink(`${ORIGIN}/?qt=not-json&q=fallback`);
+ assert.equal(d.querySource, "legacy");
+ assert.ok(d.warnings.some((w) => /malformed/.test(w)));
+ assert.match(renderQueryTree(d.tree), /transcript:"fallback"/);
+});
+
+// ─── overrides ───
+
+test("override: clearAvailability resets the fav facet to keep-all", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearAvailability: true,
+ });
+ assert.equal(d.filters!.available, true);
+ assert.equal(d.filters!.unlisted, true);
+ assert.equal(d.filters!.deleted, true);
+ assert.ok(d.warnings.some((w) => /availability filter removed/.test(w)));
+});
+
+test("override: clearFilters drops the whole filter block", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearFilters: true,
+ });
+ assert.equal(d.filters, null);
+});
+
+test("override: channels replaces the channel scope", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ channels: ["Only This"],
+ });
+ assert.deepEqual(d.channelNames, ["Only This"]);
+ assert.equal(d.hasChannelFilter, true);
+});
+
+test("override: clearChannels widens to the whole corpus", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ clearChannels: true,
+ });
+ assert.deepEqual(d.channelNames, []);
+ assert.equal(d.hasChannelFilter, false);
+});
+
+test("override: query replaces the tree with a single leaf", () => {
+ const d = applyLinkOverrides(decodeShareLink(fullLink()), {
+ query: "different term",
+ });
+ assert.match(renderQueryTree(d.tree), /transcript:"different term"/);
+});
+
+test("describeFilters: names the constrained facets only", () => {
+ const d = decodeShareLink(fullLink());
+ const lines = describeFilters(d.filters);
+ assert.ok(lines.some((l) => /videos only/.test(l)));
+ assert.ok(lines.some((l) => /all-ages only/.test(l)));
+ assert.ok(lines.some((l) => /available.*kept/.test(l)));
+ assert.ok(lines.some((l) => /uploaded/.test(l)));
+});
diff --git a/mcp/src/shareLink.ts b/mcp/src/shareLink.ts
@@ -0,0 +1,316 @@
+// Decode an archilyzer viewer **share link** into a search spec + source target.
+//
+// A share URL is `origin` + a `qt=` composite query tree (or a legacy `q`/`re`)
+// + the v1 filter params (`fc/ft/fa/fav/fk/fdf/fdt`, gated by `fv=1`). This
+// module turns that URL into the pieces the MCP needs to reproduce the search at
+// full fidelity:
+// • the origin (→ probed by the server for hub vs single-site),
+// • the query tree (parsed via the browser's own `parseRoot`),
+// • the decoded filters (via the browser's own `parseShareV1`),
+// • the requested channel NAMES (`fc`, resolved against the live corpus later),
+// • and any ignored/warned pieces (`fk` tracks are vestigial in the composite
+// share model — decoded and reported as ignored, not enforced).
+//
+// Pure: no network. The server does the origin probe and channel resolution.
+
+import {
+ parseRoot,
+ rootFromLegacy,
+ emptyRoot,
+ isLeaf,
+ isGroup,
+ isNodeActive,
+ newGroup,
+ newLeaf,
+ type GroupNode,
+ type QueryNode,
+} from "yt-dlp-transcript-common/lib/searchQuery";
+import {
+ hasShareV1,
+ parseShareV1,
+} from "yt-dlp-transcript-common/components/shareUrl";
+import type { SearchFilters } from "./search";
+
+export type DecodedLink = {
+ // The pasted URL's origin (scheme + host[:port]). The source target.
+ origin: string;
+ // The composite query tree — from `qt=`, a synthesized legacy leaf, or an
+ // empty root (filter-only browse).
+ tree: GroupNode;
+ querySource: "qt" | "legacy" | "empty";
+ // Positive share filters (ft/fa/fav + dates), or null when the link carries no
+ // v1 filter block (`fv=1`) — i.e. an unfiltered search.
+ filters: SearchFilters | null;
+ // Channel NAMES the link selects (`fc`). Resolved against the live corpus by
+ // the caller. Empty with `filters === null` means "whole corpus".
+ channelNames: string[];
+ // Whether the link constrained channels at all (had a v1 filter block). When
+ // false, channel scope is the whole corpus regardless of `channelNames`.
+ hasChannelFilter: boolean;
+ // Vestigial `fk` subtitle-track tokens — decoded but a no-op.
+ ignoredTracks: string[];
+ // Human-facing notes (malformed qt, ignored fk, …).
+ warnings: string[];
+};
+
+// Decode a share link. Never throws for a malformed query tree (falls back to a
+// legacy/empty tree with a warning); throws only if `link` isn't a URL at all.
+export function decodeShareLink(link: string): DecodedLink {
+ const url = new URL(link);
+ const search = url.search;
+ const params = new URLSearchParams(search);
+ const warnings: string[] = [];
+
+ // ── Query: qt= tree preferred, else legacy q/re, else empty (filter-only).
+ let tree: GroupNode = emptyRoot();
+ let querySource: DecodedLink["querySource"] = "empty";
+ const qt = params.get("qt");
+ if (qt) {
+ const parsed = parseRoot(qt);
+ if (parsed) {
+ tree = parsed;
+ querySource = "qt";
+ } else {
+ warnings.push("malformed `qt` query tree — ignored");
+ }
+ }
+ if (querySource === "empty") {
+ const q = params.get("q") ?? "";
+ const mode = params.get("m") === "subs" ? "subs" : "transcripts";
+ const re = params.get("re") === "1";
+ if (q.trim() !== "") {
+ tree = rootFromLegacy(q, mode, re);
+ querySource = "legacy";
+ }
+ }
+
+ // ── Filters: only meaningful when the v1 block is present (`fv=1`).
+ const v1 = hasShareV1(search);
+ // Pass the requested fc names AS the channel universe so none are dropped —
+ // real validation against the live corpus happens at apply time.
+ const requestedChannels = params.getAll("fc");
+ const sel = parseShareV1(search, requestedChannels);
+ const ignoredTracks = [...sel.tracks];
+ if (ignoredTracks.length > 0) {
+ warnings.push(
+ `\`fk\` subtitle-track filter is vestigial in the composite share model — ` +
+ `${ignoredTracks.length} track token(s) decoded but ignored: ${ignoredTracks.join(", ")}`,
+ );
+ }
+
+ const filters: SearchFilters | null = v1
+ ? {
+ videos: sel.videos,
+ livestreams: sel.livestreams,
+ allAges: sel.allAges,
+ restricted: sel.restricted,
+ available: sel.available,
+ unlisted: sel.unlisted,
+ deleted: sel.deleted,
+ ...(sel.dateFrom ? { dateFrom: sel.dateFrom } : {}),
+ ...(sel.dateTo ? { dateTo: sel.dateTo } : {}),
+ }
+ : null;
+
+ return {
+ origin: url.origin,
+ tree,
+ querySource,
+ filters,
+ channelNames: v1 ? [...sel.selectedChannels] : [],
+ hasChannelFilter: v1,
+ ignoredTracks,
+ warnings,
+ };
+}
+
+// Structured adjustments an agent can pass to honor a natural-language edit
+// ("remove the availability filter", "only channel X", "search 'foo' instead").
+// Every field is optional; only the provided facets are changed.
+export type LinkOverrides = {
+ // Drop every decoded filter (ft/fa/fav/dates) → unfiltered.
+ clearFilters?: boolean;
+ // Reset a single filter facet to its keep-everything default.
+ clearAvailability?: boolean; // fav
+ clearType?: boolean; // ft
+ clearAge?: boolean; // fa
+ clearDates?: boolean; // fdf/fdt
+ // Widen the channel scope to the whole corpus (ignore fc).
+ clearChannels?: boolean;
+ // Replace the channel scope with these names (validated against the corpus).
+ channels?: string[];
+ // Replace/clear the upload-date bounds ("YYYYMMDD"; null clears that bound).
+ dateFrom?: string | null;
+ dateTo?: string | null;
+ // Replace the whole query with a single leaf (query + optional regex/scope).
+ query?: string;
+ regex?: boolean;
+ queryScope?: "transcripts" | "chat" | "metadata" | "description" | "tags";
+};
+
+const KEEP_ALL_FILTERS: SearchFilters = {
+ videos: true,
+ livestreams: true,
+ allAges: true,
+ restricted: true,
+ available: true,
+ unlisted: true,
+ deleted: true,
+};
+
+// Apply overrides to a decoded link, returning a new DecodedLink. Records what
+// changed as warnings so the plan can echo it.
+export function applyLinkOverrides(
+ decoded: DecodedLink,
+ overrides: LinkOverrides | undefined,
+): DecodedLink {
+ if (!overrides) return decoded;
+ const next: DecodedLink = {
+ ...decoded,
+ warnings: [...decoded.warnings],
+ channelNames: [...decoded.channelNames],
+ filters: decoded.filters ? { ...decoded.filters } : null,
+ };
+ const note = (s: string): void => {
+ next.warnings.push(`override: ${s}`);
+ };
+
+ // ── Query replacement.
+ if (typeof overrides.query === "string" && overrides.query.trim() !== "") {
+ next.tree = newGroup({
+ children: [
+ newLeaf({
+ query: overrides.query,
+ scope: overrides.queryScope ?? "transcripts",
+ useRegex: overrides.regex === true,
+ }),
+ ],
+ });
+ next.querySource = "qt";
+ note(`query replaced with "${overrides.query}"`);
+ }
+
+ // ── Channel scope.
+ if (overrides.clearChannels) {
+ next.channelNames = [];
+ next.hasChannelFilter = false;
+ note("channel scope widened to the whole corpus");
+ }
+ if (Array.isArray(overrides.channels)) {
+ next.channelNames = overrides.channels.filter((c) => c.trim() !== "");
+ next.hasChannelFilter = true;
+ note(`channel scope set to [${next.channelNames.join(", ")}]`);
+ }
+
+ // ── Filters.
+ if (overrides.clearFilters) {
+ next.filters = null;
+ note("all filters removed");
+ } else if (next.filters) {
+ const f = next.filters;
+ if (overrides.clearAvailability) {
+ f.available = true;
+ f.unlisted = true;
+ f.deleted = true;
+ note("availability filter removed");
+ }
+ if (overrides.clearType) {
+ f.videos = true;
+ f.livestreams = true;
+ note("type filter removed");
+ }
+ if (overrides.clearAge) {
+ f.allAges = true;
+ f.restricted = true;
+ note("age filter removed");
+ }
+ if (overrides.clearDates) {
+ delete f.dateFrom;
+ delete f.dateTo;
+ note("date filter removed");
+ }
+ if (overrides.dateFrom !== undefined) {
+ if (overrides.dateFrom === null) delete f.dateFrom;
+ else f.dateFrom = overrides.dateFrom;
+ note(`dateFrom set to ${overrides.dateFrom ?? "(none)"}`);
+ }
+ if (overrides.dateTo !== undefined) {
+ if (overrides.dateTo === null) delete f.dateTo;
+ else f.dateTo = overrides.dateTo;
+ note(`dateTo set to ${overrides.dateTo ?? "(none)"}`);
+ }
+ } else if (
+ overrides.dateFrom !== undefined ||
+ overrides.dateTo !== undefined
+ ) {
+ // Setting a date on an otherwise-unfiltered link creates a filter block.
+ next.filters = { ...KEEP_ALL_FILTERS };
+ if (typeof overrides.dateFrom === "string") next.filters.dateFrom = overrides.dateFrom;
+ if (typeof overrides.dateTo === "string") next.filters.dateTo = overrides.dateTo;
+ note("date filter added");
+ }
+
+ return next;
+}
+
+// ── Human-readable renderings for the preview plan ──
+
+const SCOPE_LABEL: Record<string, string> = {
+ transcripts: "transcript",
+ chat: "live-chat",
+ metadata: "title/channel",
+ description: "description",
+ tags: "tags",
+};
+
+// Render a query tree readably, e.g.
+// transcript:"coffee" AND tags:"espresso" AND NOT chat:"spam"
+export function renderQueryTree(node: QueryNode): string {
+ if (isLeaf(node)) {
+ const label = SCOPE_LABEL[node.scope] ?? node.scope;
+ const kind = node.useRegex ? "/" : '"';
+ const end = node.useRegex ? "/" : '"';
+ const body = `${label}:${kind}${node.query}${end}`;
+ return node.negate ? `NOT ${body}` : body;
+ }
+ if (!isGroup(node)) return "";
+ const parts = node.children
+ .filter(isNodeActive)
+ .map((c) => {
+ const s = renderQueryTree(c);
+ return isGroup(c) ? `(${s})` : s;
+ })
+ .filter((s) => s !== "");
+ if (parts.length === 0) return "(everything in scope)";
+ const joined = parts.join(` ${node.op} `);
+ return node.negate ? `NOT (${joined})` : joined;
+}
+
+// A compact list of the active filters for the plan, or [] when unfiltered.
+export function describeFilters(f: SearchFilters | null): string[] {
+ if (!f) return [];
+ const out: string[] = [];
+ // Type (ft): only worth noting when it excludes one side.
+ if (f.videos !== f.livestreams) {
+ out.push(`type: ${f.videos ? "videos only" : "livestreams only"}`);
+ } else if (!f.videos && !f.livestreams) {
+ out.push("type: none kept (videos and livestreams both excluded)");
+ }
+ // Audience (fa).
+ if (f.allAges !== f.restricted) {
+ out.push(`audience: ${f.allAges ? "all-ages only" : "age-restricted only"}`);
+ } else if (!f.allAges && !f.restricted) {
+ out.push("audience: none kept");
+ }
+ // Availability (fav).
+ const av: string[] = [];
+ if (f.available) av.push("available");
+ if (f.unlisted) av.push("unlisted");
+ if (f.deleted) av.push("deleted");
+ if (av.length < 3) out.push(`availability: ${av.length ? av.join(" + ") : "none"} kept`);
+ // Dates (fdf/fdt).
+ if (f.dateFrom || f.dateTo) {
+ out.push(`uploaded: ${f.dateFrom ?? "…"} → ${f.dateTo ?? "…"}`);
+ }
+ return out;
+}
diff --git a/mcp/src/source.ts b/mcp/src/source.ts
@@ -2,9 +2,17 @@ import { readFile, readdir } from "node:fs/promises";
import path from "node:path";
import {
transcriptPageFileName,
+ subsPageFileName,
+ pageFileName,
type ChannelTranscriptsManifest,
+ type ChannelSubsManifest,
+ type Manifest,
} from "yt-dlp-transcript-common/lib/manifest";
-import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
+import type {
+ TranscriptDetail,
+ DisplaySummary,
+} from "yt-dlp-transcript-common/lib/transcripts";
+import type { SubsDetail } from "yt-dlp-transcript-common/lib/subs";
import {
coerceAliasConfig,
type SearchAlias,
@@ -48,6 +56,48 @@ const EMPTY_GROUPS: ChannelGroups = {
defaultGroupId: DEFAULT_GROUP_FALLBACK_ID,
};
+// Per-video availability, joined in from the summaries shards (a
+// TranscriptDetail record does not carry deleted/unlisted flags). Keyed by the
+// member-local video slug (`<channelSlug>/<id>`) — the same slug a transcript
+// page record carries — so the search engine can apply the `fav` filter.
+export type VideoAvailability = { isDeleted: boolean; isUnlisted: boolean };
+
+// Read a site's global summaries shards (summaries/manifest.json +
+// summaries/page-NNNN.json) via `readPage` and fold them into a slug →
+// availability map. Tolerant: an absent/malformed manifest yields an empty map,
+// and a page that fails to read is skipped. `readManifest`/`readPage` throw or
+// return null on absence per the source's transport.
+async function buildAvailabilityMap(
+ readManifest: () => Promise<Manifest | null>,
+ readPage: (page: number) => Promise<DisplaySummary[] | null>,
+): Promise<Map<string, VideoAvailability>> {
+ const map = new Map<string, VideoAvailability>();
+ let manifest: Manifest | null;
+ try {
+ manifest = await readManifest();
+ } catch {
+ return map;
+ }
+ if (!manifest || typeof manifest.pageCount !== "number") return map;
+ for (let page = 0; page < manifest.pageCount; page++) {
+ let records: DisplaySummary[] | null;
+ try {
+ records = await readPage(page);
+ } catch {
+ continue;
+ }
+ if (!records) continue;
+ for (const r of records) {
+ if (typeof r.slug !== "string") continue;
+ map.set(r.slug, {
+ isDeleted: r.isDeleted === true,
+ isUnlisted: r.isUnlisted === true,
+ });
+ }
+ }
+ return map;
+}
+
// Parse a summaries/manifest.json blob into channel-group defs, tolerating any
// missing/malformed shape (→ empty fallback).
function parseGroupsManifest(raw: unknown): ChannelGroups {
@@ -76,6 +126,25 @@ export interface ShardSource {
// empty fallback when absent/malformed, or in hub mode (federated per-site
// groups are a different model — deferred). Cached per source.
loadGroups(): Promise<ChannelGroups>;
+ // The public origin of the viewer that owns this source's videos, or null
+ // when there isn't one (a local dir on disk). Used to build archilyzer viewer
+ // deep links for cited moments (momentUrl). A single-site remote returns its
+ // base URL; a hub returns null because each video's origin is its member
+ // site's url (carried on the ChannelRef as `siteUrl`) — prefer that per-video.
+ publicOrigin(): string | null;
+ // A channel's live-chat/subs manifest (subs/<slug>/manifest.json), or null
+ // when the channel ships no subs shards. Same slugToPage/pageCount shape as
+ // the transcripts manifest. Fetched lazily — only the chat search scope needs
+ // it.
+ subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null>;
+ // A page of a channel's subs records (subs/<slug>/page-NNNN.json). Each record
+ // inlines its per-track cues under `tracks` (e.g. `tracks.live_chat`).
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]>;
+ // A map of every video's availability (deleted/unlisted), keyed by the
+ // member-local video slug (`<channelSlug>/<id>`), built from the summaries
+ // shards. Fetched lazily and cached — only the `fav` availability filter needs
+ // it. Empty when the source ships no summaries.
+ availabilityMap(): Promise<Map<string, VideoAvailability>>;
}
// Shape of the channels we read out of a site corpus.json (Layer 1). Kept loose
@@ -104,10 +173,58 @@ export class LocalSource implements ShardSource {
readonly label: string;
private aliases?: SearchAlias[];
private groups?: ChannelGroups;
+ private availability?: Map<string, VideoAvailability>;
constructor(private dir: string) {
this.label = `local:${dir}`;
}
+ // A local dir has no public viewer origin — cited moments fall back to
+ // platform links (momentUrl).
+ publicOrigin(): string | null {
+ return null;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ try {
+ const raw = await readFile(
+ path.join(this.dir, "subs", ch.slug, "manifest.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as ChannelSubsManifest;
+ } catch {
+ return null; // channel ships no subs shards
+ }
+ }
+
+ async subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ const raw = await readFile(
+ path.join(this.dir, "subs", ch.slug, subsPageFileName(page)),
+ "utf8",
+ );
+ return JSON.parse(raw) as SubsDetail[];
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ this.availability = await buildAvailabilityMap(
+ async () => {
+ const raw = await readFile(
+ path.join(this.dir, "summaries", "manifest.json"),
+ "utf8",
+ );
+ return JSON.parse(raw) as Manifest;
+ },
+ async (page) => {
+ const raw = await readFile(
+ path.join(this.dir, "summaries", pageFileName(page)),
+ "utf8",
+ );
+ return JSON.parse(raw) as DisplaySummary[];
+ },
+ );
+ return this.availability;
+ }
+
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
@@ -195,11 +312,45 @@ export class RemoteSource implements ShardSource {
private base: string;
private aliases?: SearchAlias[];
private groups?: ChannelGroups;
+ private availability?: Map<string, VideoAvailability>;
constructor(baseUrl: string) {
this.base = baseUrl.replace(/\/+$/, "");
this.label = `remote:${this.base}`;
}
+ // The deployed site origin — the archilyzer viewer that owns these videos.
+ publicOrigin(): string | null {
+ return this.base;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ try {
+ const res = await fetch(`${this.base}/subs/${ch.slug}/manifest.json`);
+ return res.ok ? ((await res.json()) as ChannelSubsManifest) : null;
+ } catch {
+ return null;
+ }
+ }
+
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ return this.getJson(`/subs/${ch.slug}/${subsPageFileName(page)}`);
+ }
+
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ this.availability = await buildAvailabilityMap(
+ async () => {
+ const res = await fetch(`${this.base}/summaries/manifest.json`);
+ return res.ok ? ((await res.json()) as Manifest) : null;
+ },
+ async (page) => {
+ const res = await fetch(`${this.base}/summaries/${pageFileName(page)}`);
+ return res.ok ? ((await res.json()) as DisplaySummary[]) : null;
+ },
+ );
+ return this.availability;
+ }
+
async loadAliases(): Promise<SearchAlias[]> {
if (this.aliases) return this.aliases;
try {
@@ -263,6 +414,7 @@ export class HubSource implements ShardSource {
readonly hubBase: string;
private members = new Map<string, RemoteSource>(); // siteId -> source
private aliases?: SearchAlias[];
+ private availability?: Map<string, VideoAvailability>;
// Optional subset allowlist of member siteIds. Undefined = federate every
// member; a set restricts listChannels() to those members (site discovery via
// listSites() stays unfiltered so a picker can still see all members).
@@ -359,4 +511,55 @@ export class HubSource implements ShardSource {
if (!ch.siteId) throw new Error("hub channel ref missing siteId");
return this.memberFor(ch.siteId).transcriptPage(ch, page);
}
+
+ // A hub has no single viewer origin — each video's origin is its member
+ // site's url (carried on the ChannelRef as `siteUrl`), which momentUrl prefers
+ // per-video. Return null so we never mint a wrong-origin viewer link.
+ publicOrigin(): string | null {
+ return null;
+ }
+
+ async subsManifest(ch: ChannelRef): Promise<ChannelSubsManifest | null> {
+ if (!ch.siteId) return null;
+ try {
+ return await this.memberFor(ch.siteId).subsManifest(ch);
+ } catch {
+ return null; // member not yet registered / unreachable
+ }
+ }
+
+ subsPage(ch: ChannelRef, page: number): Promise<SubsDetail[]> {
+ if (!ch.siteId) throw new Error("hub channel ref missing siteId");
+ return this.memberFor(ch.siteId).subsPage(ch, page);
+ }
+
+ // Merge each member's availability map. Keys are member-local slugs
+ // (`<channelSlug>/<id>`) — the same slug a member's transcript page records
+ // carry — so a per-record `fav` lookup joins correctly. Built lazily/cached.
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ if (this.availability) return this.availability;
+ const merged = new Map<string, VideoAvailability>();
+ let sites: HubSite[];
+ try {
+ sites = (await this.listSites()).filter(
+ (s) => !this.allowSiteIds || this.allowSiteIds.has(s.siteId),
+ );
+ } catch {
+ this.availability = merged;
+ return merged;
+ }
+ for (const site of sites) {
+ const remote = this.members.get(site.siteId) ?? new RemoteSource(site.url);
+ this.members.set(site.siteId, remote);
+ try {
+ for (const [slug, avail] of await remote.availabilityMap()) {
+ merged.set(slug, avail);
+ }
+ } catch {
+ // skip an unreachable member
+ }
+ }
+ this.availability = merged;
+ return merged;
+ }
}
diff --git a/mcp/src/sourceController.test.ts b/mcp/src/sourceController.test.ts
@@ -8,7 +8,13 @@ import { InMemoryTransport } from "@modelcontextprotocol/sdk/inMemory.js";
import type { ChannelTranscriptsManifest } from "yt-dlp-transcript-common/lib/manifest";
import type { TranscriptDetail } from "yt-dlp-transcript-common/lib/transcripts";
import type { SearchAlias } from "yt-dlp-transcript-common/lib/searchAliases";
-import type { ChannelGroups, ChannelRef, HubSite, ShardSource } from "./source";
+import type {
+ ChannelGroups,
+ ChannelRef,
+ HubSite,
+ ShardSource,
+ VideoAvailability,
+} from "./source";
import type { SourceSpec } from "./sources";
import { SourceController } from "./sourceController";
import { createServer } from "./server";
@@ -56,6 +62,18 @@ class FakeSource implements ShardSource {
async transcriptPage(): Promise<TranscriptDetail[]> {
return [];
}
+ publicOrigin(): string | null {
+ return null;
+ }
+ async subsManifest(): Promise<null> {
+ return null;
+ }
+ async subsPage(): Promise<[]> {
+ return [];
+ }
+ async availabilityMap(): Promise<Map<string, VideoAvailability>> {
+ return new Map();
+ }
}
function labelFor(spec: SourceSpec): string {